From eb77f02b16aeca9533102070de34b1f8812f4fa2 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Wed, 30 Sep 2026 00:18:45 -0700 Subject: [PATCH 01/33] ci: rebalance portable test groups and enforce a packing budget (#6192) * fix: rebalance portable CI from current duration measurements * no-mistakes(test): Test serial packing boundary and verify endpoint timeout cleanup * no-mistakes(document): Clarify timeout guidance and remove duplicated packing estimates --- bin/fm-test-run.sh | 521 +++++++++++++++++--------------- docs/fm-test-portable-shards.md | 62 ++-- tests/fm-session-start.test.sh | 11 +- tests/fm-test-run.test.sh | 56 +++- 4 files changed, 377 insertions(+), 273 deletions(-) diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 34b8232170e..dfff544fbdf 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -71,9 +71,9 @@ # --per-script-timeout-secs N # terminate a script that runs longer than N seconds and # record it as exit 124 (0 disables, the default). The -# --changed applies 1500s automatically: no measured script -# approaches it, so it only converts a HUNG -# script into a bounded failure. --max-wall-ms is checked +# --changed applies 1500s automatically, above the current +# slowest CI hint with margin; exceeding it becomes a bounded +# failure, not proof of a hang. --max-wall-ms is checked # after the run and so cannot catch a hang on its own. # External interruption cleanup is outside this runner's # guarantee; configured per-script bounds remain authoritative. @@ -135,6 +135,10 @@ # split it across separate runners, so two of its stateful scripts still never # share a machine. This script owns : a lane whose disagrees with the # configured shard count is refused, so a CI matrix cannot silently drop a shard. +# --check-coverage also reports serial_max_ms (largest packed hint sum, including +# default weights) and serial_budget_ms (the 20-minute packing target), refusing +# a split above that target. Neither figure is an execution timeout or proof of +# observed headroom: refresh growing files from CI measurements. # --changed is conservative: it over-selects related families rather than # under-selecting, and never expands to the complete suite unless --all. The one # place it is deliberately narrow is a bin/ path with no curated family: a test @@ -183,13 +187,11 @@ MAX_WALL_MS= PER_SCRIPT_TIMEOUT_SECS=0 # Bound applied automatically on the automatic --changed path, derived from # measured healthy runtimes with margin rather than picked: the slowest measured -# script is tests/fm-watch-triage.test.sh in the watcher-wake-lock family, at -# about 434s alone and about 698s under CI load (the hint table below records -# that loaded figure), and the slowest script in a runner-file changed selection -# is tests/fm-calm-pi-extension.test.sh at 77s once its Chrome reap terminates. -# 1500s keeps every measured script under the bound with roughly 2.1x headroom -# over the slowest loaded measurement, and it stays under the 30-minute normal -# CI tier so a wedged script fails here, with its output, before the job cap +# script is tests/fm-watch-triage.test.sh at about 1075s under CI load (the hint +# table below records that loaded figure). 1500s keeps every measured script +# under the bound with roughly 1.4x headroom over the slowest loaded measurement, +# and it stays under the 30-minute normal CI tier so a wedged script fails here, +# with its output, before the job cap # cancels the lane. It is a guard, not a speed control: a HUNG script becomes a # bounded failure instead of an unbounded suite, which is the shape that # silently outruns a caller's invocation budget. @@ -199,10 +201,14 @@ CHANGED_DEFAULT_TIMEOUT_SECS=1500 # One owner: CI lane names carry this count and are refused when they disagree. PORTABLE_SERIAL_SHARDS=9 -# Balance hint for a portable-serial script with no measured duration, close to -# the measured per-script mean so a newly added test neither starves nor -# overloads the shard it lands in. -PORTABLE_SERIAL_DEFAULT_WEIGHT_MS=27000 +# Conservative balance hint for a portable-serial script with no measurement. +# Rounded above the current CI mean, including the capability-skipped scripts. +PORTABLE_SERIAL_DEFAULT_WEIGHT_MS=45000 + +# Packing target, not an execution timeout: leave at least ten minutes of the +# normal CI tier for setup and runtime variance. --check-coverage refuses a +# modeled serial shard above this target; refresh hints or rebalance instead. +PORTABLE_SERIAL_MAX_WEIGHT_MS=1200000 # Largest share of the serial lane allowed to run on the default weight above. # Hints are what keep the shards balanced, so once too much of the lane is @@ -513,30 +519,30 @@ EOF # refresh procedure are owned by docs/fm-test-portable-shards.md. portable_parallel_weight_hints() { cat <<'EOF' -tests/fm-arm-pretool-check.test.sh 30898 -tests/fm-backend-herdr.test.sh 22144 -tests/fm-brief.test.sh 1625 -tests/fm-captain-hold-lifecycle.test.sh 296481 -tests/fm-cd-pretool-check.test.sh 16964 -tests/fm-composer-ghost.test.sh 2120 -tests/fm-composer-lib.test.sh 4798 -tests/fm-crew-state.test.sh 11557 -tests/fm-ensure-agents-md.test.sh 901 -tests/fm-grok-harness.test.sh 6563 -tests/fm-herdr-lab.test.sh 9800 -tests/fm-lint.test.sh 164262 -tests/fm-pi-primary-types.test.sh 8624 -tests/fm-pr-merge.test.sh 111145 -tests/fm-review-diff.test.sh 2747 -tests/fm-send-popup-settle.test.sh 4939 -tests/fm-send-settle.test.sh 2051 -tests/fm-send-strict.test.sh 3861 -tests/fm-spawn-batch.test.sh 2265 -tests/fm-supervision-instructions.test.sh 297 -tests/fm-test-run.test.sh 92944 -tests/fm-tmux-submit-busy.test.sh 2477 -tests/fm-transition-lib.test.sh 99 -tests/fm-x-mode.test.sh 31870 +tests/fm-arm-pretool-check.test.sh 33778 +tests/fm-backend-herdr.test.sh 36331 +tests/fm-brief.test.sh 10594 +tests/fm-captain-hold-lifecycle.test.sh 343658 +tests/fm-cd-pretool-check.test.sh 16801 +tests/fm-composer-ghost.test.sh 2292 +tests/fm-composer-lib.test.sh 9521 +tests/fm-crew-state.test.sh 82058 +tests/fm-ensure-agents-md.test.sh 895 +tests/fm-grok-harness.test.sh 7666 +tests/fm-herdr-lab.test.sh 18325 +tests/fm-lint.test.sh 252498 +tests/fm-pi-primary-types.test.sh 5426 +tests/fm-pr-merge.test.sh 300199 +tests/fm-review-diff.test.sh 4134 +tests/fm-send-popup-settle.test.sh 6624 +tests/fm-send-settle.test.sh 2310 +tests/fm-send-strict.test.sh 4804 +tests/fm-spawn-batch.test.sh 2987 +tests/fm-supervision-instructions.test.sh 809 +tests/fm-test-run.test.sh 156781 +tests/fm-tmux-submit-busy.test.sh 2600 +tests/fm-transition-lib.test.sh 101 +tests/fm-x-mode.test.sh 29896 EOF } @@ -559,17 +565,19 @@ portable_parallel_lane_weight() { # workflow step moved with it. list_portable_parallel_1() { cat <<'EOF' -tests/fm-lint.test.sh tests/fm-pr-merge.test.sh -tests/fm-test-run.test.sh +tests/fm-lint.test.sh +tests/fm-backend-herdr.test.sh +tests/fm-x-mode.test.sh tests/fm-cd-pretool-check.test.sh -tests/fm-pi-primary-types.test.sh -tests/fm-grok-harness.test.sh tests/fm-composer-lib.test.sh +tests/fm-send-popup-settle.test.sh +tests/fm-pi-primary-types.test.sh tests/fm-review-diff.test.sh -tests/fm-tmux-submit-busy.test.sh -tests/fm-composer-ghost.test.sh -tests/fm-brief.test.sh +tests/fm-send-settle.test.sh +tests/fm-ensure-agents-md.test.sh +tests/fm-supervision-instructions.test.sh +tests/fm-transition-lib.test.sh EOF } @@ -577,18 +585,16 @@ EOF list_portable_parallel_2() { cat <<'EOF' tests/fm-captain-hold-lifecycle.test.sh -tests/fm-x-mode.test.sh -tests/fm-arm-pretool-check.test.sh -tests/fm-backend-herdr.test.sh +tests/fm-test-run.test.sh tests/fm-crew-state.test.sh +tests/fm-arm-pretool-check.test.sh tests/fm-herdr-lab.test.sh -tests/fm-send-popup-settle.test.sh +tests/fm-brief.test.sh +tests/fm-grok-harness.test.sh tests/fm-send-strict.test.sh tests/fm-spawn-batch.test.sh -tests/fm-send-settle.test.sh -tests/fm-ensure-agents-md.test.sh -tests/fm-supervision-instructions.test.sh -tests/fm-transition-lib.test.sh +tests/fm-tmux-submit-busy.test.sh +tests/fm-composer-ghost.test.sh EOF } @@ -672,193 +678,214 @@ list_portable_serial() { # Measured portable-serial script durations in milliseconds, from the CI timing # artifacts recorded in docs/fm-test-portable-shards.md. Each value is the -# slowest successful sample in the referenced complete/partial CI runs, rather -# than only on the fastest one measured. These are balance hints only: the shard +# slowest successful sample in the referenced complete/partial CI runs, with +# the version-specific host and native-Windows exceptions documented there. +# These are balance hints only: the shard # partition stays complete and disjoint whatever they say, so a stale hint costs # balance rather than coverage. That doc owns the refresh procedure. portable_serial_weight_hints() { cat <<'EOF' -tests/fm-afk-contract.test.sh 15645 -tests/fm-afk-inject-e2e.test.sh 35889 -tests/fm-afk-pi-herdr-return-e2e.test.sh 45 -tests/fm-afk-return.test.sh 20385 -tests/fm-agy-harness.test.sh 47933 -tests/fm-agy-signals-live-e2e.test.sh 49 -tests/fm-ask-user-authority.test.sh 131 -tests/fm-backend-cmux-smoke.test.sh 33 -tests/fm-backend-cmux.test.sh 3498 -tests/fm-backend-orca.test.sh 23381 -tests/fm-backend-tmux-smoke.test.sh 363 -tests/fm-backend-zellij-smoke.test.sh 21 -tests/fm-backend-zellij.test.sh 9064 -tests/fm-backend.test.sh 21658 -tests/fm-backlog-atomicity.test.sh 196948 -tests/fm-backlog-handoff.test.sh 51990 -tests/fm-backlog-read-bound.test.sh 24288 -tests/fm-bearings-board-lavish-live-e2e.test.sh 48 -tests/fm-bearings-board-render.test.sh 12591 -tests/fm-bearings-board.test.sh 36490 -tests/fm-bearings-snapshot.test.sh 171176 -tests/fm-bootstrap-network-parallel.test.sh 9539 -tests/fm-bootstrap.test.sh 46634 -tests/fm-branch-supervision.test.sh 8915 -tests/fm-busy-adapter-wiring.test.sh 27817 -tests/fm-busy-state.test.sh 2990 -tests/fm-calm-claude-mod-live-e2e.test.sh 46 -tests/fm-calm-claude-mod-plugin.test.sh 172 -tests/fm-calm-claude-mod.test.sh 1252 -tests/fm-calm-pi-extension.test.sh 45128 -tests/fm-check-unregister.test.sh 464 -tests/fm-ci-workflow.test.sh 2073 -tests/fm-classify-corr-token.test.sh 49294 -tests/fm-classify-decision-key.test.sh 3336 -tests/fm-claude-stop-autoarm-live-e2e.test.sh 45 -tests/fm-claude-stop-autoarm.test.sh 60797 -tests/fm-claude-trust.test.sh 10410 -tests/fm-cmux-claude-composer-live-e2e.test.sh 47 -tests/fm-codex-continuity-live-e2e.test.sh 71 -tests/fm-codex-hook-layer-live-e2e.test.sh 47 -tests/fm-composer-codex-idle-live-e2e.test.sh 229 -tests/fm-composer-matrix-live-e2e.test.sh 47 -tests/fm-contributions.test.sh 35676 -tests/fm-control-relaunch.test.sh 137013 -tests/fm-control.test.sh 39524 -tests/fm-cursor-harness.test.sh 30212 -tests/fm-cursor-primary-live-e2e.test.sh 72 -tests/fm-cursor-primary.test.sh 52269 -tests/fm-daemon.test.sh 27262 -tests/fm-dispatch-resolve.test.sh 4397 -tests/fm-documentation-audiences.test.sh 847 -tests/fm-dod-lib.test.sh 4000 -tests/fm-extension-binding.test.sh 9053 -tests/fm-fleet-snapshot-view.test.sh 17465 -tests/fm-fleet-sync.test.sh 35983 -tests/fm-forge-detect.test.sh 160 -tests/fm-gate-refuse.test.sh 5328 -tests/fm-gemini-harness.test.sh 938 -tests/fm-gitignore-config.test.sh 58 -tests/fm-gotmp.test.sh 1320 -tests/fm-grok-continuity-live-e2e.test.sh 45 -tests/fm-grok-stop-live-e2e.test.sh 46 -tests/fm-guard-stale-banner.test.sh 14968 -tests/fm-harness-adapter-instructions-live-e2e.test.sh 48 -tests/fm-harness-adapter-references.test.sh 83 -tests/fm-harness-liveness-drift-live-e2e.test.sh 881 -tests/fm-harness-precedence.test.sh 3661 -tests/fm-herdr-pi-stale-registration-live-e2e.test.sh 47 -tests/fm-herdr-session-cleanup.test.sh 6828 -tests/fm-herdr-submit-confirm-live-e2e.test.sh 46 -tests/fm-herdr-version-floor-live-e2e.test.sh 72 -tests/fm-home-summary-refresh.test.sh 37264 -tests/fm-inactive-reconcile.test.sh 53178 -tests/fm-kimi-harness.test.sh 19151 -tests/fm-lint-workflows.test.sh 785 -tests/fm-live-gate.test.sh 1755 -tests/fm-mail-check.test.sh 9162 -tests/fm-mail.test.sh 9703 -tests/fm-muse-harness.test.sh 40970 -tests/fm-muse-signals-live-e2e.test.sh 77 -tests/fm-nm-test-contract.test.sh 128 -tests/fm-no-mistakes-required.test.sh 247 -tests/fm-omp-harness.test.sh 47734 -tests/fm-omp-primary-live-e2e.test.sh 46 -tests/fm-on.test.sh 11001 -tests/fm-opencode-primary-live-e2e.test.sh 48 -tests/fm-operational-input.test.sh 221 -tests/fm-peek-remote.test.sh 964 -tests/fm-pending-reply.test.sh 28255 -tests/fm-pi-branch-extension.test.sh 60394 -tests/fm-pi-branch-live-e2e.test.sh 72 -tests/fm-pi-branch-responsiveness-live-e2e.test.sh 13121 -tests/fm-pi-codex-native.test.sh 46 -tests/fm-pi-primary-live-e2e.test.sh 47 -tests/fm-pi-watch-extension.test.sh 50637 +tests/fm-afk-contract.test.sh 11101 +tests/fm-afk-inject-e2e.test.sh 41958 +tests/fm-afk-pi-herdr-return-e2e.test.sh 52 +tests/fm-afk-return.test.sh 47380 +tests/fm-agy-harness.test.sh 50959 +tests/fm-agy-signals-live-e2e.test.sh 53 +tests/fm-ask-user-authority.test.sh 171 +tests/fm-backend-cmux-smoke.test.sh 34 +tests/fm-backend-cmux.test.sh 3754 +tests/fm-backend-orca.test.sh 27102 +tests/fm-backend-tmux-smoke.test.sh 291 +tests/fm-backend-zellij-smoke.test.sh 23 +tests/fm-backend-zellij.test.sh 10453 +tests/fm-backend.test.sh 23932 +tests/fm-backlog-atomicity.test.sh 219379 +tests/fm-backlog-handoff.test.sh 57458 +tests/fm-backlog-read-bound.test.sh 24743 +tests/fm-bearings-board-lavish-live-e2e.test.sh 51 +tests/fm-bearings-board-render.test.sh 15612 +tests/fm-bearings-board.test.sh 40817 +tests/fm-bearings-snapshot.test.sh 186219 +tests/fm-bootstrap-network-parallel.test.sh 30424 +tests/fm-bootstrap.test.sh 50965 +tests/fm-branch-supervision.test.sh 22979 +tests/fm-busy-adapter-wiring.test.sh 31642 +tests/fm-busy-state.test.sh 3185 +tests/fm-calm-claude-mod-live-e2e.test.sh 47 +tests/fm-calm-claude-mod-plugin.test.sh 77 +tests/fm-calm-claude-mod.test.sh 2527 +tests/fm-calm-pi-extension.test.sh 56463 +tests/fm-calm-pi-queue-retention-live-e2e.test.sh 1345 +tests/fm-check-unregister.test.sh 469 +tests/fm-ci-workflow.test.sh 5833 +tests/fm-classify-corr-token.test.sh 23085 +tests/fm-classify-decision-key.test.sh 4362 +tests/fm-claude-stop-autoarm-live-e2e.test.sh 73 +tests/fm-claude-stop-autoarm.test.sh 61189 +tests/fm-claude-trust.test.sh 12010 +tests/fm-cmux-claude-composer-live-e2e.test.sh 77 +tests/fm-codex-continuity-live-e2e.test.sh 108 +tests/fm-codex-hook-layer-live-e2e.test.sh 108 +tests/fm-composer-codex-idle-live-e2e.test.sh 77 +tests/fm-composer-matrix-live-e2e.test.sh 51 +tests/fm-contributions.test.sh 140911 +tests/fm-control-relaunch.test.sh 114115 +tests/fm-control.test.sh 72794 +tests/fm-cursor-harness.test.sh 30088 +tests/fm-cursor-primary-live-e2e.test.sh 75 +tests/fm-cursor-primary.test.sh 69845 +tests/fm-daemon.test.sh 33606 +tests/fm-devin-harness.test.sh 3725 +tests/fm-devin-signals-live-e2e.test.sh 49 +tests/fm-dispatch-resolve.test.sh 10051 +tests/fm-documentation-audiences.test.sh 1301 +tests/fm-dod-lib.test.sh 2035 +tests/fm-extension-binding.test.sh 11105 +tests/fm-fleet-ledger.test.sh 19980 +tests/fm-fleet-snapshot-view.test.sh 23334 +tests/fm-fleet-sync.test.sh 40541 +tests/fm-forge-detect.test.sh 193 +tests/fm-fork-free-helpers.test.sh 746 +tests/fm-gate-refuse.test.sh 9953 +tests/fm-gemini-harness.test.sh 947 +tests/fm-git-strip-ai-trailers.test.sh 2067 +tests/fm-gitignore-config.test.sh 59 +tests/fm-gotmp.test.sh 1509 +tests/fm-grok-continuity-live-e2e.test.sh 46 +tests/fm-grok-stop-live-e2e.test.sh 48 +tests/fm-guard-stale-banner.test.sh 17234 +tests/fm-harness-adapter-instructions-live-e2e.test.sh 72 +tests/fm-harness-adapter-references.test.sh 64 +tests/fm-harness-liveness-drift-live-e2e.test.sh 1309 +tests/fm-harness-precedence.test.sh 4083 +tests/fm-herdr-pi-stale-registration-live-e2e.test.sh 55 +tests/fm-herdr-session-cleanup.test.sh 7425 +tests/fm-herdr-submit-confirm-live-e2e.test.sh 51 +tests/fm-herdr-version-floor-live-e2e.test.sh 50 +tests/fm-home-summary-refresh.test.sh 37057 +tests/fm-host-mirror-live-e2e.test.sh 79 +tests/fm-host-mirror.test.sh 11587 +tests/fm-inactive-reconcile.test.sh 60823 +tests/fm-inbox.test.sh 6062 +tests/fm-jev-mem-guard.test.sh 336 +tests/fm-kimi-harness.test.sh 58917 +tests/fm-launch-prompt-signals-live-e2e.test.sh 50 +tests/fm-lint-workflows.test.sh 872 +tests/fm-live-gate.test.sh 7452 +tests/fm-live-lab-up-mate.test.sh 17363 +tests/fm-live-lab.test.sh 79639 +tests/fm-mail-check.test.sh 7524 +tests/fm-mail.test.sh 9684 +tests/fm-muse-harness.test.sh 46548 +tests/fm-muse-signals-live-e2e.test.sh 52 +tests/fm-nm-test-contract.test.sh 853 +tests/fm-no-mistakes-required.test.sh 270 +tests/fm-omp-harness.test.sh 63796 +tests/fm-omp-primary-live-e2e.test.sh 74 +tests/fm-on.test.sh 11473 +tests/fm-opencode-primary-live-e2e.test.sh 47 +tests/fm-operational-input.test.sh 2404 +tests/fm-peek-remote.test.sh 1082 +tests/fm-pending-reply.test.sh 41090 +tests/fm-pi-branch-extension.test.sh 77218 +tests/fm-pi-branch-live-e2e.test.sh 48 +tests/fm-pi-branch-responsiveness-live-e2e.test.sh 12834 +tests/fm-pi-codex-native.test.sh 75 +tests/fm-pi-primary-live-e2e.test.sh 72 +tests/fm-pi-watch-extension.test.sh 56515 tests/fm-pi-windows-shell-invocation.test.sh 5121 -tests/fm-pr-check-security.test.sh 226546 -tests/fm-pr-reviewers.test.sh 273 -tests/fm-pr-state-live-e2e.test.sh 45 -tests/fm-pr-state.test.sh 531 -tests/fm-procevent-quota.test.sh 1900 -tests/fm-procevent-when.test.sh 23805 -tests/fm-procevent.test.sh 221745 -tests/fm-project-origin.test.sh 136 -tests/fm-public-followup.test.sh 153508 -tests/fm-quota-array-dispatch-live-e2e.test.sh 71 -tests/fm-quota-choose.test.sh 1484 -tests/fm-remote-backlog-handoff.test.sh 73123 -tests/fm-remote-doctor.test.sh 13889 -tests/fm-remote-entrypoint.test.sh 108 -tests/fm-remote-herdr-guard.test.sh 3044 -tests/fm-remote-job-orphan-reap.test.sh 2905 -tests/fm-remote-job.test.sh 59354 -tests/fm-remote-reply.test.sh 118669 -tests/fm-remote-secondmate-lifecycle-e2e.test.sh 241208 -tests/fm-remote-secondmate-parent-binding.test.sh 32176 -tests/fm-remote-secondmate-trace-context.test.sh 59689 -tests/fm-remote-transport-lanes.test.sh 62635 -tests/fm-rovo-harness.test.sh 14322 -tests/fm-rovo-signals-live-e2e.test.sh 48 -tests/fm-secondmate-harness.test.sh 163801 -tests/fm-secondmate-lifecycle-e2e.test.sh 9633 -tests/fm-secondmate-liveness.test.sh 10402 -tests/fm-secondmate-reconcile.test.sh 97544 -tests/fm-secondmate-restart.test.sh 44488 -tests/fm-secondmate-safety.test.sh 127260 -tests/fm-secondmate-sync.test.sh 54502 -tests/fm-send-agy-confirm.test.sh 3983 -tests/fm-send-inbox-doorbell-live-e2e.test.sh 46 -tests/fm-send-inbox.test.sh 38632 -tests/fm-send-remote-delivery.test.sh 27717 -tests/fm-send-resolve-key.test.sh 28685 -tests/fm-send-secondmate-marker-herdr-e2e.test.sh 52 -tests/fm-send-secondmate-marker.test.sh 5309 -tests/fm-session-lock-ancestry.test.sh 2857 -tests/fm-session-start.test.sh 179350 -tests/fm-sessionstart-hook-live-e2e.test.sh 97 -tests/fm-sessionstart-instruction-refresh-live-e2e.test.sh 46 -tests/fm-sessionstart-nudge.test.sh 66247 -tests/fm-shared-captain-inheritance.test.sh 5687 -tests/fm-spawn-dispatch-profile.test.sh 138433 -tests/fm-spawn-pool-base-freshen.test.sh 62249 -tests/fm-spawn-worktree-settle.test.sh 8482 -tests/fm-startup-memory-budget.test.sh 7392 -tests/fm-startup-network.test.sh 61336 -tests/fm-stat-shadowing.test.sh 48 -tests/fm-stow-cascade.test.sh 3022 -tests/fm-subagent-pretool-check.test.sh 949 -tests/fm-supervision-events.test.sh 659 -tests/fm-supervision-host-live-e2e.test.sh 50 -tests/fm-supervision-host.test.sh 41512 -tests/fm-tangle-guard.test.sh 7470 -tests/fm-task-delivery.test.sh 19784 -tests/fm-task-inbox.test.sh 30004 -tests/fm-tasks-axi.test.sh 1953 -tests/fm-teardown-endpoint-safety.test.sh 33210 -tests/fm-teardown.test.sh 145174 -tests/fm-test-fixture-cleanup.test.sh 937 -tests/fm-test-fixtures.test.sh 1562 -tests/fm-test-isolation-proof.test.sh 2692 -tests/fm-timeout-lib.test.sh 8541 -tests/fm-tmux-agent-liveness.test.sh 1953 -tests/fm-tool-update-check.test.sh 13832 -tests/fm-trace-context-lib.test.sh 227 -tests/fm-trace-context-spawn.test.sh 49071 -tests/fm-turnend-foreign-owner-arm-fix.test.sh 2397 -tests/fm-turnend-guard.test.sh 33450 -tests/fm-update.test.sh 11572 -tests/fm-vendor-auth-probe.test.sh 45255 -tests/fm-voice-relay.test.sh 32486 -tests/fm-wake-daemon-lifecycle-e2e.test.sh 7477 -tests/fm-wake-drain-open-decisions-cursor.test.sh 38506 -tests/fm-wake-drain-open-decisions.test.sh 6890 -tests/fm-wake-drain-outcome-backstop.test.sh 44076 -tests/fm-wake-drain-unread-status.test.sh 16169 -tests/fm-wake-queue.test.sh 85252 -tests/fm-watch-arm.test.sh 68479 -tests/fm-watch-checkpoint.test.sh 6076 -tests/fm-watch-recovery-loop.test.sh 58946 -tests/fm-watch-triage.test.sh 697969 -tests/fm-watcher-lock.test.sh 108940 +tests/fm-pr-check-security.test.sh 300675 +tests/fm-pr-reviewers.test.sh 157 +tests/fm-pr-state-live-e2e.test.sh 47 +tests/fm-pr-state.test.sh 525 +tests/fm-procevent-quota.test.sh 2459 +tests/fm-procevent-when.test.sh 25674 +tests/fm-procevent.test.sh 292297 +tests/fm-project-origin.test.sh 123 +tests/fm-public-followup.test.sh 381564 +tests/fm-quota-array-dispatch-live-e2e.test.sh 50 +tests/fm-quota-choose.test.sh 2860 +tests/fm-remote-backlog-handoff.test.sh 82063 +tests/fm-remote-doctor.test.sh 14460 +tests/fm-remote-entrypoint.test.sh 134 +tests/fm-remote-herdr-guard.test.sh 3140 +tests/fm-remote-job-orphan-reap.test.sh 2985 +tests/fm-remote-job.test.sh 81046 +tests/fm-remote-reply.test.sh 140887 +tests/fm-remote-secondmate-lifecycle-e2e.test.sh 345655 +tests/fm-remote-secondmate-parent-binding.test.sh 42294 +tests/fm-remote-secondmate-relaunch.test.sh 879 +tests/fm-remote-secondmate-trace-context.test.sh 74870 +tests/fm-remote-transport-lanes.test.sh 66089 +tests/fm-rovo-harness.test.sh 15691 +tests/fm-rovo-signals-live-e2e.test.sh 52 +tests/fm-secondmate-harness.test.sh 188187 +tests/fm-secondmate-lifecycle-e2e.test.sh 11268 +tests/fm-secondmate-liveness.test.sh 24564 +tests/fm-secondmate-reconcile.test.sh 100853 +tests/fm-secondmate-restart.test.sh 52591 +tests/fm-secondmate-safety.test.sh 69424 +tests/fm-secondmate-sync.test.sh 55501 +tests/fm-send-agy-confirm.test.sh 4440 +tests/fm-send-inbox-doorbell-live-e2e.test.sh 108 +tests/fm-send-inbox.test.sh 41713 +tests/fm-send-remote-delivery.test.sh 31964 +tests/fm-send-resolve-key.test.sh 47317 +tests/fm-send-secondmate-marker-herdr-e2e.test.sh 80 +tests/fm-send-secondmate-marker.test.sh 7574 +tests/fm-session-lock-ancestry.test.sh 18918 +tests/fm-session-start.test.sh 363574 +tests/fm-sessionstart-hook-live-e2e.test.sh 50 +tests/fm-sessionstart-instruction-refresh-live-e2e.test.sh 49 +tests/fm-sessionstart-nudge.test.sh 71802 +tests/fm-shared-captain-inheritance.test.sh 7991 +tests/fm-spawn-compact-adviser-disable-remote.test.sh 38561 +tests/fm-spawn-compact-adviser-disable.test.sh 21654 +tests/fm-spawn-dispatch-profile.test.sh 197548 +tests/fm-spawn-orca-worktree.test.sh 2400 +tests/fm-spawn-pool-base-freshen.test.sh 68652 +tests/fm-spawn-worktree-settle.test.sh 9309 +tests/fm-startup-memory-budget.test.sh 8086 +tests/fm-startup-network.test.sh 72106 +tests/fm-stat-shadowing.test.sh 75 +tests/fm-stow-cascade.test.sh 3058 +tests/fm-subagent-pretool-check.test.sh 998 +tests/fm-supervision-events.test.sh 673 +tests/fm-supervision-host-attended-live-e2e.test.sh 49 +tests/fm-supervision-host-live-e2e.test.sh 75 +tests/fm-supervision-host.test.sh 789123 +tests/fm-tangle-guard.test.sh 8501 +tests/fm-task-delivery.test.sh 32789 +tests/fm-task-inbox.test.sh 31965 +tests/fm-tasks-axi.test.sh 2293 +tests/fm-teardown-endpoint-safety.test.sh 40851 +tests/fm-teardown.test.sh 202132 +tests/fm-test-fixture-cleanup.test.sh 866 +tests/fm-test-fixtures.test.sh 1802 +tests/fm-test-isolation-proof.test.sh 2866 +tests/fm-timeout-lib.test.sh 10750 +tests/fm-tmux-agent-liveness.test.sh 3770 +tests/fm-tool-update-check.test.sh 14383 +tests/fm-trace-context-lib.test.sh 221 +tests/fm-trace-context-spawn.test.sh 57488 +tests/fm-turnend-foreign-owner-arm-fix.test.sh 5575 +tests/fm-turnend-guard.test.sh 34727 +tests/fm-update.test.sh 11894 +tests/fm-vendor-auth-probe.test.sh 43278 +tests/fm-voice-relay.test.sh 28917 +tests/fm-wake-daemon-lifecycle-e2e.test.sh 7345 +tests/fm-wake-drain-open-decisions-cursor.test.sh 47677 +tests/fm-wake-drain-open-decisions.test.sh 8781 +tests/fm-wake-drain-outcome-backstop.test.sh 46316 +tests/fm-wake-drain-unread-status.test.sh 24251 +tests/fm-wake-queue.test.sh 165906 +tests/fm-watch-arm.test.sh 113076 +tests/fm-watch-checkpoint.test.sh 11234 +tests/fm-watch-recovery-loop.test.sh 59092 +tests/fm-watch-triage.test.sh 1074843 +tests/fm-watcher-lock.test.sh 72022 +tests/fm-worker-account-live-e2e.test.sh 3179 +tests/fm-worker-account.test.sh 37445 EOF } @@ -874,6 +901,15 @@ portable_serial_unhinted() { rm -rf "$tmp" } +# Sum serial weights for paths on stdin, including the unmeasured default. +portable_serial_lane_weight() { + awk -v fallback="$PORTABLE_SERIAL_DEFAULT_WEIGHT_MS" ' + NR == FNR { if (NF) { hint[$1] = $2 }; next } + NF { total += ($1 in hint) ? hint[$1] : fallback } + END { printf "%d\n", total + 0 } + ' <(portable_serial_weight_hints) - +} + portable_parallel_weight_for() { local want=$1 ms ms=$(portable_parallel_weight_hints | awk -v want="$want" '$1 == want { print $2; exit }') @@ -1011,7 +1047,7 @@ select_lane() { } run_coverage_guard() { - local tmp missing extra a b shard unhinted serial_total + local tmp missing extra a b shard unhinted serial_total serial_ms serial_max_ms=0 local p1_ms p1_unhinted p2_ms p2_unhinted parallel_max_ms parallel_imbalance_ms local -a saved_scripts=() tmp=$(mktemp -d "${TMPDIR:-/tmp}/fm-test-coverage.XXXXXX") @@ -1057,6 +1093,8 @@ run_coverage_guard() { return 1 fi printf '%s\n' "${SCRIPTS[@]+"${SCRIPTS[@]}"}" >>"$tmp/serial_shards_raw" + serial_ms=$(printf '%s\n' "${SCRIPTS[@]+"${SCRIPTS[@]}"}" | portable_serial_lane_weight) + [ "$serial_ms" -le "$serial_max_ms" ] || serial_max_ms=$serial_ms shard=$((shard + 1)) done SCRIPTS=() @@ -1132,6 +1170,13 @@ run_coverage_guard() { return 1 fi + if [ "$serial_max_ms" -gt "$PORTABLE_SERIAL_MAX_WEIGHT_MS" ]; then + log "coverage guard: largest portable serial shard packs ${serial_max_ms}ms above the ${PORTABLE_SERIAL_MAX_WEIGHT_MS}ms target" + log "refresh CI hints and rebalance or add shards; do not raise the job timeout: docs/fm-test-portable-shards.md" + rm -rf "$tmp" + return 1 + fi + if [ -x "$ROOT/bin/fm-test-isolation-proof.sh" ]; then "$ROOT/bin/fm-test-isolation-proof.sh" --list | LC_ALL=C sort -u >"$tmp/proof_list" if ! cmp -s "$tmp/proven" "$tmp/proof_list"; then @@ -1151,7 +1196,7 @@ run_coverage_guard() { parallel_imbalance_ms=$((p1_ms - p2_ms)) [ "$parallel_imbalance_ms" -ge 0 ] || parallel_imbalance_ms=$((-parallel_imbalance_ms)) - printf 'FM_TEST_COVERAGE ok total=%s parallel=%s parallel_max_ms=%s parallel_imbalance_ms=%s parallel_unhinted=%s serial=%s serial_shards=%s serial_unhinted=%s herdr=%s\n' \ + printf 'FM_TEST_COVERAGE ok total=%s parallel=%s parallel_max_ms=%s parallel_imbalance_ms=%s parallel_unhinted=%s serial=%s serial_shards=%s serial_unhinted=%s serial_max_ms=%s serial_budget_ms=%s herdr=%s\n' \ "$(wc -l <"$tmp/all" | tr -d ' ')" \ "$(wc -l <"$tmp/shards_union" | tr -d ' ')" \ "$parallel_max_ms" \ @@ -1160,6 +1205,8 @@ run_coverage_guard() { "$(wc -l <"$tmp/serial" | tr -d ' ')" \ "$PORTABLE_SERIAL_SHARDS" \ "$unhinted" \ + "$serial_max_ms" \ + "$PORTABLE_SERIAL_MAX_WEIGHT_MS" \ "$(wc -l <"$tmp/herdr" | tr -d ' ')" rm -rf "$tmp" return 0 diff --git a/docs/fm-test-portable-shards.md b/docs/fm-test-portable-shards.md index 859ab9da099..1e152cda2e9 100644 --- a/docs/fm-test-portable-shards.md +++ b/docs/fm-test-portable-shards.md @@ -9,27 +9,26 @@ Balance hints come from serial runs of the real lanes on `ubuntu-latest`. The concurrent isolation proof in [fm-test-isolation-proof.md](fm-test-isolation-proof.md) establishes concurrency safety, not serial CI duration. Local timings are not interchangeable with CI timings: platform and machine load can affect each script differently and change their relative weights. -The retained hints are the slowest completed value each script reached across six CI runs on 2026-09-10: [34459949083](https://github.com/kunchenguid/firstmate/actions/runs/34459949083), [34460760299](https://github.com/kunchenguid/firstmate/actions/runs/34460760299), [34462530836](https://github.com/kunchenguid/firstmate/actions/runs/34462530836), [34462758357](https://github.com/kunchenguid/firstmate/actions/runs/34462758357), [34466966385](https://github.com/kunchenguid/firstmate/actions/runs/34466966385), and [34470382458](https://github.com/kunchenguid/firstmate/actions/runs/34470382458). -Shard 2 completed in all six, so its scripts come from the uploaded `fm-test-timing-portable-parallel-2` artifacts. -Shard 1 was cancelled at its job cap in five of the six, so its scripts come from the `FM_TEST_END duration_ms=` markers in each cancelled job's log, which record every script that finished before the cancellation, plus the one complete `fm-test-timing-portable-parallel-1` artifact from run 34462758357. +Both hint tables were refreshed on 2026-09-30 from five Ubuntu CI runs: [36583881812](https://github.com/kunchenguid/firstmate/actions/runs/36583881812), [36658498535](https://github.com/kunchenguid/firstmate/actions/runs/36658498535), [36663947738](https://github.com/kunchenguid/firstmate/actions/runs/36663947738), [36664663190](https://github.com/kunchenguid/firstmate/actions/runs/36664663190), and [36669175457](https://github.com/kunchenguid/firstmate/actions/runs/36669175457). +Use the slowest successful `duration_ms` per script across their uploaded portable timing artifacts and completed `FM_TEST_END` log markers, with the two version/platform exceptions below. +All artifact records were cross-checked against the corresponding job's markers. +This covers all 24 parallel and 201 serial members; an existing live-capability skip is a portable-runner measurement, not a timing claim for the unavailable live integration. Observed maxima provide conservative packing weights, not an upper bound on future durations. -The measurements cover all 24 candidates, with six samples per script except: +Two serial-5 jobs were cancelled at their 30-minute cap and uploaded no artifact. +Their completed log markers supplement the complete runs, but a cancelled job's wall time is only a lower bound and its unfinished or never-started scripts have no completed sample. +A failed script's duration is excluded even when its lane uploaded an artifact. +In particular, run 36664663190's serial 5 finished in 22m15s with an assertion failure, not a timeout; treating that as a healthy whole-lane sample would hide the failure. +Collect successful per-script measurements for every member before calculating a split. -| Samples | Scripts | -|---:|---| -| 4 | `tests/fm-lint.test.sh` | -| 3 | `tests/fm-pi-primary-types.test.sh`, `tests/fm-review-diff.test.sh` | -| 1 | `tests/fm-brief.test.sh`, `tests/fm-transition-lib.test.sh` | - -The two scripts with one sample are the tail of shard 1 that only the complete run reached. -Collect completed per-script measurements for every member before calculating a split. -A cancelled lane's elapsed duration is only a lower bound; its unfinished scripts have no completed duration for that invocation. -The complete historical run supplies tail-script hints, not a completion time for any later cancelled invocation or for the rebalanced jobs. +`tests/fm-supervision-host.test.sh` uses 789123 ms from run 36669175457, after the merged [host runtime fix](https://github.com/kunchenguid/firstmate/pull/6179), rather than its pre-fix maximum of 1065298 ms. +That post-fix value has only one sample in this baseline, so further green runs must establish its variance. +The native-Windows-only `tests/fm-pi-windows-shell-invocation.test.sh` retains its separate 5121 ms measurement from 2026-09-06T21:02Z instead of a portable capability skip. +The session-start hint retains its pre-optimization maximum until CI measures the shorter fixture-only home-summary bound; do not discount a local speedup from CI packing weights. ## Parallel lanes -The two parallel lanes use longest-processing-time assignment over those hints. +The two parallel lanes use longest-processing-time assignment over those hints, with the Pi typecheck pinned to the job that installs its prerequisite. [`bin/fm-test-run.sh`](../bin/fm-test-run.sh) holds the duration values in `portable_parallel_weight_hints` and the ordered memberships and lane-specific prerequisite constraints beside `list_portable_parallel_1` and `list_portable_parallel_2`. Read the derived packing estimates with that runner's `--check-coverage`; its header and `--help` own the output fields and the selection-specific `--list-scheduled` weight rules. The largest individual hint sets a lower bound on the estimated duration of any split, regardless of how evenly the remaining work is assigned. @@ -57,21 +56,21 @@ Each shard is still strictly serial in itself, and separate runners mean no two `.github/workflows/ci.yml` derives the same `n` from `strategy.job-total` rather than a literal, so changing the shard count in either file without the other fails the lane loudly instead of leaving part of the required suite unrun. Assignment is longest-processing-time bin packing over per-script duration hints embedded in `bin/fm-test-run.sh`. -The serial hints were refreshed from successful per-script records in the `fm-test-timing-portable-serial-*` artifacts of the complete green [run 35279383618](https://github.com/kunchenguid/firstmate/actions/runs/35279383618) and the available completed shards of [run 35282466441](https://github.com/kunchenguid/firstmate/actions/runs/35282466441) on 2026-09-17. -Together these cover all 176 serial scripts at refresh time; retain the slower successful sample where both exist. -The native-Windows-only `tests/fm-pi-windows-shell-invocation.test.sh` retains its separate 5121 ms measurement from 2026-09-06T21:02Z instead of a portable capability skip. -An unfinished or failed invocation is not a healthy duration sample. +[Verification inputs](#verification-inputs) owns the measurement provenance and exceptions. A script with no hint gets the conservative `PORTABLE_SERIAL_DEFAULT_WEIGHT_MS` default. Hints only affect balance: the coverage guard keeps the partition complete and disjoint whatever they say, so a stale hint costs a slower shard rather than lost coverage. Balance is still worth keeping current, because enough unmeasured scripts let one shard carry more than twice another shard's real work and reach the job cap while another runner sits idle. -That is not hypothetical: by 2026-09-01 the lane had grown from 116 to 139 scripts and from ~42 to ~63 minutes, 17 scripts were still unmeasured, and several hints were low by 2-5x, so shard 3 of 4 ran 17-20 minutes against its 20-minute cap while shard 1 ran 11.5 minutes and run [33574154856](https://github.com/kunchenguid/firstmate/actions/runs/33574154856) timed out seconds after a passing test. -`bin/fm-test-run.sh --check-coverage` now reports the unmeasured share as `serial_unhinted=` and refuses past `PORTABLE_SERIAL_MAX_UNHINTED_PERCENT`, so hint drift fails the coverage guard instead of silently pushing one shard into its job cap. -Refresh the hints whenever the serial lane gains scripts, rather than waiting for that bound to trip. +`bin/fm-test-run.sh --check-coverage` reports the unmeasured share as `serial_unhinted=` and refuses past `PORTABLE_SERIAL_MAX_UNHINTED_PERCENT`. +That catches missing hints, not stale existing hints: the host suite still had a 41512 ms hint after growing to over 1000 seconds in CI, so the old split placed it beside another 12 minutes of work while passing the guard. +Refresh the hints whenever a serial member grows materially or the lane gains scripts, rather than waiting for missing-hint coverage to trip. `bin/fm-test-run.sh` owns the per-shard packing, so its `--check-coverage` output is the current account of lane size and coverage rather than a copied inventory. -Nine serial runners pack the refreshed measurements into a longest modeled script sum of 697969 ms (11m38s), with other shards near 10m36s. -The longest script, `tests/fm-watch-triage.test.sh`, legitimately occupies one whole shard and is the indivisible floor for this layout. -This is a packing estimate, not measured new-workflow execution or an end-to-end latency guarantee. +Its header and `--help` own the modeled-budget check and output fields; read the current estimates from `--check-coverage` instead of retaining copied lane sums here. +[`tests/fm-test-run.test.sh`](../tests/fm-test-run.test.sh), in `test_portable_serial_packing_budget_boundary`, verifies acceptance exactly at the budget and refusal one millisecond above it through the executable runner. +The longest script, `tests/fm-watch-triage.test.sh`, is the indivisible floor for this layout. +The estimates use per-file maxima from different runs, not measured rebalanced jobs or an end-to-end latency guarantee. +The baseline watch-triage samples range from 944375 to 1074843 ms, while each observed completed portable job adds at most 30 seconds beyond its summed scripts in these runs. +Even so, maxima from five runs do not establish a P95 or guarantee future headroom. Job timeouts remain hang tripwires under the policy in [Timeouts](#timeouts) below; they are not the desired healthy duration. `tests/fm-ci-workflow.test.sh` compares the parsed CI matrix to the executable runner lanes, and the runner rejects parallel `--jobs` on a serial lane even when that shard has only one member. @@ -79,9 +78,9 @@ Refresh the CI-derived hints by downloading the per-shard timing artifacts from ```sh for run in ; do - gh run download "$run" -R kunchenguid/firstmate --pattern 'fm-test-timing-portable-serial-*' -D "/tmp/fm-serial/$run" + gh-axi run download "$run" -R kunchenguid/firstmate --dir "/tmp/fm-serial/$run" done -jq -r '.scripts[] | select(.exit == 0) | [.path, .duration_ms] | @tsv' /tmp/fm-serial/*/*/*.json \ +jq -r '.scripts[] | select(.exit == 0) | [.path, .duration_ms] | @tsv' /tmp/fm-serial/*/fm-test-timing-portable-serial-*/*.json \ | awk -F'\t' '$2 > m[$1] { m[$1] = $2 } END { for (p in m) print p, m[p] }' \ | LC_ALL=C sort bin/fm-test-run.sh --check-coverage @@ -96,7 +95,7 @@ Measure native-Windows-only scripts through the focused Git Bash runner and reta `bin/fm-test-run.sh --check-coverage` verifies that both parallel lanes partition the proven-isolated set. It also verifies that the parallel lanes, portable serial lane, and real-Herdr family are disjoint and cover every `tests/*.test.sh` script. It separately verifies that the portable serial CI shards are non-empty, disjoint, and together equal the portable serial lane. -It reports the unmeasured serial share as `serial_unhinted=` and refuses when that share exceeds `PORTABLE_SERIAL_MAX_UNHINTED_PERCENT`, so the shards stay balanced on evidence rather than on the default weight. +Its hint-coverage and modeled-budget checks are described in [Portable serial CI shards](#portable-serial-ci-shards); neither replaces inspection of actual CI timing artifacts. ## Timing artifacts @@ -112,8 +111,9 @@ Its `--list-files` interface exposes partition membership; `tests/fm-lint.test.s The workflow uploads each partition's quiet telemetry plus its per-root lifecycle sidecar to distinguish analysis cost, memory use, and host contention. No fast mode, path skips, reduced checks, or paid runner provisioning is part of this layout. -The performance objective is a complete green run under fifteen minutes including start delay: roughly twelve minutes of longest-path execution, at most two minutes of runner delay, and less than one minute of other overhead. -The candidate uses fourteen long-lived Linux jobs (nine serial, two parallel, Herdr, two lint), plus short checks and macOS; insufficient shared account capacity can erase the packing gain. +The longer-term performance objective remains a complete green run under fifteen minutes including start delay, but the current watch-triage floor alone exceeds that objective. +The immediate packing target is the runner's modeled script budget, not a claim that more shards alone can make an indivisible script faster. +The layout uses fourteen long-lived Linux jobs (nine serial, two parallel, Herdr, two lint), plus short checks and macOS; insufficient shared account capacity can erase the packing gain. Compare complete before/after runs, preserve cancelled and partial-run evidence, and measure a representative normal-run sample before claiming a P95 improvement. The workflow retains per-PR supersession without cancelling main pushes or changing the compliance workflow's event semantics. @@ -126,7 +126,7 @@ The workflow retains per-PR supersession without cancelling main pushes or chang CI job timeouts follow one three-tier policy, so the workflow reads as a policy rather than as a collection of per-job numbers. Every tier is a hang tripwire with headroom above the healthy duration, never a packing estimate or a runtime target. -A lane that reaches its tier bound is wedged, not slow, so change the policy here rather than treating the bound as a way to fit a slower lane. +A lane that reaches its tier bound needs investigation and a distribution or runtime fix, not a larger timeout to fit the same work. | Tier | Jobs | Bound | Rationale | |---|---|---|---| diff --git a/tests/fm-session-start.test.sh b/tests/fm-session-start.test.sh index 913c3734311..729edd36055 100755 --- a/tests/fm-session-start.test.sh +++ b/tests/fm-session-start.test.sh @@ -1474,7 +1474,11 @@ EOF printf 'window=sess:p-slow\nkind=ship\nbackend=herdr\n' > "$home/state/task-a-slow.meta" printf 'window=sess:p-live\nkind=ship\nbackend=herdr\n' > "$home/state/task-z-live.meta" - out=$(FM_SESSION_START_ENDPOINT_TIMEOUT=2 run_session_start "$home" "$root" "$fakebin:$BASE_PATH") || status=$? + # The same fake hangs the side-band home summary before the endpoint section. + # Bound that unrelated refresh at 5s instead of paying its production 60s; + # the endpoint's own 2s bound and descendant-cleanup assertions stay real. + out=$(FM_HOME_SUMMARY_TIMEOUT=5 FM_SESSION_START_ENDPOINT_TIMEOUT=2 \ + run_session_start "$home" "$root" "$fakebin:$BASE_PATH") || status=$? expect_code 0 "$status" "a hung endpoint read must not fail the digest" assert_contains "$out" \ @@ -1506,7 +1510,10 @@ EOF printf 'window=sess:p-slow\nkind=ship\nbackend=herdr\n' > "$home/state/task-a-slow.meta" printf 'window=sess:p-live\nkind=ship\nbackend=herdr\n' > "$home/state/task-z-live.meta" - out=$(FM_SESSION_START_ENDPOINT_TIMEOUT=00 run_session_start "$home" "$root" "$fakebin:$BASE_PATH") || status=$? + # Only the unrelated summary gets a shorter fixture budget. The invalid + # endpoint value must still fall back to the real 10s production bound. + out=$(FM_HOME_SUMMARY_TIMEOUT=5 FM_SESSION_START_ENDPOINT_TIMEOUT=00 \ + run_session_start "$home" "$root" "$fakebin:$BASE_PATH") || status=$? expect_code 0 "$status" "a padded-zero per-read bound must not fail the digest" assert_contains "$out" \ diff --git a/tests/fm-test-run.test.sh b/tests/fm-test-run.test.sh index 211a6fde79c..5f70ea8a557 100755 --- a/tests/fm-test-run.test.sh +++ b/tests/fm-test-run.test.sh @@ -1028,11 +1028,11 @@ test_list_scheduled_non_lane_selections_use_serial_weights() { printf '\n' >>"$repo/$script" done printf '%s\n' \ + tests/fm-kimi-harness.test.sh \ tests/fm-muse-harness.test.sh \ tests/fm-brief.test.sh \ tests/fm-captain-hold-lifecycle.test.sh \ tests/fm-lint.test.sh \ - tests/fm-kimi-harness.test.sh \ tests/fm-operational-input.test.sh >"$tmp/expected" for selection in family all changed scripts; do case "$selection" in @@ -1157,7 +1157,7 @@ test_portable_serial_shards_partition_the_serial_lane() { } test_portable_serial_hint_coverage_is_reported_and_bounded() { - local out serial unhinted + local out serial unhinted max budget # Shards are packed from measured duration hints, so an unmeasured script is # placed on a guess. Enough of them and the partition still looks balanced by # script count while one shard carries far more real work than another and @@ -1178,7 +1178,56 @@ test_portable_serial_hint_coverage_is_reported_and_bounded() { # this trips (docs/fm-test-portable-shards.md). [ "$((unhinted * 100))" -le "$((serial * 15))" ] \ || fail "$unhinted of $serial portable serial scripts lack a measured hint; refresh them" - pass "coverage guard reports and bounds the unmeasured portable serial share" + # A complete partition can still overflow a CI job. Assert the runner's + # modeled packing target through its executable interface, not source hints. + max=$(printf '%s\n' "$out" | sed -n 's/.*serial_max_ms=\([0-9][0-9]*\).*/\1/p') + budget=$(printf '%s\n' "$out" | sed -n 's/.*serial_budget_ms=\([0-9][0-9]*\).*/\1/p') + [ -n "$max" ] && [ -n "$budget" ] \ + || fail "coverage summary must carry serial packing and budget: $out" + [ "$budget" -eq 1200000 ] || fail "packing must leave ten minutes of the normal CI tier" + [ "$max" -gt 0 ] && [ "$max" -le "$budget" ] \ + || fail "largest serial shard packs ${max}ms above the ${budget}ms target" + pass "coverage guard bounds the unmeasured share and serial packing within twenty minutes" +} + +test_portable_serial_packing_budget_boundary() { + local tmp repo script weight out rc + tmp=$(fm_test_tmproot fm-test-run-packing-boundary) + repo="$tmp/repo" + mkdir -p "$repo/bin" "$repo/tests" + # Preserve the real inventory and packing policy without executing suites. + # Only the fixture's measured timing input changes at the boundary. + while IFS= read -r script; do + printf '#!/usr/bin/env bash\nexit 0\n' >"$repo/$script" + done < <("$RUNNER" --list --all) + + for weight in 1200000 1200001; do + cp "$RUNNER" "$repo/bin/fm-test-run.sh" + python3 - "$repo/bin/fm-test-run.sh" "$weight" <<'PY' \ + || fail "could not seed the fixture's measured timing input" +from pathlib import Path +import re, sys +runner = Path(sys.argv[1]) +runner.write_text(re.sub( + r"(?m)^tests/fm-watch-triage\.test\.sh [0-9]+$", + f"tests/fm-watch-triage.test.sh {sys.argv[2]}", + runner.read_text(), +)) +PY + out=$(bash "$repo/bin/fm-test-run.sh" --check-coverage 2>&1) && rc=0 || rc=$? + if [ "$weight" -eq 1200000 ]; then + expect_code 0 "$rc" "packing exactly at the budget must be accepted" + assert_contains "$out" "FM_TEST_COVERAGE ok" "boundary coverage did not pass" + assert_contains "$out" "serial_max_ms=1200000" "fixture did not pack exactly at the budget" + assert_contains "$out" "serial_budget_ms=1200000" "fixture changed the packing budget" + else + expect_code 1 "$rc" "packing one millisecond above the budget must be refused" + assert_contains "$out" "largest portable serial shard packs 1200001ms above the 1200000ms target" \ + "over-budget refusal did not explain the modeled excess" + assert_not_contains "$out" "FM_TEST_COVERAGE ok" "over-budget packing reported success" + fi + done + pass "serial packing accepts the exact budget and refuses one millisecond above it" } test_portable_serial_shard_lane_refusals() { @@ -1805,6 +1854,7 @@ test_portable_shard_union_and_coverage_guard test_portable_parallel_lanes_stay_duration_balanced test_portable_serial_shards_partition_the_serial_lane test_portable_serial_hint_coverage_is_reported_and_bounded +test_portable_serial_packing_budget_boundary test_portable_serial_shard_lane_refusals test_jobs_requires_proven_isolated test_jobs_admits_a_concurrent_safe_family From bd744684ee9d7457c000496d16d6d2c136c24e9e Mon Sep 17 00:00:00 2001 From: Christopher McKay <101884182+karotkriss@users.noreply.github.com> Date: Wed, 30 Sep 2026 10:19:43 -0400 Subject: [PATCH 02/33] fix(bin): run no repository hook when core.hooksPath is empty (#6216) * fix(bin): run no repository hook when core.hooksPath is empty The per-task hook wrapper refused every commit in a repository whose own config sets core.hooksPath to the empty string, because git rev-parse --git-path hooks fails on it. Plain git reads that setting as no hooks, so the wrapper now runs none; every other lookup failure still refuses and shows git's error. Fixes #6171 * no-mistakes(review): Refuse commits when core.hooksPath is a valueless key * no-mistakes(document): Document empty core.hooksPath handling in commit attribution docs * no-mistakes(ci): When the wrapper refuses a commit, Git's hook-lookup error now shows up once instead of twice. That required changing one line in the wrapper, and the tests were extended so both bad-config cases would catch the duplicate. Invariant: when the wrapper refuses, Git's lookup error must appear exactly once. In the failure path, the only Git call besides the deliberate second lookup is the `git config --get --type=path core.hooksPath` check in `runtime_chain_body` (`bin/fm-git-strip-ai-trailers.sh:168`). That check prints the same error, so it was the one place to fix. I added `2>/dev/null` to it. Its exit status still decides the outcome: an empty value still runs no hook, and anything else goes on to the second lookup, which prints Git's error once, and the commit is refused. Tests (`tests/fm-git-strip-ai-trailers.test.sh`): - The unresolvable-path test (`~fm-no-such-user-6171/hooks`) now requires `failed to expand user dir` to appear exactly once in the refused commit's output. - The valueless-key test now requires `missing value for 'core.hookspath'` to appear exactly once. - Pre-existing bug in the unresolvable-path test: its `git add` ran after the bad config was set, so it failed silently (exit 128) and the "refused commit" had nothing staged. The test now stages the file before writing the config, the same way the valueless test does, so a real commit gets refused. - The empty-string test is unchanged and still passes, so an empty `core.hooksPath` still runs no hook. Verification: - With the wrapper change reverted, both new checks fail with `expected '1', got '2'`. With the change in place, the whole suite passes. - `bash -n` passes. shellcheck shows only an info-level SC1091 note about sourcing `lib.sh`, which was already there before this change. - `git status` lists only the two intended files --- bin/fm-git-strip-ai-trailers.sh | 16 +++++-- docs/configuration.md | 2 +- tests/fm-git-strip-ai-trailers.test.sh | 58 ++++++++++++++++++++++++++ 3 files changed, 71 insertions(+), 5 deletions(-) diff --git a/bin/fm-git-strip-ai-trailers.sh b/bin/fm-git-strip-ai-trailers.sh index 5469dfe5557..dfa01616177 100755 --- a/bin/fm-git-strip-ai-trailers.sh +++ b/bin/fm-git-strip-ai-trailers.sh @@ -19,8 +19,9 @@ # because git -c core.hooksPath= (or a child process that # inherits it) carries the override there, and a lookup that honored it # would find this directory again and never run the repository's own -# hook - a skipped pre-push guard. A lookup that fails exits nonzero -# rather than skipping the repository's hook. Does not touch the +# hook - a skipped pre-push guard. An empty core.hooksPath means no +# repository hook, as in plain git; any other failed lookup exits +# nonzero rather than skipping the repository's hook. Does not touch the # project's git config; the caller prefixes the pane with # GIT_CONFIG_COUNT / GIT_CONFIG_KEY_0 / GIT_CONFIG_VALUE_0. # @@ -153,14 +154,21 @@ write_executable() { # is the other environment channel that can carry this directory as # core.hooksPath; only the repository's config files name its own hooks. Skip # when the lookup still names this launch's own hooks dir, meaning those files -# point here, so the wrapper cannot recurse into itself. +# point here, so the wrapper cannot recurse into itself. An empty +# core.hooksPath makes that lookup fail, but plain git reads it as "no hooks", +# so the wrapper runs none; any other failure reruns the lookup to show git's +# error and refuses. runtime_chain_body() { local ours=$1 cat </dev/null) || { + if hooks_path=\$(unset GIT_CONFIG_PARAMETERS; git config --get --type=path core.hooksPath 2>/dev/null) && [ -z "\$hooks_path" ]; then + exit 0 + fi + (unset GIT_CONFIG_PARAMETERS; git rev-parse --path-format=absolute --git-path hooks >/dev/null) echo "fm-git-strip-ai-trailers: cannot resolve this repository's hooks directory; refusing to skip its \$name hook" >&2 exit 1 } diff --git a/docs/configuration.md b/docs/configuration.md index d06de2f5efa..af304a22495 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -983,7 +983,7 @@ The optional local, gitignored `config/keep-ai-trailers` presence flag opts this With the flag absent, every Claude launch's inline `--settings` JSON carries `"attribution":{"commit":"","pr":"","sessionUrl":false}`, every Devin worker config sets `"attribution": false`, and every fleet launch receives a pane-scoped `GIT_CONFIG` `core.hooksPath` pointing at `state/.git-hooks`, where git's `commit-msg` hook strips known AI trailers even when a runtime injects them after the typed message. When the flag is present, Claude launches omit those attribution-off settings, Devin worker configs keep the user config's `attribution` setting (Devin's default is on), and fleet launches do not install or select the strip hooks, so Git uses the repository's configured hooks directly. `bin/fm-git-strip-ai-trailers.sh` owns the identities, the install, and chaining the hooks of whichever repository git is running in, including when `git -c core.hooksPath` supplies the pane's hook override, so a project hook such as husky still runs when stripping is enabled. -If the wrapper cannot resolve that repository's hooks directory, the git operation fails rather than silently skipping a project hook such as a pre-push guard. +A repository whose config sets `core.hooksPath` to the empty string runs no project hook, as in plain git; if the wrapper otherwise cannot resolve that repository's hooks directory, the git operation fails rather than silently skipping a project hook such as a pre-push guard. When stripping is enabled, the hooks directory is read-only, so a hook manager run inside a fleet pane (lefthook's npm postinstall, `pre-commit install`) fails instead of displacing the strip; install a project's hooks from outside the pane, where the wrappers chain them. The flag is a home-wide attribution choice, so it is inherited into secondmate homes under the [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md) inherited-local-material contract and a secondmate's own workers keep AI trailers too. Per-machine Cursor `cli-config.json` attribution-off is not this contract: it does not travel with Firstmate, defaults back to on when unset, and only feeds the CLI's request to the server, so it suppresses the trailer rather than preventing it. diff --git a/tests/fm-git-strip-ai-trailers.test.sh b/tests/fm-git-strip-ai-trailers.test.sh index c12124194ba..2f66b1008fe 100644 --- a/tests/fm-git-strip-ai-trailers.test.sh +++ b/tests/fm-git-strip-ai-trailers.test.sh @@ -234,6 +234,61 @@ test_pane_hookspath_does_not_reroute_another_repository() { pass "a pane GIT_CONFIG hooksPath still chains the repository git is actually in" } +test_empty_project_hookspath_runs_no_repository_hook() { + local repo hooks err + repo="$TMP_ROOT/empty-hookspath" + make_repo "$repo" + write_marker_hook "$repo/.git/hooks/pre-commit" default-pre-commit + git -C "$repo" config core.hooksPath '' + hooks="$TMP_ROOT/hooks-empty" + "$STRIP" install "$hooks" "$repo" || fail "install should succeed with an empty core.hooksPath" + printf 'note\n' >>"$repo/README.md" + git -C "$repo" add README.md + err=$(with_hooks_env "$hooks" git -C "$repo" commit -q --trailer 'Co-authored-by: Cursor ' -m 'fix: empty hooksPath' 2>&1) || + fail "a commit in a repo with an empty core.hooksPath was refused: $err" + assert_equals "" "$err" "an empty core.hooksPath commit printed errors" + [ -f "$repo/default-pre-commit.ran" ] && fail "a repository hook ran although core.hooksPath is empty" + assert_not_contains "$(git -C "$repo" log -1 --format=%B)" "Co-authored-by: Cursor" \ + "Cursor trailer survived an empty-hooksPath commit" + pass "an empty project core.hooksPath runs no repository hook and still strips the trailer" +} + +test_unresolvable_project_hookspath_still_refuses() { + local repo hooks head err + repo="$TMP_ROOT/unresolvable-hookspath" + make_repo "$repo" + printf 'note\n' >>"$repo/README.md" + git -C "$repo" add README.md + git -C "$repo" config core.hooksPath '~fm-no-such-user-6171/hooks' + hooks="$TMP_ROOT/hooks-unresolvable" + "$STRIP" install "$hooks" "$repo" || fail "install should succeed with an unresolvable core.hooksPath" + head=$(git -C "$repo" rev-parse HEAD) + err=$(with_hooks_env "$hooks" git -C "$repo" commit -q -m 'fix: unresolvable hooksPath' 2>&1) && + fail "a commit succeeded although the repository's hooks directory cannot be resolved" + assert_contains "$err" "refusing to skip its pre-commit hook" "the refusal did not name the skipped hook" + assert_equals 1 "$(printf '%s\n' "$err" | grep -c 'failed to expand user dir')" "git's lookup error was not shown exactly once" + assert_equals "$head" "$(git -C "$repo" rev-parse HEAD)" "a refused commit still moved HEAD" + pass "an unresolvable project core.hooksPath still refuses the commit" +} + +test_valueless_project_hookspath_still_refuses() { + local repo hooks head err + repo="$TMP_ROOT/valueless-hookspath" + make_repo "$repo" + hooks="$TMP_ROOT/hooks-valueless" + "$STRIP" install "$hooks" "$repo" || fail "install should succeed before the valueless key is written" + head=$(git -C "$repo" rev-parse HEAD) + printf 'note\n' >>"$repo/README.md" + git -C "$repo" add README.md + printf '[core]\n\thooksPath\n' >>"$repo/.git/config" + err=$(with_hooks_env "$hooks" git -C "$repo" commit -q -m 'fix: valueless hooksPath' 2>&1) && + fail "a commit succeeded although core.hooksPath has no value" + assert_contains "$err" "refusing to skip its pre-commit hook" "the refusal did not name the skipped hook" + assert_equals 1 "$(printf '%s\n' "$err" | grep -c "missing value for 'core.hookspath'")" "git's lookup error was not shown exactly once" + assert_equals "$head" "$(git -C "$repo" -c core.hooksPath=x rev-parse HEAD)" "a refused commit still moved HEAD" + pass "a valueless project core.hooksPath still refuses the commit" +} + write_refusing_pre_push() { # cat >"$1" < Date: Wed, 30 Sep 2026 10:19:52 -0400 Subject: [PATCH 03/33] fix(bin): let a stale record on a reassigned slot retire records-only (#6213) * fix(bin): let a stale record on a reassigned slot retire records-only When a pool slot's owner claim names another task, the stale record's teardown touches nothing under the slot, so the exclusive-slot record scan no longer refuses it. Full teardowns of a slot this task still claims, or one with no claim, keep the refusal. Fixes #6184 * no-mistakes(document): Note claim-over-record precedence for reassigned teardown slots --- bin/fm-teardown.sh | 21 +++++++++---- docs/architecture.md | 2 +- tests/fm-teardown-endpoint-safety.test.sh | 38 +++++++++++++++++++++++ 3 files changed, 54 insertions(+), 7 deletions(-) diff --git a/bin/fm-teardown.sh b/bin/fm-teardown.sh index 10ee713e767..24ed4644c76 100755 --- a/bin/fm-teardown.sh +++ b/bin/fm-teardown.sh @@ -98,7 +98,10 @@ # cleanup step, teardown verifies record exclusivity: no OTHER task record in # this home or any locally registered Firstmate home may name the same live path # in its worktree= or home=. One live path with two task records is the reuse -# collision itself, whichever record is stale. +# collision itself, whichever record is stale. The one exception is a slot whose +# owner claim (below) names another task: this teardown is then records-only and +# touches nothing under the slot, so the scan is skipped rather than stranding +# the stale record and, with it, the claimant's own teardown. # That scan alone cannot prove THIS record is the current owner, because the task # that took the slot next may leave no record it can reach - its own worker may # have exited and its record been cleaned up, or it may live in a home this @@ -2357,6 +2360,12 @@ require_exclusive_worktree_slot_record() { local record_meta=$1 record_id=$2 record_state=$3 worktree=$4 local slot state_dir other other_id field other_path other_slot slot=$(canonical_existing_dir "$worktree") || return 0 + # A slot whose owner claim names another task was reassigned, so this record's + # teardown is records-only and touches nothing under it; another record naming + # the slot is then no hazard, and refusing would strand this stale record and + # block the claimant's own teardown behind it. + fm_treehouse_slot_owner_state "$slot" "$record_id" + [ "$FM_TREEHOUSE_SLOT_OWNER" != other ] || return 0 collect_local_firstmate_states "$record_state" || return 1 for state_dir in "${TREEHOUSE_OWNER_STATES[@]}"; do for other in "$state_dir"/*.meta; do @@ -2390,11 +2399,11 @@ require_exclusive_task_worktree_slot() { # Positive slot ownership, read from the claim the task that took the slot wrote # into the slot itself (bin/fm-wake-lib.sh owns the claim and its states). # -# The record scan above proves that no OTHER task record names this slot. It -# cannot prove that THIS record is not the stale one, because the task that took -# the slot next may leave no record this scan can reach: its own worker may have -# exited and its record been cleaned up, or it may belong to a home this machine -# does not register. The claim closes that gap from the other side - it names the +# For a slot this task still claims, or one with no claim, the record scan above +# proves that no OTHER task record names it. It cannot prove that THIS record is +# not the stale one, because the task that took the slot next may leave no record +# this scan can reach: its own worker may have exited and its record been cleaned +# up, or it may belong to a home this machine does not register. The claim closes that gap from the other side - it names the # task that actually took the slot, and it is written under the same project lock # that allocates it - so a claim naming another task is proof the slot was # reassigned after this record was written. diff --git a/docs/architecture.md b/docs/architecture.md index eafab774866..4b2b6f9cbfe 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -412,7 +412,7 @@ A later merged poll consumes only that matching persisted value; with no match i [`bin/fm-merge-authority-lib.sh`](../bin/fm-merge-authority-lib.sh)'s header owns resolution, private atomic persistence, identity-checked consumption, and retirement, while only the merge path gates on the answer. Teardown is fail-closed for ship worktrees: dirty worktrees refuse, and committed work must be landed before the worktree is returned. A pool worktree is only returned after teardown passes the slot-ownership proof: a contradictory task record or a supported live endpoint refuses without touching either task, and no discard authority relaxes that. -A slot's own owner claim, written by the spawn that takes it under the allocation lock and owned by [`bin/fm-wake-lib.sh`](../bin/fm-wake-lib.sh), covers a slot reassigned to a task that left no record the scan could reach: a claim naming a different task releases nothing - teardown warns, names the claimant, and finishes only the task's own cleanup - because Treehouse's own live process lease cannot answer ownership once the worker's exit releases it. +A slot's own owner claim, written by the spawn that takes it under the allocation lock and owned by [`bin/fm-wake-lib.sh`](../bin/fm-wake-lib.sh), covers a slot reassigned to another task, including one that left no record the scan could reach: a claim naming a different task releases nothing, even alongside that task's contradictory record - teardown warns, names the claimant, and finishes only the task's own cleanup - because Treehouse's own live process lease cannot answer ownership once the worker's exit releases it. Allocation and return serialize on one project lock per machine-local Firstmate tree: every home reachable through local parent links shares that lock, and a home seeded from another machine anchors its own, because a lock taken on this filesystem is neither held nor observable across that boundary. Before the worktree is returned, teardown concludes the task's own no-mistakes run when it is parked at a gate, including a run whose head the task copy cannot resolve - the shared runs-ledger continuation proof is the only recognition for that case, so cleanup never orphans a parked run the pipeline advanced past the submitted head. [`bin/fm-teardown.sh`](../bin/fm-teardown.sh)'s header owns the landed-work proofs, slot-ownership proof, endpoint-close refusal, PR-discovery fallback, pre-teardown run conclusion, and stale-lock recovery procedure; [`tests/fm-teardown-endpoint-safety.test.sh`](../tests/fm-teardown-endpoint-safety.test.sh) and [`tests/fm-secondmate-safety.test.sh`](../tests/fm-secondmate-safety.test.sh) pin the slot-collision boundary. diff --git a/tests/fm-teardown-endpoint-safety.test.sh b/tests/fm-teardown-endpoint-safety.test.sh index 01dc74239e6..7ee608d2024 100755 --- a/tests/fm-teardown-endpoint-safety.test.sh +++ b/tests/fm-teardown-endpoint-safety.test.sh @@ -983,6 +983,43 @@ test_reassigned_pool_slot_finishes_own_cleanup_without_touching_the_slot() { pass "fm-teardown: a pool slot claimed by another task is left alone while the task's own cleanup finishes" } +# The reuse collision where BOTH records survive: the stale task's record still +# names the slot the pool handed on, and the claimant's own record names it too. +# The claim proves the stale record's teardown is records-only, so the record +# scan must not refuse it; once it is gone, the claimant tears down normally. +test_stale_record_on_claimed_slot_retires_then_claimant_tears_down() { + local dir id=stale-task other=live-task rc + + dir=$(make_case slot-reassigned-both-records) + mark_case_as_treehouse_pool "$dir" + fm_write_meta "$dir/home/state/$id.meta" \ + "window=firstmate:fm-$id" "endpoint_task_id=$id" \ + "worktree=$dir/worktree" "project=$dir/project" "kind=scout" + fm_write_meta "$dir/home/state/$other.meta" \ + "window=firstmate:fm-$other" "endpoint_task_id=$other" \ + "worktree=$dir/worktree" "project=$dir/project" "kind=scout" + claim_pool_slot "$dir" "$other" + + set +e + run_case "$dir" "$id" > "$dir/stdout" 2> "$dir/stderr" + rc=$? + set -e + [ "$rc" -eq 0 ] || fail "records-only teardown of a stale record on a claimed slot failed: $(cat "$dir/stderr")" + assert_reassigned_slot_left_alone "$dir" "$id" "$other" "stale record beside the claimant's record" + assert_present "$dir/worktree/sentinel" "records-only teardown reset the claimant's slot" + assert_present "$dir/home/state/$other.meta" "records-only teardown removed the claimant's record" + + : > "$dir/runtime.log" + run_case "$dir" "$other" > "$dir/stdout" 2> "$dir/stderr" \ + || fail "claimant teardown failed after the stale record retired: $(cat "$dir/stderr")" + assert_absent "$dir/home/state/$other.meta" "claimant teardown left its record" + assert_absent "$dir/pool/1/.fm-slot-owner" "claimant teardown left its spent slot claim behind" + grep -Fq "treehouse " "$dir/runtime.log" \ + || fail "claimant teardown did not return its pool slot: $(cat "$dir/runtime.log")" + + pass "fm-teardown: a stale record on a claimed slot retires, then the claimant tears down" +} + # The two states that must never become a false refusal: the task's own claim, # and no claim at all (a slot taken before claims existed, or already returned). test_own_and_absent_slot_claims_still_tear_down() { @@ -1403,6 +1440,7 @@ test_reused_pool_slot_refuses_before_touching_the_other_task test_cross_home_pool_slot_collision_refuses test_sole_slot_record_still_tears_down test_reassigned_pool_slot_finishes_own_cleanup_without_touching_the_slot +test_stale_record_on_claimed_slot_retires_then_claimant_tears_down test_own_and_absent_slot_claims_still_tear_down test_recorded_endpoint_that_changed_directory_still_tears_down test_project_lock_anchors_at_the_local_root_across_home_layouts From 23e5584714e6765cc223a1740d385d0e85f8ad8e Mon Sep 17 00:00:00 2001 From: Christopher McKay <101884182+karotkriss@users.noreply.github.com> Date: Wed, 30 Sep 2026 14:21:20 -0400 Subject: [PATCH 04/33] fix(bin): keep the steering doorbell short under deep homes (#6240) * fix(bin): keep the steering doorbell short under deep homes The doorbell printed the task inbox's absolute path twice, so under a deep home it grew to about 290 characters and a Herdr submit reported it never reached the pane on every re-ring. It now names the inbox once by its short .inbox name and points at the full path the worker's brief already gives, so its length no longer depends on the home's depth. Fixes #6120 * no-mistakes(review): Export FM_TASK_INBOX at launch and name it in doorbell * no-mistakes(ci): ci-1 (Behavior portable serial 9) was caused by this PR, and I fixed it in the test. tests/fm-claude-trust.test.sh failed with "the launch command did not carry a brief doorbell". Its claude_launch_doorbell helper stripped exactly two leading `export ...;` statements before reading the final prompt argument. This PR adds a third one (`export FM_TASK_INBOX=...`) to every launch, so the helper was reading the wrong command. The invariant: a test that parses the launch command must skip every leading export statement, however many there are. I checked every test that parses the launch this way. The only other ones are the two helpers in tests/fm-spawn-dispatch-profile.test.sh, and they already loop over all exports. The kimi and dispatch-profile exact-string checks were updated earlier in this PR. The fix makes claude_launch_doorbell use the same loop (`while [[ "$command" == export\ *\;* ]]; do command=${command#*; }; done`) and then take the last argument. The ordinary path still works: the claude spawn test and the secondmate-clone spawn test both resolve the brief record through the same helper. Verified locally: `bash tests/fm-claude-trust.test.sh` exits 0 with no failing cases. ci-2 (Behavior tests (Herdr)) was not caused by this change, and I made no code change for it. In tests/fm-backend-herdr-presentation-e2e.test.sh, the concurrent secondmate recovery failed with "herdr presentation recovery could not acquire its session lock; refusing a concurrent resume". Two reasons it is not this PR: - The same failure, in the same test and case, happened on run 36655209015 for the unrelated branch fm/fm-contributions-old-gh-compat about 14 hours earlier. - This PR's change cannot lengthen how long the lock is held. The launch is written to a file and sent to the pane as `. launch.N.sh`, so the extra export changes neither the pane submit nor the lock hold time. The cause is a race that was already there: spawn_herdr_presentation_order_lock_acquire gives up after 5 seconds, and a concurrent real-Herdr recovery can hold the lock longer. Fixing that means changing the product's lock timeout, which is outside this PR. It should be tracked separately, and a rerun of the Herdr job is expected to pass. The only file changed is tests/fm-claude-trust.test.sh --- bin/fm-brief.sh | 7 +-- bin/fm-spawn.sh | 10 +++- bin/fm-task-inbox-lib.sh | 34 +++++++----- docs/configuration.md | 1 + docs/verification/runtime-backends.md | 18 +++++++ tests/fm-claude-trust.test.sh | 9 ++-- tests/fm-kimi-harness.test.sh | 10 +++- tests/fm-send-inbox-doorbell-live-e2e.test.sh | 21 ++++---- tests/fm-send-inbox.test.sh | 45 +++++++++++++++- .../fm-spawn-compact-adviser-disable.test.sh | 42 +++++++++++++-- tests/fm-spawn-dispatch-profile.test.sh | 10 +++- tests/fm-startup-memory-budget.test.sh | 2 +- tests/fm-task-inbox.test.sh | 54 ++++++++++--------- 13 files changed, 199 insertions(+), 64 deletions(-) diff --git a/bin/fm-brief.sh b/bin/fm-brief.sh index e3fe51fd2ae..864b2c857df 100755 --- a/bin/fm-brief.sh +++ b/bin/fm-brief.sh @@ -348,9 +348,10 @@ INBOX_DIR=$(shell_quote "$STATE/$ID.inbox") # The receive-and-ack half of the steering-inbox contract, included in every # scaffold kind. The record format, doorbell line, and re-ring ladder are -# owned by bin/fm-task-inbox-lib.sh; the doorbell itself is self-describing, -# so this section is reinforcement for the natural-checkpoint habit, not the -# only carrier of the instruction. +# owned by bin/fm-task-inbox-lib.sh. The doorbell names the inbox as +# "$FM_TASK_INBOX", which bin/fm-spawn.sh exports into every launch; the full +# path here remains the fallback for a worker launched without that export, +# plus the natural-checkpoint habit. IFS= read -r -d '' INBOX_SECTION <.inbox +# path the steering doorbell names). Raw commands must # be POSIX sh compatible under this opt-in; the absent-file path is unchanged. # This is an exec environment boundary, not a sandbox for the pane's startup # shell, credential files, same-user processes, or later shell initialization. @@ -5166,6 +5168,12 @@ fi if [ "$LAVISH_AXI_HOST_CONFIG_PRESENT" = 1 ]; then LAUNCH="export LAVISH_AXI_HOST=$(shell_quote "$LAVISH_AXI_HOST"); $LAUNCH" fi +# Every launch also exports the absolute path of this task's steering inbox, so +# the constant doorbell line (bin/fm-task-inbox-lib.sh) can name +# "$FM_TASK_INBOX" instead of a path that grows with the home's depth. Like the +# kill switch below it is an export statement, so it survives a compound raw +# launch and the launch-env-allowlist `env -i` wrapper. +LAUNCH="export FM_TASK_INBOX=$(shell_quote "$STATE_REAL/$ID.inbox"); $LAUNCH" LAUNCH="export COMPACT_ADVISER_DISABLE=1; $LAUNCH" # When the live-harness gate has exported DISABLE_AUTOUPDATER into this spawn's # own environment, carry it into the launch command text so Claude Code's diff --git a/bin/fm-task-inbox-lib.sh b/bin/fm-task-inbox-lib.sh index 27c3aeda623..c1d48689975 100644 --- a/bin/fm-task-inbox-lib.sh +++ b/bin/fm-task-inbox-lib.sh @@ -59,7 +59,7 @@ # crash or marker failure may produce a rare duplicate rather than silently lose # a wake. # -# Inbox paths containing bytes outside printable ASCII are unsupported. The +# Inbox names containing bytes outside printable ASCII are unsupported. The # doorbell refuses them rather than sending terminal control bytes to a pane. # # fm_task_inbox_ring requires bin/fm-backend.sh's dispatch (sourced below); the @@ -252,22 +252,30 @@ fm_task_inbox_body() { # } # The constant self-describing doorbell line for the inbox containing a record. -# Self-describing on purpose: a worker whose brief predates the inbox contract -# still receives the complete instruction in the line itself. The leading `: ` -# is the POSIX shell no-op, so the same line typed into a pane whose agent has -# exited (a bare shell) runs nothing; see the dead-pane note in the header. -# A non-printable path fails without output so terminal controls never reach -# the pane's line discipline. +# It names the inbox by the literal "$FM_TASK_INBOX", which bin/fm-spawn.sh +# exports into every launch as the inbox's absolute path, so the worker can +# resolve it from its own environment even after losing its brief context. +# The short `.inbox` name follows as the fallback for a worker launched +# before that export, whose brief carries the full path (bin/fm-dod-lib.sh +# role contract, bin/fm-brief.sh inbox section). No absolute path is printed, +# so the line's length never grows with the home's depth: a long line wraps +# past what a harness composer read can prove, and a Herdr submit then reports +# it did not reach the pane on every re-ring. The leading `: ` is the POSIX +# shell no-op, so the same line typed into a pane whose agent has exited (a +# bare shell) runs nothing; see the dead-pane note in the header. A +# non-printable inbox name fails without output so terminal controls never +# reach the pane's line discipline. fm_task_inbox_doorbell_line() { # - local dir=${1%/*} abs quoted LC_ALL=C + local dir=${1%/*} abs name quoted LC_ALL=C abs=$(cd "$dir" 2>/dev/null && pwd) || abs=$dir abs=${abs%/handled} - case "$abs" in - *[![:print:]]*) return 1 ;; + name=${abs##*/} + case "$name" in + ''|*[![:print:]]*) return 1 ;; esac - quoted=$(printf '%s' "$abs" | sed "s/'/'\\\\''/g") - printf ": Firstmate instruction waiting: list '%s'/*.msg and, in numeric order, read and act on each, then mv each handled file to '%s'/handled/." \ - "$quoted" "$quoted" + quoted=$(printf '%s' "$name" | sed "s/'/'\\\\''/g") + printf ": Firstmate instruction waiting: list \"\$FM_TASK_INBOX\"/*.msg in your '%s' steering inbox, read and act on each in numeric order, then mv each into its handled/." \ + "$quoted" } # Ring the doorbell, best-effort: one endpoint-liveness pre-check, one advisory diff --git a/docs/configuration.md b/docs/configuration.md index af304a22495..5dc5b92f47e 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -2254,6 +2254,7 @@ FM_PROC_ROOT_OVERRIDE= # alternate /proc root for Linux process-identity reads FM_BACKEND= # optional runtime backend override for new spawns; tmux/herdr/zellij/orca/cmux support ship/scout spawns, codex-app is not accepted FM_TRACE_CONTEXT= # optional trace-context override; see "Trace context propagation" FM_TASK_ID= # internal task-worker marker fm-spawn.sh exports into ship and scout panes, never set by hand; bin/fm-test-run.sh refuses to execute in the repository primary checkout while it is set +FM_TASK_INBOX= # internal: absolute path of the task's steering inbox (state/.inbox) that fm-spawn.sh exports into every ship, scout, and secondmate launch, never set by hand; the steering doorbell names "$FM_TASK_INBOX" HERDR_SESSION=default # herdr-only: named session for normal backend ops; not enough for destructive cleanup (docs/herdr-backend.md) FM_BACKEND_HERDR_SUBMIT_POLLS=6 # herdr-only: agent-state samples spread across each Enter attempt's budget when confirming a submit (docs/herdr-backend.md "Current transport behavior") FM_BACKEND_HERDR_SUBMIT_MIN_SLEEP=0.6 # herdr-only: minimum per-Enter confirmation budget before polling agent-state after an idle baseline diff --git a/docs/verification/runtime-backends.md b/docs/verification/runtime-backends.md index 807d1272052..28593006c4a 100644 --- a/docs/verification/runtime-backends.md +++ b/docs/verification/runtime-backends.md @@ -826,6 +826,24 @@ The current pending-composer ring contract is owned by `bin/fm-task-inbox-lib.sh Kimi was not installed on the verification machine; its receive path is the same one-line-plus-shell contract, and the portable ladder and enqueue regressions in `tests/fm-task-inbox.test.sh` and `tests/fm-send-inbox.test.sh` cover every harness-independent half. This guard is the refresh command after any harness upgrade; it spends a small number of real tokens per installed harness, reports an absent harness explicitly, and refuses a run that verified nothing. +The doorbell no longer prints the inbox's absolute path, so its length no longer grows with the home's depth. +It names the inbox as `"$FM_TASK_INBOX"`, which `bin/fm-spawn.sh` exports into every launch as the absolute `state/.inbox` path, followed by the short `.inbox` name; the brief's full path remains the fallback for a worker launched without that export. +The guard now launches each worker with `FM_TASK_INBOX` exported and no brief, so the worker must resolve the inbox from the doorbell and its environment alone. +It is the refresh command for that shape, which has not yet been recorded live here. +The run below, on 2026-09-30 on tmux 3.6, Linux (WSL2), with the same command, covered the earlier brief-primed shape, whose doorbell named only the short `.inbox` name and whose guard gave each worker the brief's steering-inbox sentence before the steer: + +```text +ok - claude (2.1.285 (Claude Code)): the doorbell reached a real worker, which acted and acked with the mv +ok - codex (codex-cli 0.157.0): the doorbell reached a real worker, which acted and acked with the mv +ok - opencode (1.18.33): the doorbell reached a real worker, which acted and acked with the mv +# harness absent, not verified here: grok +# harness absent, not verified here: kimi +# harness absent, not verified here: muse +``` + +OpenCode needed `FM_SEND_INBOX_LIVE_TIMEOUT=560` because its configured model was still mid-turn at the default 240 seconds. +Pi 0.87.1 was installed but not verified: its configured model returned an account error (`The 'gpt-5.6-sol' model is not supported when using Codex with a ChatGPT account`) before it read the inbox. + ## Gemini The Gemini crewmate adapter was verified on 2026-09-04 with gemini-cli 0.58.0 on Linux, Node v24.20.0, tmux 3.4. diff --git a/tests/fm-claude-trust.test.sh b/tests/fm-claude-trust.test.sh index cafcb327dfd..70a807c04fa 100755 --- a/tests/fm-claude-trust.test.sh +++ b/tests/fm-claude-trust.test.sh @@ -626,11 +626,14 @@ test_refused_spawn_leaves_no_task_state() { } # Resolve the final prompt argument using the same shell argument splitting the -# pane sees after the two leading export statements. +# pane sees after the leading export statements. claude_launch_doorbell() { # - local command=${1#*; } + local command=$1 + while [[ "$command" == export\ *\;* ]]; do + command=${command#*; } + done ( - eval "set -- ${command#*; }" + eval "set -- $command" printf '%s' "${!#}" ) } diff --git a/tests/fm-kimi-harness.test.sh b/tests/fm-kimi-harness.test.sh index 817bb3a1a92..8dd483c0bb5 100755 --- a/tests/fm-kimi-harness.test.sh +++ b/tests/fm-kimi-harness.test.sh @@ -23,6 +23,12 @@ PYTHON_BIN_DIR=$(dirname "$PYTHON_BIN") JQ_BIN=$(command -v jq) || fail "test needs jq" BASE_PATH=${FM_TEST_BASE_PATH:-$PYTHON_BIN_DIR:/usr/bin:/bin:/usr/sbin:/sbin} +task_inbox_export() { # + local state + state=$(CDPATH='' cd -- "$1/state" && pwd -P) || fail "cannot resolve state dir $1/state" + printf "export FM_TASK_INBOX='%s'; " "$state/$2.inbox" +} + ai_trailer_hooks_prefix() { # local state state=$(CDPATH='' cd -- "$1/state" && pwd -P) || fail "cannot resolve state dir $1/state" @@ -300,7 +306,7 @@ test_kimi_launch_then_send_is_verified() { assert_contains "$out" "spawned $id harness=kimi" "kimi spawn did not report success" launch=$(cat "$CASE_DIR/launch.log") - [ "$launch" = "export COMPACT_ADVISER_DISABLE=1; $(ai_trailer_hooks_prefix "$HOME_DIR" "$id")env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI '$FAKEBIN_DIR/kimi' --model 'kimi-code/k3' --auto" ] \ + [ "$launch" = "export COMPACT_ADVISER_DISABLE=1; $(task_inbox_export "$HOME_DIR" "$id")$(ai_trailer_hooks_prefix "$HOME_DIR" "$id")env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI '$FAKEBIN_DIR/kimi' --model 'kimi-code/k3' --auto" ] \ || fail "kimi launch did not use the absolute binary, model, and --auto only: $launch" assert_not_contains "$launch" "--effort" "kimi launch emitted a nonexistent effort flag" assert_not_contains "$launch" "turn-ended" "kimi launch embedded a turn-end path" @@ -676,7 +682,7 @@ test_kimi_falls_back_to_expanded_home_binary() { rc=$? expect_code 0 "$rc" "Kimi HOME fallback spawn should succeed" launch=$(cat "$CASE_DIR/launch.log") - [ "$launch" = "export COMPACT_ADVISER_DISABLE=1; $(ai_trailer_hooks_prefix "$HOME_DIR" "$id")env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI '$fallback' --auto" ] \ + [ "$launch" = "export COMPACT_ADVISER_DISABLE=1; $(task_inbox_export "$HOME_DIR" "$id")$(ai_trailer_hooks_prefix "$HOME_DIR" "$id")env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI '$fallback' --auto" ] \ || fail "Kimi fallback did not expand HOME into an absolute executable: $launch" pass "fm-spawn: Kimi fallback expands the active HOME" } diff --git a/tests/fm-send-inbox-doorbell-live-e2e.test.sh b/tests/fm-send-inbox-doorbell-live-e2e.test.sh index 36de34f56b4..0b1240ddddf 100644 --- a/tests/fm-send-inbox-doorbell-live-e2e.test.sh +++ b/tests/fm-send-inbox-doorbell-live-e2e.test.sh @@ -4,14 +4,17 @@ # # The steering inbox's one behavioral assumption is that a real worker agent # follows the constant self-describing doorbell line: list the inbox, read and -# act on its records in numeric order, then mv each into handled/. A stub can -# only confirm the assumption already -# written into the stub, so per .agents/skills/firstmate-coding-guidelines -# this is proven against every INSTALLED verified harness: each is launched -# idle in an isolated tmux server, steered through the REAL fm-send (durable -# record + doorbell), and must both ACT on the instruction (create a named -# file) and ACKNOWLEDGE it (the mv into handled/), failing loudly with the -# harness name and version. +# act on its records in numeric order, then mv each into handled/. The +# doorbell names the inbox as "$FM_TASK_INBOX", so each worker is launched the +# way bin/fm-spawn.sh launches it, with FM_TASK_INBOX exported to its home's +# state/.inbox, and receives no brief at all: it must resolve the inbox +# from the doorbell plus its own environment. A stub can only confirm the +# assumption already written into the stub, so per +# .agents/skills/firstmate-coding-guidelines this is proven against every +# INSTALLED verified harness: each is launched idle in an isolated tmux server, +# steered through the REAL fm-send (durable record + doorbell), and must both +# ACT on the instruction (create a named file) and ACKNOWLEDGE it (the mv into +# handled/), failing loudly with the harness name and version. # # Run explicitly with FM_SEND_INBOX_LIVE_E2E=1. This test spends a small # number of real model tokens per installed harness (one short turn each) - @@ -131,7 +134,7 @@ check_harness_doorbell() { # task="live-$name" acted="$LAB/acted-$name" tmux -L "$SOCKET" new-window -d -t "$SESSION:" -n "$win" -c "$ROOT" \ - -- bash -lc "$cmd" \ + -- bash -lc "export FM_TASK_INBOX=$(printf '%q' "$home/state/$task.inbox"); $cmd" \ || { FAILED=1; printf 'not ok - %s (%s): could not launch in the isolated tmux server\n' "$name" "$version" >&2; return 0; } wait_ready "$win"; ready_rc=$? if [ "$ready_rc" -eq 1 ]; then diff --git a/tests/fm-send-inbox.test.sh b/tests/fm-send-inbox.test.sh index b669a6d9860..c0c61f62ae5 100644 --- a/tests/fm-send-inbox.test.sh +++ b/tests/fm-send-inbox.test.sh @@ -7,6 +7,7 @@ # drive the real fm-send executable over a stubbed tmux and pin: # 1. The payload is durably recorded and never typed; only the doorbell # crosses the terminal, and the send exits 0 at enqueue. +# The doorbell names the inbox once and never grows with the home's depth. # 2. Multi-line steers are legal and round-trip byte-exact. # 3. A re-send enqueues a NEW sequence and still never retypes a payload, # so the terminal can never truncate, garble, or duplicate a steer. @@ -129,7 +130,7 @@ test_text_steer_rides_inbox() { body=$(record_body _ "$rec") [ "$body" = "please rebase onto main" ] || fail "the recorded body differs: $body" typed=$(cat "$dir/send.log") - assert_contains "$typed" "Firstmate instruction waiting: list '$dir/home/state/t1.inbox'/*.msg" \ + assert_contains "$typed" "Firstmate instruction waiting: list \"\$FM_TASK_INBOX\"/*.msg in your 't1.inbox' steering inbox" \ "the doorbell should direct the worker to drain the inbox" case "$typed" in *"please rebase onto main"*) fail "the payload must never be typed:"$'\n'"$typed" ;; @@ -137,6 +138,47 @@ test_text_steer_rides_inbox() { pass "fm-send inbox: the payload is recorded durably and only the doorbell is typed" } +# A home nested deep must not lengthen the doorbell: a long line wraps past +# what a composer read can prove, so a Herdr submit reports it never reached +# the pane and every re-ring fails the same way. +test_deep_home_doorbell_stays_short() { + local shallow deep home err rest typed shallow_typed found + shallow=$(setup_case shallow-home) + run_send "$shallow" "$shallow/send.err" -- t1 "please continue" || fail "the shallow-home send failed" + shallow_typed=$(cat "$shallow/send.log") + deep="$TMP_ROOT/deep-home" + home="$deep/one/two/three/four/five/six/seven/eight-secondmate-homes-nest-under-long-worktree-paths" + mkdir -p "$home/state" + make_stubs "$deep" >/dev/null + fm_write_meta "$home/state/t1.meta" "window=sess:fm-t1" "kind=ship" "harness=claude" + err="$deep/send.err" + env PATH="$deep/fakebin:$PATH" \ + FM_ROOT_OVERRIDE="$home" FM_HOME="$home" FM_SEND_LOG="$deep/send.log" \ + FM_SEND_SETTLE=0 "$SEND" t1 "please continue" >/dev/null 2>"$err" || + fail "the deep-home send failed: $(cat "$err")" + [ -f "$home/state/t1.inbox/001.msg" ] || fail "the deep-home steer was not durably recorded" + typed=$(cat "$deep/send.log") + [ "$typed" = "$shallow_typed" ] || + fail "the doorbell should not depend on the home's depth:"$'\n'"shallow: $shallow_typed"$'\n'"deep: $typed" + [ "${#typed}" -le 200 ] || fail "the doorbell should stay under 200 characters, got ${#typed}: $typed" + case "$typed" in + *"$deep"* | *"$TMP_ROOT"*) fail "the doorbell should not carry the home's absolute path: $typed" ;; + esac + rest=${typed#*t1.inbox} + [ "$rest" != "$typed" ] || fail "the doorbell should name the inbox: $typed" + case "$rest" in + *t1.inbox*) fail "the doorbell should name the inbox once: $typed" ;; + esac + found=$(cd / && FM_TASK_INBOX="$home/state/t1.inbox" bash -c 'ls "$FM_TASK_INBOX"/*.msg') || + fail "a shell with FM_TASK_INBOX exported could not list the deep inbox" + [ "$found" = "$home/state/t1.inbox/001.msg" ] || + fail "the doorbell's list instruction did not resolve the deep inbox from an unrelated cwd: $found" + (cd / && FM_TASK_INBOX="$home/state/t1.inbox" bash -c 'mv "$FM_TASK_INBOX"/001.msg "$FM_TASK_INBOX"/handled/') || + fail "the doorbell's mv instruction did not acknowledge through FM_TASK_INBOX" + [ -f "$home/state/t1.inbox/handled/001.msg" ] || fail "the acknowledged record did not land in handled/" + pass "fm-send inbox: a deep home rings the same short doorbell naming the inbox once" +} + test_multiline_steer_is_legal() { local dir err rc body dir=$(setup_case multiline) @@ -412,6 +454,7 @@ test_empty_message_refused() { } test_text_steer_rides_inbox +test_deep_home_doorbell_stays_short test_multiline_steer_is_legal test_resend_enqueues_new_sequence test_pending_composer_skips_ring_advisorily diff --git a/tests/fm-spawn-compact-adviser-disable.test.sh b/tests/fm-spawn-compact-adviser-disable.test.sh index d2713604caf..f9b7f7482d1 100755 --- a/tests/fm-spawn-compact-adviser-disable.test.sh +++ b/tests/fm-spawn-compact-adviser-disable.test.sh @@ -59,10 +59,10 @@ run_case_spawn() { # Replace the harness binary with a probe that reports the single environment # fact under test, so executing the emitted launch answers "what would the agent # have seen" rather than "what does the command text look like". -install_env_probe() { # - cat > "$1/$2" <<'SH' +install_env_probe() { # [variable] + cat > "$1/$2" < "$HOME_DIR/config/launch-env-allowlist" + if [ "$kind" = ship ]; then + out=$(run_case_spawn "$id" "$PROJ_DIR" --mode no-mistakes --yolo off) + else + sm="$CASE_DIR/secondmate-home" + mkdir -p "$sm/bin" "$sm/data" + printf '# Firstmate\n' > "$sm/AGENTS.md" + printf '%s\n' "$id" > "$sm/.fm-secondmate-home" + printf 'charter for %s\n' "$id" > "$sm/data/charter.md" + printf '%s\n' 'projects/' 'state/' 'data/' 'config/' '.no-mistakes/' > "$sm/.gitignore" + git -C "$sm" init -q -b main + out=$(run_case_spawn "$id" "$sm" --secondmate) + fi + status=$? + expect_code 0 "$status" "$kind spawn should succeed: $out" + install_env_probe "$FAKEBIN_DIR" codex FM_TASK_INBOX + seen=$(emitted_launch_env "$FAKEBIN_DIR" "$LAUNCH_LOG" "$PANE_LOG") \ + || fail "$kind: the emitted launch failed to run" + want="$(cd "$HOME_DIR/state" && pwd -P)/$id.inbox" + assert_equals "$want" "$seen" \ + "a $kind agent must start with FM_TASK_INBOX set to its absolute steering inbox" + done + pass "ship and secondmate launches export their absolute steering inbox as FM_TASK_INBOX" +} + # --- relaunch --------------------------------------------------------------- # # bin/fm-control.sh relaunch stops the agent and rebuilds the launch through @@ -351,5 +386,6 @@ test_ship_allowlist_absent test_ship_allowlist_enabled test_launch_command_carries_the_switch_without_the_pane_export test_secondmate_launch +test_launch_exports_task_inbox test_relaunch_rebuilds_the_switch test_raw_compound_launch_command_carries_the_switch diff --git a/tests/fm-spawn-dispatch-profile.test.sh b/tests/fm-spawn-dispatch-profile.test.sh index 56b7241d605..98c273950a2 100755 --- a/tests/fm-spawn-dispatch-profile.test.sh +++ b/tests/fm-spawn-dispatch-profile.test.sh @@ -87,6 +87,12 @@ make_seeded_secondmate_home() { git -C "$home" init -q -b main } +task_inbox_export() { # + local state + state=$(CDPATH='' cd -- "$1/state" && pwd -P) || fail "cannot resolve state dir $1/state" + printf "export FM_TASK_INBOX='%s'; " "$state/$2.inbox" +} + ai_trailer_hooks_prefix() { # local state state=$(CDPATH='' cd -- "$1/state" && pwd -P) || fail "cannot resolve state dir $1/state" @@ -476,7 +482,7 @@ test_active_dispatch_profile_allows_raw_launch_command() { # The unverified-adapter escape hatch is still an agent this fleet launched, # so it carries the compact-adviser floor and the AI-trailer strip; nothing # else may rewrite the captain's own command. - [ "$launch" = "export COMPACT_ADVISER_DISABLE=1; $(ai_trailer_hooks_prefix "$HOME_DIR" "$id")custom-agent --flag" ] || fail "raw launch command changed"$'\n'"actual: $launch" + [ "$launch" = "export COMPACT_ADVISER_DISABLE=1; $(task_inbox_export "$HOME_DIR" "$id")$(ai_trailer_hooks_prefix "$HOME_DIR" "$id")custom-agent --flag" ] || fail "raw launch command changed"$'\n'"actual: $launch" pass "active crew-dispatch profile allows the raw launch-command escape hatch" } @@ -1682,7 +1688,7 @@ claude_expected_launch() { # [ "$(printf '%s' "$doorbell" | "$ROOT/bin/fm-operational-input.sh" doorbell-kind)" = launch-brief ] \ || doorbell="not a launch-brief doorbell" quoted="'$(printf '%s' "$doorbell" | sed "s/'/'\\\\''/g")'" - printf '%s' "export COMPACT_ADVISER_DISABLE=1; $(ai_trailer_hooks_prefix "$2" "$3")env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false CLAUDE_CODE_SEND_FEEDBACK=0 claude $4 $(claude_worker_add_dirs "$2" "$3")--settings '{\"feedbackDrafts\":\"off\",\"attribution\":{\"commit\":\"\",\"pr\":\"\",\"sessionUrl\":false}}' $CLAUDE_CONTROL_CHANNEL_FLAG $quoted" + printf '%s' "export COMPACT_ADVISER_DISABLE=1; $(task_inbox_export "$2" "$3")$(ai_trailer_hooks_prefix "$2" "$3")env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false CLAUDE_CODE_SEND_FEEDBACK=0 claude $4 $(claude_worker_add_dirs "$2" "$3")--settings '{\"feedbackDrafts\":\"off\",\"attribution\":{\"commit\":\"\",\"pr\":\"\",\"sessionUrl\":false}}' $CLAUDE_CONTROL_CHANNEL_FLAG $quoted" } test_claude_permission_mode_bypass_matches_absent_launch() { diff --git a/tests/fm-startup-memory-budget.test.sh b/tests/fm-startup-memory-budget.test.sh index 6e53816c024..fa4a6daca22 100755 --- a/tests/fm-startup-memory-budget.test.sh +++ b/tests/fm-startup-memory-budget.test.sh @@ -280,7 +280,7 @@ test_primary_budget_converges_with_exact_reread_and_safe_failures() { "budget propagation did not enqueue the pointer to its exact reread generation" assert_contains "$(<"$log")" "Firstmate instruction waiting: list " \ "budget propagation did not ring the durable inbox doorbell" - assert_contains "$(<"$log")" "/state/sm.inbox'/*.msg" \ + assert_contains "$(<"$log")" "'sm.inbox' steering inbox" \ "budget propagation doorbell did not identify the durable inbox" outside="$world/unsafe-budget" diff --git a/tests/fm-task-inbox.test.sh b/tests/fm-task-inbox.test.sh index eda6f37170c..7a188c76422 100644 --- a/tests/fm-task-inbox.test.sh +++ b/tests/fm-task-inbox.test.sh @@ -162,9 +162,10 @@ test_write_is_durable_and_exact() { doorbell2=$(inbox_lib "$state" fm_task_inbox_doorbell_line "$rec2") [ "$doorbell" = "$doorbell2" ] \ || fail "every record in one inbox should ring the same drain-all doorbell" - assert_contains "$doorbell" "'$state/t1.inbox'/*.msg" "doorbell should quote and name all unhandled records" + assert_contains "$doorbell" "list \"\$FM_TASK_INBOX\"/*.msg" "doorbell should list all unhandled records through FM_TASK_INBOX" + assert_contains "$doorbell" "'t1.inbox' steering inbox" "doorbell should quote and name the inbox" assert_contains "$doorbell" "numeric order" "doorbell should require ordered processing" - assert_contains "$doorbell" "'$state/t1.inbox'/handled/" "doorbell should quote and name the handled dir" + assert_contains "$doorbell" "handled/" "doorbell should name the handled dir" assert_contains "$doorbell" "Firstmate instruction waiting" "doorbell should be self-describing" case "$doorbell" in *$'\n'*) fail "the doorbell must be a single line" ;; @@ -181,69 +182,70 @@ test_write_is_durable_and_exact() { # command line. Execute the real line in real shells and assert it is inert: # exit 0, no output, and nothing in the inbox touched. test_doorbell_is_a_shell_noop() { - local state rec doorbell sh out before after marker - state="$TMP_ROOT/noop/x; touch marker; #'s space/state" + local state task rec doorbell sh out before after marker + state="$TMP_ROOT/noop/state" + task="x; touch marker; #'s space" marker="$state/marker" mkdir -p "$state" - rec=$(inbox_lib "$state" fm_task_inbox_write "$state" t1 "please continue") + rec=$(inbox_lib "$state" fm_task_inbox_write "$state" "$task" "please continue") doorbell=$(inbox_lib "$state" fm_task_inbox_doorbell_line "$rec") case "$doorbell" in ': '*) ;; *) fail "the doorbell must start with the shell no-op prefix, got: $doorbell" ;; esac - assert_contains "$doorbell" "'\\''s space/state/t1.inbox'" \ - "the doorbell should escape an embedded single quote in its quoted path" - before=$(ls -R "$state/t1.inbox") + assert_contains "$doorbell" "'\\''s space.inbox'" \ + "the doorbell should escape an embedded single quote in its quoted inbox name" + before=$(ls -R "$state/$task.inbox") for sh in sh bash zsh; do command -v "$sh" >/dev/null 2>&1 || continue - out=$(cd "$state" && "$sh" -c "$doorbell" 2>&1) \ + out=$(cd "$state" && FM_TASK_INBOX="$state/$task.inbox" "$sh" -c "$doorbell" 2>&1) \ || fail "$sh executed the hostile-path doorbell with a non-zero status: $out" [ -z "$out" ] || fail "$sh produced output while executing the hostile-path doorbell: $out" - [ ! -e "$marker" ] || fail "$sh executed shell syntax embedded in the inbox path" + [ ! -e "$marker" ] || fail "$sh executed shell syntax embedded in the inbox name" done # An interactive-style zsh with the line fed on stdin, the closest portable # stand-in for a dead pane's login shell reading typed keystrokes. if command -v zsh >/dev/null 2>&1; then - out=$(cd "$state" && printf '%s\n' "$doorbell" | zsh -s 2>&1) \ + out=$(cd "$state" && printf '%s\n' "$doorbell" | FM_TASK_INBOX="$state/$task.inbox" zsh -s 2>&1) \ || fail "zsh reading the hostile-path doorbell from stdin failed: $out" [ -z "$out" ] || fail "zsh printed while reading the hostile-path doorbell: $out" [ ! -e "$marker" ] || fail "zsh executed shell syntax from the stdin doorbell" fi - after=$(ls -R "$state/t1.inbox") + after=$(ls -R "$state/$task.inbox") [ "$before" = "$after" ] || fail "executing the doorbell changed the inbox:"$'\n'"$after" [ -f "$rec" ] || fail "executing the doorbell removed the unhandled record" - pass "inbox: a hostile-path doorbell executes as a no-op in bare shells" + pass "inbox: a hostile-name doorbell executes as a no-op in bare shells" } test_doorbell_rejects_terminal_controls() { - local dir state rec doorbell control label log marker rc + local dir state task rec doorbell control label log marker rc dir="$TMP_ROOT/control-path" + state="$dir/state" marker="$dir/marker" - mkdir -p "$dir" + mkdir -p "$state" make_watch_stubs "$dir" >/dev/null for label in etx esc; do case "$label" in etx) control=$'\003' ;; esc) control=$'\033' ;; esac - state="$dir/${control}touch marker; # $label/state" - mkdir -p "$state" - rec=$(inbox_lib "$state" fm_task_inbox_write "$state" t1 "please continue") + task="${control}touch marker; # $label" + rec=$(inbox_lib "$state" fm_task_inbox_write "$state" "$task" "please continue") doorbell= rc=0 doorbell=$(inbox_lib "$state" fm_task_inbox_doorbell_line "$rec") || rc=$? - [ "$rc" -ne 0 ] || fail "a $label path should make doorbell construction fail" - [ -z "$doorbell" ] || fail "a rejected $label path emitted doorbell bytes" + [ "$rc" -ne 0 ] || fail "a $label inbox name should make doorbell construction fail" + [ -z "$doorbell" ] || fail "a rejected $label inbox name emitted doorbell bytes" log="$dir/$label.send.log"; : > "$log" rc=0 PATH="$dir/fakebin:$PATH" FM_SEND_LOG="$log" \ inbox_lib "$state" fm_task_inbox_ring tmux sess:fm-t1 "$rec" fm-t1 || rc=$? - [ "$rc" = 2 ] || fail "a rejected $label path should return send-failed status 2, got $rc" - [ ! -s "$log" ] || fail "a $label path reached send-keys:"$'\n'"$(cat "$log")" - [ ! -e "$marker" ] || fail "a $label path executed its crafted command" - [ -f "$rec" ] || fail "rejecting a $label path removed the durable record" + [ "$rc" = 2 ] || fail "a rejected $label inbox name should return send-failed status 2, got $rc" + [ ! -s "$log" ] || fail "a $label inbox name reached send-keys:"$'\n'"$(cat "$log")" + [ ! -e "$marker" ] || fail "a $label inbox name executed its crafted command" + [ -f "$rec" ] || fail "rejecting a $label inbox name removed the durable record" done - pass "inbox: terminal-control paths are rejected without typing" + pass "inbox: terminal-control inbox names are rejected without typing" } # fm_task_inbox_ring against a backend whose agent classifies dead or missing: @@ -616,7 +618,7 @@ test_watcher_rerings_idle_pane_quietly() { sleep 0.1 i=$((i + 1)) done - grep -qF "Firstmate instruction waiting: list '$state/t1.inbox'/*.msg" "$log" \ + grep -qF "Firstmate instruction waiting: list \"\$FM_TASK_INBOX\"/*.msg in your 't1.inbox' steering inbox" "$log" \ || { kill "$pid" 2>/dev/null; fail "the watcher never re-rang the doorbell:"$'\n'"$(cat "$log")"; } kill -0 "$pid" 2>/dev/null \ || fail "a healthy re-ring must not wake firstmate (watcher exited):"$'\n'"$(cat "$out")" From fd325b1ba16b0caa2c19fa0992b130bacba0dca8 Mon Sep 17 00:00:00 2001 From: Tiago Date: Wed, 30 Sep 2026 19:20:11 -0300 Subject: [PATCH 05/33] feat(bin): add opt-in config/wait-no-turns so a waiting worker spends no turns (#4859) * fix(dod): drive no-mistakes with one foreground call, not a background poll The brief told workers to background the drive call and poll `axi status` because one call "routinely outlives what your harness lets a single command run". That advice contradicts the tool it drives: `no-mistakes axi run --help` documents `--wait` with an 8m default, existing precisely "so an agent harness with a 10-minute tool cap gets a structured return instead of an unbounded hang". Following the old text, a worker could never idle - a backgrounded call returns in milliseconds, so it does not wait at all - and each attempt leaked a live timer that later fired as a paid wake. Tell workers to make one foreground call, let it block, and repeat it when it returns on elapsed wait rather than on a gate or outcome. Also drops the generalisation that told workers on any unestablished harness to assume a command cap and use the same shape, which exported the defect to harnesses with no such cap. * fix(bin): let a waiting worker spend no turns until it is answered A worker waiting on a decision, a pipeline gate, CI, or a heavy-test slot kept taking model turns: the brief told it to list its inbox at any natural checkpoint, and six automatic senders nudged secondmates whatever their open decisions. - The ship and scout briefs gain one Waiting section: end the turn after needs-decision or blocked, and hold an external wait inside ONE blocking command bounded by the harness's own command ceiling. The checkpoint clause is deleted. Forbidding the wrong shapes is not enough on its own, so the section also names the blocking foreground `until` loop as the wait a Claude Code worker may use, because that harness can refuse a sleep-then-check command while pointing at backgrounding, which is the one shape a waiting worker must not take. - fm-send --automatic defers (exit 4, nothing written or rung) while the target has an open decision or blocker of its own; every automatic sender passes it and keeps its retry state, and the pending-reply recovery waits the same way. - The two senders that report the result classified it by matching the text of the send's captured output against `deferred:*`. fm-send runs bin/fm-guard.sh as a supervision warning, and that guard prints its worktree-tangle banner whenever the primary checkout is on a feature branch, which is exactly what a CI pull-request checkout is. The banner lands ahead of the `deferred:` line, so the match fell through and a waiting mate was reported as a failed send, with the banner as the reason. Both senders now classify on fm-send's exit status, which is the contract the deferral is actually stated in, and select the `deferred:` line out of the output rather than assuming it came first. The third root cause, a no-mistakes definition of done that backgrounded the drive call and polled axi status, is fixed by this branch's parent commit "drive no-mistakes with one foreground call, not a background poll"; this commit takes that text as is and adds the regression test. Upstream's spawn abort path no longer calls the lease-return helper at all, so the fork's missing-helper guard and its pin-feature test line are moot here and are not ported. The command ceilings each harness enforces, and the probes behind the named Claude Code wait, are recorded in docs/verification/runtime-backends.md. * no-mistakes(review): Exempt captain holds, quiet deferred reconcile, clarify worker pauses * no-mistakes(document): Document deferred automatic nudges, rereads, and reply recovery * no-mistakes(document): Ring unlanded fire-and-forget steers exactly once more * no-mistakes(ci): The failing check, "PR must be raised via no-mistakes", reads the pipeline's attestation record, which says document=skipped. No file in the repository can change that record, so I did not touch the check or the PR body. As you said, the no-mistakes rerun after this run finishes will re-execute the document step and record document=completed. The one change is the documentation sentence you ordered. It adds a line to docs/remote-secondmates.md, right after the line saying the remote host runs no re-ring ladder of its own: "A fire-and-forget record, such as a reconcile ask, gets its single retry ring only on the local plane: the remote steer leg owes no re-ring, so a swallowed remote doorbell for one waits for the next ring into that inbox, and a remote-side retry is known follow-up scope." No behavior changed. Checks: tests/fm-documentation-audiences.test.sh passes (4/4) and bin/fm-lint.sh is clean. The change is left uncommitted in the working tree for the pipeline to pick up * no-mistakes(review): Hold automatic wakes until a mate's own decision closes * no-mistakes(document): Document watcher delivery of deferred remote re-read nudges * no-mistakes(review): Merge duplicate elapsed-wait reattach instructions in DOD * no-mistakes(test): Resolve merged default decision in remote-reply recovery fixture * no-mistakes(test): Source classify lib so config-push retry-deferred honors open decisions * no-mistakes(ci): Fixed a flaky test that also fails on main. Neither this PR's bin/fm-brief.sh nor its bin/fm-dod-lib.sh change is involved: bin/fm-dispatch-resolve.sh sources neither file. Another branch (fm-attended-cutover-smoothing-s1, run 36343879084) failed the same shard 8 check the same way, on a different case ("a rule-criterion match prints one diagnostic line, got 2"). Root cause: `fm_quota_single_provider_for_harness` in bin/fm-quota-axi-lib.sh returned from its `while read` loop as soon as it found a match. That closed the pipe while `fm_quota_single_provider_table`'s `printf` was sometimes still writing. GitHub Actions runners ignore SIGPIPE, so bash printed `fm-quota-axi-lib.sh: line 138: printf: write error: Broken pipe` to the resolver's stderr. That is the extra line. I reproduced it locally by running the test with SIGPIPE ignored: 2 of 20 runs failed, one with the resolver's diagnostic line plus two broken-pipe lines. Invariant: looking up a harness in the provider table must never make the table writer fail. The only reader of that table is this function, and all of the resolver's lookups (line 208 without stderr redirected, line 222 with it) go through it. So the fix is in that one place: read the whole table, then print the match. The same file now shows it reads the full table first, like `fm_control_harness_supported` does. Return values and output are unchanged. Verification: with SIGPIPE ignored, tests/fm-dispatch-resolve.test.sh failed 0 of 30 runs after the fix (2 of 20 before). tests/fm-dispatch-resolve.test.sh, tests/fm-brief.test.sh, tests/fm-send-inbox.test.sh, tests/fm-quota-choose.test.sh and tests/fm-quota-array-dispatch-live-e2e.test.sh all pass, and shellcheck is clean. tests/fm-procevent-quota.test.sh fails locally with or without the change ("process-event state root is not a private directory"), so that failure comes from the local environment, not from this fix. No new test was added: the existing one-diagnostic-line assertions already catch this whenever SIGPIPE is ignored, as it is in CI * Revert "no-mistakes(ci): Fixed a flaky test that also fails on main. Neither this PR's bin/fm-brief.sh nor its bin/fm-dod-lib.sh change is involved: bin/fm-dispatch-resolve.sh sources neither file. Another branch (fm-attended-cutover-smoothing-s1, run 36343879084) failed the same shard 8 check the same way, on a different case ("a rule-criterion match prints one diagnostic line, got 2"). Root cause: `fm_quota_single_provider_for_harness` in bin/fm-quota-axi-lib.sh returned from its `while read` loop as soon as it found a match. That closed the pipe while `fm_quota_single_provider_table`'s `printf` was sometimes still writing. GitHub Actions runners ignore SIGPIPE, so bash printed `fm-quota-axi-lib.sh: line 138: printf: write error: Broken pipe` to the resolver's stderr. That is the extra line. I reproduced it locally by running the test with SIGPIPE ignored: 2 of 20 runs failed, one with the resolver's diagnostic line plus two broken-pipe lines. Invariant: looking up a harness in the provider table must never make the table writer fail. The only reader of that table is this function, and all of the resolver's lookups (line 208 without stderr redirected, line 222 with it) go through it. So the fix is in that one place: read the whole table, then print the match. The same file now shows it reads the full table first, like `fm_control_harness_supported` does. Return values and output are unchanged. Verification: with SIGPIPE ignored, tests/fm-dispatch-resolve.test.sh failed 0 of 30 runs after the fix (2 of 20 before). tests/fm-dispatch-resolve.test.sh, tests/fm-brief.test.sh, tests/fm-send-inbox.test.sh, tests/fm-quota-choose.test.sh and tests/fm-quota-array-dispatch-live-e2e.test.sh all pass, and shellcheck is clean. tests/fm-procevent-quota.test.sh fails locally with or without the change ("process-event state root is not a private directory"), so that failure comes from the local environment, not from this fix. No new test was added: the existing one-diagnostic-line assertions already catch this whenever SIGPIPE is ignored, as it is in CI" This reverts commit c7199284297ea278c8da7d0a698cc5823ac37cec. * no-mistakes(review): Retry deferred local instruction nudges via the watcher * no-mistakes(review): Document watcher retry for deferred local instruction nudges * no-mistakes(ci): I fixed both review findings you selected (ci-1 and ci-3). I did not touch the deferral check in bin/fm-send.sh. ci-1 (bin/fm-config-push.sh, retry_deferred_rereads) - Rule that must hold: a deferred reread stays flagged until it is actually delivered. - Before the fix, the flag was removed before any of the steps that can skip a mate: the remote lock-path lookup, validate_secondmate_home, the local lock-path lookup, and the lock acquire. A skip at any of those dropped the flag, so the watcher lost track of the reread. - Now the flag is removed in one place only, when the send succeeds (rc 0). A skipped home, a busy lock, a deferred send (rc 4) or a failed send all leave it in place. The re-mark calls on a busy lock and on rc 4 were no longer needed, so I removed them. I updated the comment above the function to match. - Side effect: a send that keeps failing now stays flagged, so the watcher retries it on every poll and logs each failure. That follows your "don't clear until delivered" rule, but it replaces the old behaviour of leaving a failed send to the next config push or session start. - New test in tests/fm-secondmate-sync.test.sh: T8j "a deferred flag survives a skipped invalid home and is retried once it validates". It takes the home's marker away to make validation fail, checks that nothing is sent and the flag stays, then puts the marker back and checks that the nudge is delivered and both the flag and the retry marker are cleared. It fails on the old code and passes now. ci-3 (bin/fm-secondmate-restart.sh) - Rule that must hold: no automatic send wakes a mate that is waiting on its own open decision. - The two automatic sends in this script are the fallback reread nudge (fall_back_to_nudge) and the persist request. Both now pass --automatic. If a persist request is deferred, its correlation is discarded and the mate goes to the fallback nudge, which is also deferred, so the mate is reported as unreached. - New test in tests/fm-secondmate-restart.test.sh: T3b. It gives a mate an open needs-decision and runs a restart. It checks that both sends report as deferred, the mate's doorbell is never rung, its inbox gets no message, nothing is stopped, and the mate is reported as unreached with exit status 3. It fails on the old code and passes now. - The test marks the watcher as alive first. Without that, the watcher-down warning is printed first and becomes the reported reason instead of the deferral message. Verification - tests/fm-secondmate-sync.test.sh passes. - tests/fm-secondmate-restart.test.sh passes. - tests/fm-secondmate-harness.test.sh (the other test that exercises --retry-deferred) passes. - The fm-send-inbox test that covers automatic deferral passes. I only looked at the last lines of that run, not the whole file. - `shellcheck -x` on the four changed files is clean * Pin autoarm supervision model in secondmate restart T3b The fresh watcher beat the test writes proves a live watcher only under the autoarm model; on CI hosts with no detected harness the persistent model demands a lock-holding watcher, so the watcher-down banner became the reported reason and the deferral assertion failed. Co-Authored-By: Claude Opus 5.5 * Keep deferred secondmate nudges retryable under the inheritance lock. A bootstrap instruction nudge could write its deferral flag outside the lock the watcher retry holds, so a concurrent retry could delete a flag that had just been set. A restart fallback that is deferred now records the same marker and flag, so the watcher delivers it once the decision closes. * no-mistakes(document): Document watcher retry of deferred restart re-read nudges * Send secondmate reread and restart nudges immediately again. Deferring those nudges let a later config push drop an incomplete transfer once the decision closed. They now send as they do on main. * Make the no-turn wait opt-in behind config/wait-no-turns. Homes that do not create the file keep the previous briefs, drive text, and sends. * no-mistakes(document): Document wait-no-turns inbox wording change in configuration * no-mistakes(review): Keep checkpoint inbox check; forbid only polling while waiting * no-mistakes(ci): Fixed ci-2 (Greptile: a concurrent retry marker gets lost). The rule that was broken: the watcher may remove only the `.retry-ring` mark for the record it just processed. A newer mark written in the meantime is owed its own retry. `fm_task_inbox_clear_retry` is the one shared function that removes the mark, and I fixed it there. In `bin/fm-task-inbox-lib.sh` it now takes the record path. It compares the mark's content with that record's name and removes the mark only when they match. When the mark names a different record it returns success and leaves the mark alone. It still fails only when the processed record's own mark can't be removed. Both callers in `bin/fm-watch.sh` now pass `"$rec"`: the dead or missing pane path and the path after a retry ring. So the fix holds at both removal sites. Tests, in `tests/fm-task-inbox.test.sh`: - I added an optional `FM_RING_MARKS_RETRY` hook to the fake tmux. It writes a newer record's mark while the doorbell is being typed, which reproduces the race deterministically. - I added `test_watcher_retry_keeps_a_newer_mark`. The owed retry rings once, the newer mark survives, and a later check rings the newer record once and then clears its mark. The test fails without the fix ("the spent retry removed a newer record's mark written during its ring") and passes with it. - I updated the direct `clear_retry` call in the existing unit test to pass the record. Results: `tests/fm-task-inbox.test.sh` passes in full and `tests/fm-send-inbox.test.sh` passes 15/15. Shellcheck reports only SC1091 "not following sourced file" notices. As instructed, I didn't change the brief inbox wording * no-mistakes(document): Fix stale wait-no-turns inbox wording in inbox lib comment --------- Co-authored-by: Claude Opus 5.5 Co-authored-by: Cursor Agent Co-authored-by: Kun Chen --- .../skills/operational-home-layout/SKILL.md | 1 + bin/fm-brief.sh | 38 ++++- bin/fm-classify-lib.sh | 19 +++ bin/fm-dod-lib.sh | 24 ++- bin/fm-pending-reply-lib.sh | 11 +- bin/fm-send.sh | 21 ++- bin/fm-task-inbox-lib.sh | 46 +++++ bin/fm-watch.sh | 24 ++- docs/configuration.md | 7 + docs/remote-secondmates.md | 1 + docs/verification/runtime-backends.md | 37 ++++ tests/fm-brief.test.sh | 75 +++++++- tests/fm-pending-reply.test.sh | 71 ++++++++ tests/fm-remote-reply.test.sh | 3 + tests/fm-send-inbox.test.sh | 55 +++++- tests/fm-task-inbox.test.sh | 160 ++++++++++++++++++ 16 files changed, 573 insertions(+), 20 deletions(-) diff --git a/.agents/skills/operational-home-layout/SKILL.md b/.agents/skills/operational-home-layout/SKILL.md index bf17f233ea3..f51340058bb 100644 --- a/.agents/skills/operational-home-layout/SKILL.md +++ b/.agents/skills/operational-home-layout/SKILL.md @@ -39,6 +39,7 @@ config/trace-context optional presence flag enabling default-off native W3C tra config/lavish-axi-host optional one-line per-machine Lavish server address; LOCAL, gitignored, inherited by secondmate homes, and exported into every worker launch; see docs/configuration.md "Lavish server address" for opening versus polling config/brief-include.md optional standing worker instructions appended verbatim as the last section of every ship and scout scaffold; LOCAL, gitignored, and not inherited; keep its text out of `## Firstmate spec`; see docs/configuration.md "Home brief include" config/fleet-ledger optional presence flag opting this home in to the default-off fleet activity ledger state/fleet-ledger.jsonl that outside tools can follow; LOCAL, gitignored, and not inherited; see docs/fleet-ledger.md +config/wait-no-turns optional presence flag opting this home into default-off waiting-worker behavior (brief waiting section, foreground pipeline drive, pending-reply hold, one fire-and-forget retry ring); LOCAL, gitignored, and not inherited; see docs/configuration.md "Waiting worker spends no turns" config/turnend-churn-absorb optional presence flag opting this home into the default-off absorb of bare turn-end wakes on pane churn; LOCAL, gitignored, and not inherited; see docs/configuration.md "Turn-end pane-churn absorb" config/wedge-defer-parked-gate optional presence flag opting this home into the default-off deferral of a wedge escalation for a lane parked at a validation gate awaiting the supervisor's own still-open decision; LOCAL, gitignored, and not inherited; see docs/configuration.md "Parked-gate wait deferral" config/cmux-socket-password optional cmux control-socket password; LOCAL, gitignored; read fresh on every cmux CLI call and passed through without ever overriding an operator's own ambient CMUX_SOCKET_PASSWORD when absent (docs/cmux-backend.md "Setup") diff --git a/bin/fm-brief.sh b/bin/fm-brief.sh index 864b2c857df..2ff9c727762 100755 --- a/bin/fm-brief.sh +++ b/bin/fm-brief.sh @@ -350,16 +350,46 @@ INBOX_DIR=$(shell_quote "$STATE/$ID.inbox") # scaffold kind. The record format, doorbell line, and re-ring ladder are # owned by bin/fm-task-inbox-lib.sh. The doorbell names the inbox as # "$FM_TASK_INBOX", which bin/fm-spawn.sh exports into every launch; the full -# path here remains the fallback for a worker launched without that export, -# plus the natural-checkpoint habit. +# path here remains the fallback for a worker launched without that export. +# The doorbell itself is self-describing, so this section is reinforcement +# for the natural-checkpoint habit, not the only carrier of the instruction. +# config/wait-no-turns (docs/configuration.md) adds the line that a waiting +# worker does not poll the inbox: checkpoint checks happen during active work, +# so waiting still spends no turns. IFS= read -r -d '' INBOX_SECTION < --watch`, or `until ; do sleep 30; done` for anything else. +Never spend turns on `sleep` followed by a status check, and never background a command in order to poll it. +In Claude Code that `until` loop in a single Bash call is the sanctioned foreground wait: when the harness refuses a sleep-then-check command and points you at backgrounding instead, reissue the wait as the loop rather than accepting the background. +Bound that command by what your harness lets one command run: in Pi pass the bash tool a `timeout` of at most 2700 seconds, because Pi sets none by default; in Claude Code pass the Bash tool its maximum `timeout` of 600000 ms, because its default is 2 minutes; in Codex keep waiting on a still-running command with empty `write_stdin` polls of up to 300000 ms; elsewhere pass your shell tool its largest timeout and assume at most 10 minutes. +Give any `--wait` a duration a little under that bound. +When the bound passes with nothing changed, run the same blocking command again, with no status check in between. +The one exception is `respond`: it sent its answer before it began waiting, so reattach with `no-mistakes axi run --wait` instead, and never send the same `respond` again, because it would answer whichever gate parks next without you reading it. +A wait your shell can watch this way needs no `paused:` line, except your own pipeline run, a long foreground command, or your own validation round, which you declare once just before its blocking hold: append `paused:` once just before its first blocking command, then stay in the command, and never append it again as you reissue that command. +EOF +WAIT_SECTION=${WAIT_SECTION%$'\n'} +WAIT_BLOCK= +if [ -e "$CONFIG/wait-no-turns" ]; then + WAIT_BLOCK="$WAIT_SECTION"$'\n\n' +fi + if [ "$KIND" = secondmate ]; then SECONDMATE_PROJECTS="" idx=1 @@ -577,7 +607,7 @@ $CREWMATE_PAUSE_INSTRUCTIONS Firstmate's reply normally writes that closing line at answer time; when a blocker or wait clears WITHOUT a firstmate reply, append \`resolved [at=]: {how it cleared}\` yourself (same \`[key=]\` if you opened it with one) as you resume. $SHARED_INFRA_RULE -$INBOX_SECTION +$WAIT_BLOCK$INBOX_SECTION # Definition of done Write your findings to \`$DATA/$ID/report.md\`. @@ -655,7 +685,7 @@ $ASK_USER_BLOCK Firstmate's reply normally writes that closing line at answer time; when a blocker or wait clears WITHOUT a firstmate reply, append \`resolved [at=]: {how it cleared}\` yourself (same \`[key=]\` if you opened it with one) as you resume. $SHARED_INFRA_RULE -$INBOX_SECTION +$WAIT_BLOCK$INBOX_SECTION # Project memory A project's \`AGENTS.md\` or \`CLAUDE.md\` is loaded into every agent session in that project, so edit it only to correct information that is factually wrong - including information your own change made wrong - and never to add knowledge because it is missing. diff --git a/bin/fm-classify-lib.sh b/bin/fm-classify-lib.sh index 3d4e1f62282..11cc7f24cfb 100755 --- a/bin/fm-classify-lib.sh +++ b/bin/fm-classify-lib.sh @@ -922,6 +922,25 @@ EOF printf '%s\n' "$current" } +# The subset of status_open_decisions the task raised about its own work: a +# reserved-namespace key is raised by a supervisor library about the task (a +# pending-reply escalation), a `remote-reply-continuity-` key is the parent's +# own blocker about a broken remote reply mirror +# (bin/fm-procevent-remote-reply.sh), and a `captain-hold-` key relays a child +# decision a secondmate escalated to the captain (bin/fm-captain-hold.sh) while +# it keeps working, so the task is not waiting on any of them. Pending-reply +# recovery and a fire-and-forget retry ring consult this set and leave a task +# alone while it is non-empty. +status_own_open_decisions() { # + local line prefix + status_open_decisions "$1" | while IFS= read -r line || [ -n "$line" ]; do + for prefix in ${FM_CLASSIFY_RESERVED_KEY_PREFIXES:-$FM_CLASSIFY_RESERVED_KEY_PREFIXES_DEFAULT} remote-reply-continuity- captain-hold-; do + case "$line" in "$prefix"*) continue 2 ;; esac + done + printf '%s\n' "$line" + done +} + # 0 when the fold above still holds at least one decision OPENED by # `needs-decision` - the status side's own record that a human was asked # something and has not answered. A `blocked` record is deliberately not this: a diff --git a/bin/fm-dod-lib.sh b/bin/fm-dod-lib.sh index 52cc1cc1c14..aed6f5d0d22 100755 --- a/bin/fm-dod-lib.sh +++ b/bin/fm-dod-lib.sh @@ -280,12 +280,28 @@ EOF # Written once; only the two sentences about a green PR depend on the forge, # because on gerrit the ci step is skipped and there is no PR to report. fm_nm_driving_block() { # - local pr_return_line='' pr_reattach_clause=';' + local pr_return_line='' pr_reattach_clause=';' drive_block wait_cfg if [ "$1" != gerrit ]; then pr_return_line="Only a drive call's return reports the green PR: \`no-mistakes axi status\` shows progress but never reports \`checks-passed\` while the ci step is still monitoring the PR for merge, so never wait on a status poll for the next gate or outcome. " pr_reattach_clause="; once checks are green it returns \`checks-passed\` immediately, and" fi + # config/wait-no-turns selects the foreground drive. Absent, the text matches + # the backgrounded drive a home had before that flag. + wait_cfg=${CONFIG:-${FM_CONFIG_OVERRIDE:-${FM_HOME:-}/config}} + if [ -e "$wait_cfg/wait-no-turns" ]; then + drive_block="Drive the run with ONE foreground \`no-mistakes axi run\` and let it block. +It bounds its own hold for you: \`--wait\` (default 8m) exists precisely so a harness with a ten-minute command cap gets a structured return instead of being killed mid-hold. +Declare that wait using the brief's status-reporting rule before the foreground drive call. +Never background a wait, and never arm a timer to stand in for one: a backgrounded call returns in milliseconds, so it does not wait at all, and every timer left behind fires later as a paid wake for nothing. +${pr_return_line}Whenever a drive call returns without a gate or an outcome - its own wait elapsed, or it was killed or timed out - that is not a failure: reattach at once by re-running \`no-mistakes axi run\` without flags, and issue the same foreground call again, one at a time, until a gate or outcome comes back${pr_reattach_clause} if it refuses because no run is active, read the finished outcome from \`no-mistakes axi status\`." + else + drive_block="One drive call blocks until the next gate or outcome, which routinely outlives what your harness lets a single command run: Claude Code kills a command at ten minutes maximum, while one fix round is capped around thirty minutes and up to three rounds chain. +So background the drive call instead of sitting in one blocking hold your harness will kill, and read its return when it finishes. +Declare that wait using the brief's status-reporting rule before waiting on the backgrounded drive call. +Where a harness's own command limit is not established, assume it bounds commands and use that same backgrounded shape. +${pr_return_line}Whenever a drive call returns without a gate or an outcome - its own wait elapsed, or it was killed or timed out - reattach at once by re-running \`no-mistakes axi run\` without flags, backgrounded the same way${pr_reattach_clause} if it refuses because no run is active, read the finished outcome from \`no-mistakes axi status\`." + fi cat < task_id=$(fm_pending_reply_get "$rec" task_id) # A remote mate's report may exist and simply not have been mirrored yet. fm_pending_reply_missing_report_is_evidence "$state" "$task_id" "$completed" || return 1 + # config/wait-no-turns: a mate waiting on its own open decision or blocker + # is never poked. The recovery stays unattempted until the answer lands. + if [ -e "${FM_CONFIG_OVERRIDE:-${FM_HOME:-}/config}/wait-no-turns" ]; then + [ -z "$(status_own_open_decisions "$state/$task_id.status")" ] || return 1 + fi status_file=$(fm_pending_reply_get "$rec" parent_status) parent_home=$(fm_pending_reply_get "$rec" parent_home) msg=$(fm_pending_reply_recovery_message "$rec") diff --git a/bin/fm-send.sh b/bin/fm-send.sh index af09392a4d0..19562680313 100755 --- a/bin/fm-send.sh +++ b/bin/fm-send.sh @@ -51,7 +51,9 @@ # watcher re-rings an unacknowledged message while its endpoint remains # available, escalates after the bounded ladder, and instead routes a positively # dead or missing endpoint directly to recovery without typing. An explicit -# fire-and-forget record is excluded from that ladder. +# fire-and-forget record is excluded from that ladder; when config/wait-no-turns +# is present and its ring here was skipped or failed, the watcher rings it +# exactly once more. # bin/fm-task-inbox-lib.sh owns the record format, the doorbell line, and the # re-ring ladder. The composer pre-check before the ring is ADVISORY only: when # the composer visibly holds pending text the ring is skipped with a notice and @@ -1085,9 +1087,22 @@ else # bounded re-ring ladder or direct unavailable-endpoint recovery. ring_rc=0 fm_task_inbox_ring "$TARGET_BACKEND" "$T" "$INBOX_RECORD" "$EXPECTED_LABEL" || ring_rc=$? + ring_retry="the watcher will re-ring" + if [ -n "$FIRE_AND_FORGET_ID" ] \ + && [ -e "${FM_CONFIG_OVERRIDE:-$FM_HOME/config}/wait-no-turns" ]; then + case "$ring_rc" in + 1|2) + if fm_task_inbox_mark_retry "$STATE" "$INBOX_TASK_ID" "$INBOX_RECORD"; then + ring_retry="the watcher will ring it once more" + else + ring_retry="its one retry ring could not be recorded, so nothing will ring it again" + fi + ;; + esac + fi case "$ring_rc" in - 1) echo "fm-send: doorbell skipped (composer visibly holds pending text); the steer is durably recorded at $INBOX_RECORD and the watcher will re-ring" >&2 ;; - 2) echo "fm-send: doorbell did not reach $T; the steer is durably recorded at $INBOX_RECORD and the watcher will re-ring" >&2 ;; + 1) echo "fm-send: doorbell skipped (composer visibly holds pending text); the steer is durably recorded at $INBOX_RECORD and $ring_retry" >&2 ;; + 2) echo "fm-send: doorbell did not reach $T; the steer is durably recorded at $INBOX_RECORD and $ring_retry" >&2 ;; 3) echo "fm-send: doorbell not typed because the agent in $T has exited; the steer is durably recorded at $INBOX_RECORD for recovery (stuck-crewmate-recovery), and the watcher will not re-ring a dead pane" >&2 ;; esac exit 0 diff --git a/bin/fm-task-inbox-lib.sh b/bin/fm-task-inbox-lib.sh index c1d48689975..257da1dd54d 100644 --- a/bin/fm-task-inbox-lib.sh +++ b/bin/fm-task-inbox-lib.sh @@ -30,11 +30,14 @@ # .inbox/.ring-state watcher re-ring ladder: "\t\t" # .inbox/.escalated oldest-message name already surfaced as stale, # so later polls suppress another escalation +# .inbox/.retry-ring name of a fire-and-forget record still owed its +# one retry ring (fm_task_inbox_mark_retry) # # Record format (fm_task_inbox_write / fm_task_inbox_body): # schema=fm-task-inbox.v1 # at= # delivery=fire-and-forget present only when the re-ring ladder must ignore it +# (it still gets one retry ring; see below) # -- # @@ -59,6 +62,19 @@ # crash or marker failure may produce a rare duplicate rather than silently lose # a wake. # +# Retry ring (fm_task_inbox_mark_retry): only while config/wait-no-turns is +# present. A fire-and-forget record never enters the ladder, but when +# fm-send's ring at enqueue did not land +# (fm_task_inbox_ring returned 1 or 2) it marks the record, and one grace later +# the due action is `retry`: once the worker has no open decision of its own, +# the watcher rings once more and spends the mark +# whatever the result, so the record never rings a third time and never +# escalates. A waiting worker does not poll its inbox (bin/fm-brief.sh), so +# without this retry the record could sit unread until a checkpoint. A pending ordinary record's +# ladder rings the same inbox, so the retry waits behind it, and an +# acknowledged record drops its mark. The remote steer leg has no watcher +# ladder and owes no retry. +# # Inbox names containing bytes outside printable ASCII are unsupported. The # doorbell refuses them rather than sending terminal control bytes to a pane. # @@ -369,11 +385,30 @@ fm_task_inbox_oldest_unhandled() { # printf '%s' "$best" } +# Owe a fire-and-forget record its one retry ring (see the header). A newer +# mark replaces an older one: a ring names the whole inbox, not one record. +fm_task_inbox_mark_retry() { # + local dir + dir=$(fm_task_inbox_dir "$1" "$2") + { printf '%s\n' "${3##*/}" > "$dir/.retry-ring"; } 2>/dev/null +} + +# Spend the retry mark after its ring, only while it still names that record: +# a newer mark written meanwhile is owed its own retry and survives. Fails only +# when the processed record's mark stays behind. +fm_task_inbox_clear_retry() { # + local dir + dir=$(fm_task_inbox_dir "$1" "$2") + [ "$(cat "$dir/.retry-ring" 2>/dev/null)" = "${3##*/}" ] || return 0 + rm -f "$dir/.retry-ring" 2>/dev/null +} + # The re-ring ladder decision for one task. Prints exactly one of: # quiet nothing due (healthy, within grace or spacing, # or already escalated for the current oldest) # ring one doorbell re-ring is due # escalate attempt budget spent; surface as stale +# retry a fire-and-forget record's one retry ring is due # An empty inbox also resets the ladder bookkeeping so the next message starts # a fresh ladder. fm_task_inbox_due_action() { # @@ -381,6 +416,17 @@ fm_task_inbox_due_action() { # dir=$(fm_task_inbox_dir "$1" "$2") if ! oldest=$(fm_task_inbox_oldest_unhandled "$1" "$2"); then rm -f "$dir/.ring-state" "$dir/.escalated" 2>/dev/null || true + # The one retry ring exists only while config/wait-no-turns is present. + # Absent, a mark is left untouched and the inbox stays quiet, as before. + if [ -e "${FM_CONFIG_OVERRIDE:-${FM_HOME:-}/config}/wait-no-turns" ]; then + base=$(cat "$dir/.retry-ring" 2>/dev/null || true) + if ! fm_task_inbox_seq_of "$base" >/dev/null || [ ! -f "$dir/$base" ]; then + rm -f "$dir/.retry-ring" 2>/dev/null || true + elif [ "$(fm_path_age "$dir/.retry-ring")" -ge "$(fm_task_inbox_grace_secs)" ]; then + printf 'retry %s' "$dir/$base" + return 0 + fi + fi printf 'quiet' return 0 fi diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index 96bae225fa5..35c5a9f1b75 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -529,7 +529,10 @@ inbox_steer_escalate_unavailable() { # # stale path instead of silently re-ringing forever; acknowledgement or teardown # still makes the race quiet. The attempt is data-plane typing or a # composer-protected skip, never a wake, so normal retries keep the watcher -# blocking. Runs for secondmates +# blocking. A fire-and-forget record's one retry ring follows the same busy +# wait, also waits while the worker has an open decision or blocker of its own +# (status_own_open_decisions), and never escalates: a dead pane just spends it. +# Runs for secondmates # too: their pane-staleness exemption is about quiet panes being healthy, # while an unacknowledged instruction past the ladder is a stuck steer. inbox_steer_check() { # @@ -537,6 +540,9 @@ inbox_steer_check() { # action=$(fm_task_inbox_due_action "$STATE" "$task") || return 0 verb=${action%% *} [ "$verb" != quiet ] || return 0 + if [ "$verb" = retry ] && [ -n "$(status_own_open_decisions "$STATE/$task.status")" ]; then + return 0 + fi rec=${action#* } count= case "$verb" in @@ -549,7 +555,11 @@ inbox_steer_check() { # agent_state=$(fm_backend_agent_state "$backend" "$w" 2>/dev/null || true) case "$agent_state" in dead|missing) - inbox_steer_escalate_unavailable "$w" "$task" "$rec" + if [ "$verb" = retry ]; then + fm_task_inbox_clear_retry "$STATE" "$task" "$rec" || true + else + inbox_steer_escalate_unavailable "$w" "$task" "$rec" + fi return 0 ;; esac @@ -578,6 +588,16 @@ inbox_steer_check() { # fi triage_log "steer-inbox delivery attempt: $task ${rec##*/} result=$ring_rc" ;; + retry) + ring_rc=0 + fm_task_inbox_ring "$backend" "$w" "$rec" "$(window_label "$w")" || ring_rc=$? + if ! fm_task_inbox_clear_retry "$STATE" "$task" "$rec" && [ -f "$rec" ]; then + reason="stale: $w (steering-inbox retry mark unremovable: ${rec%/*}/.retry-ring cannot be removed, so $rec would ring on every poll - inspect the inbox directory)" + fm_wake_append stale "$w" "$reason" || exit 1 + wake "$reason" + fi + triage_log "steer-inbox retry ring: $task ${rec##*/} result=$ring_rc" + ;; escalate) reason="stale: $w (unread firstmate instruction: $rec still unhandled after $count doorbell delivery attempts with an idle pane; inspect the worker)" if [ ! -d "${rec%/*}" ] || [ ! -f "$rec" ]; then diff --git a/docs/configuration.md b/docs/configuration.md index 5dc5b92f47e..965ddbce5a9 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -573,6 +573,13 @@ See [`trace-context.md`](trace-context.md) for carrier semantics, supported rout See [`fleet-ledger.md`](fleet-ledger.md) for the opt-in setup, record contract, and limits. +## Waiting worker spends no turns (config/wait-no-turns) + +The optional local, gitignored `config/wait-no-turns` presence flag opts this home into keeping a waiting worker from spending turns until it is answered. +With it present, ship and scout briefs gain the `# Waiting` section and the foreground no-mistakes drive text, every brief's inbox section keeps the natural-checkpoint check and adds that a waiting worker does not poll or list its inbox because a waiting instruction rings, a pending-reply recovery waits while that mate has its own open decision or blocker, and a fire-and-forget steer whose doorbell did not land gets one later ring. +With the file absent, generated briefs omit the waiting section and the no-poll inbox line, the drive text backgrounds the call, recovery sends during an open decision, and a fire-and-forget steer is not owed a retry ring. +The flag is a home-local preference and is not inherited by secondmate homes. + ## Turn-end pane-churn absorb (config/turnend-churn-absorb) The optional local, gitignored `config/turnend-churn-absorb` presence flag opts this home into a default-off third form of positive work evidence in watcher triage. diff --git a/docs/remote-secondmates.md b/docs/remote-secondmates.md index 42d496acb39..eb7537cf84b 100644 --- a/docs/remote-secondmates.md +++ b/docs/remote-secondmates.md @@ -487,6 +487,7 @@ When deduplication finds that the worker already moved the matching record into The remote host runs no doorbell re-ring ladder of its own. A swallowed doorbell for an ordinary reply-bearing request surfaces through the parent's pending-reply recovery and escalation. Its recovery request rings the doorbell again when it is enqueued. +A fire-and-forget record, such as a reconcile ask, gets its single retry ring only on the local plane, and only when `config/wait-no-turns` is present: the remote steer leg owes no re-ring, so a swallowed remote doorbell for one waits for the next ring into that inbox, and a remote-side retry is known follow-up scope. ### Remote reads diff --git a/docs/verification/runtime-backends.md b/docs/verification/runtime-backends.md index 28593006c4a..bef0095bb91 100644 --- a/docs/verification/runtime-backends.md +++ b/docs/verification/runtime-backends.md @@ -844,6 +844,43 @@ ok - opencode (1.18.33): the doorbell reached a real worker, which acted and ack OpenCode needed `FM_SEND_INBOX_LIVE_TIMEOUT=560` because its configured model was still mid-turn at the default 240 seconds. Pi 0.87.1 was installed but not verified: its configured model returned an account error (`The 'gpt-5.6-sol' model is not supported when using Codex with a ChatGPT account`) before it read the inbox. +## Waiting-worker command ceilings + +The `# Waiting` section of the ship and scout briefs (`bin/fm-brief.sh`) has a worker hold every external wait inside one blocking shell command, bounded by what its harness lets one command run. +That section is generated only when `config/wait-no-turns` is present. +Those bounds were read from the installed vendor code on 2026-09-11, macOS arm64, with Pi 0.85.1, codex-cli 0.154.0, and Claude Code 2.1.268. + +```sh +grep -n "Timeout in seconds" "$(npm root -g)/@earendil-works/pi-coding-agent/dist/core/tools/bash.js" +strings -n 20 "$(readlink -f "$(command -v codex)")" | grep -o "Non-empty writes default to [^.]*; empty polls wait [^.]*\." +strings -n 8 "$(readlink -f "$(command -v claude)")" | grep -oE '=120000,[A-Za-z0-9_$]+=600000;' | head -1 +``` + +Observed output: + +```text +28: timeout: Type.Optional(Type.Number({ description: "Timeout in seconds (optional, no default timeout)" })), +Non-empty writes default to 250 ms and cap at 30000 ms; empty polls wait 5000-300000 ms by default. +=120000,ARo=600000; +``` + +Pi's bash tool runs a command with no time limit unless the call passes `timeout`, so the brief asks for at most 2700 seconds, which stays under the watcher's 3600-second busy-turn bound. +Codex yields a still-running command back to the model, and one empty `write_stdin` poll then waits up to 300000 ms. +Claude Code's Bash tool defaults to 120000 ms and accepts at most 600000 ms; `BASH_DEFAULT_TIMEOUT_MS` and `BASH_MAX_TIMEOUT_MS` override those two values. + +Claude Code also constrains the shape of a wait, not only its length, so the brief has to name the shape that is allowed rather than only forbid the ones that are not. +Run as separate Bash tool calls on 2026-09-14 with Claude Code 2.1.268: + +```sh +until [ -e /tmp/fm-wait-probe ]; do sleep 30; done # ran to completion, rc=0 +sleep 61; echo "rc=$?" # rc=0 +sleep 40; echo "checked at $(date +%s)" # rc=0 +``` + +An earlier `sleep 60` chained ahead of a status check was refused before execution, with a message pointing at `Monitor` with an until-loop and at `run_in_background: true`, and adding "Do not chain shorter sleeps to work around this block". +The blocking foreground `until` loop is therefore the wait a Claude Code worker may use, and it is what the brief names, because the refusal's own `run_in_background` suggestion is the one shape a waiting worker must not take: a backgrounded call returns at once and so does not wait at all. +The brief's portable regression is `tests/fm-brief.test.sh`; rerun these commands after upgrading any of the three harnesses and update the numbers in the brief when they move. + ## Gemini The Gemini crewmate adapter was verified on 2026-09-04 with gemini-cli 0.58.0 on Linux, Node v24.20.0, tmux 3.4. diff --git a/tests/fm-brief.test.sh b/tests/fm-brief.test.sh index 7234ae435ce..225d1c0975c 100755 --- a/tests/fm-brief.test.sh +++ b/tests/fm-brief.test.sh @@ -920,7 +920,8 @@ SIGNALS test_ship_and_scout_teach_validation_round_pause() { local home kind id brief home="$TMP_ROOT/validation-round-pause-home" - mkdir -p "$home/data" + mkdir -p "$home/data" "$home/config" + : > "$home/config/wait-no-turns" for kind in ship scout; do id="brief-validation-round-pause-$kind" @@ -930,6 +931,12 @@ test_ship_and_scout_teach_validation_round_pause() { FM_HOME="$home" "$ROOT/bin/fm-brief.sh" "$id" firstmate --mode no-mistakes >/dev/null 2>&1 fi brief="$home/data/$id/brief.md" + assert_grep "your own validation round, which you declare once just before its blocking hold" "$brief" \ + "$kind brief did not teach workers to declare their validation-round wait before holding it" + assert_grep "append \`paused:\` once just before its first blocking command, then stay in the command" "$brief" \ + "$kind brief's Waiting section does not declare the validation round once and then hold it" + assert_no_grep "is not a \`paused:\` wait" "$brief" \ + "$kind brief still tells workers never to declare a wait they hold in a command" assert_grep "your own validation round" "$brief" \ "$kind brief did not teach workers to declare their validation-round wait" assert_grep 'Before ending your turn with your own background shell or monitor still running' "$brief" \ @@ -941,7 +948,7 @@ test_ship_and_scout_teach_validation_round_pause() { assert_grep 'Do not declare active implementation or reasoning as a wait' "$brief" \ "$kind brief did not limit the declaration to actual waits" done - pass "fm-brief.sh: ship and scout scaffolds teach validation-round pauses" + pass "fm-brief.sh: ship and scout scaffolds declare a validation-round pause once, then hold it" } test_scout_and_secondmate_load_decision_hold_policy() { @@ -1026,6 +1033,68 @@ test_scout_and_secondmate_scaffold() { pass "fm-brief: scout and secondmate code paths still scaffold well-formed briefs" } +# Contract: a waiting worker spends no turns. A decision wait ends the turn, an +# external wait sleeps in one bounded blocking shell command sized per harness, +# and a waiting worker neither polls its inbox nor polls a pipeline between holds. +test_workers_wait_without_spending_turns() { + local home id brief + home="$TMP_ROOT/wait-home" + mkdir -p "$home/data" "$home/config" + : > "$home/config/wait-no-turns" + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" brief-wait-ship some-proj --mode no-mistakes >/dev/null 2>&1 \ + || fail "fm-brief.sh ship scaffold exited non-zero" + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" brief-wait-scout some-proj --scout >/dev/null 2>&1 \ + || fail "fm-brief.sh scout scaffold exited non-zero" + for id in brief-wait-ship brief-wait-scout; do + brief="$home/data/$id/brief.md" + assert_grep "end your turn at once" "$brief" "$id: a decision wait must end the turn" + assert_grep "with ONE blocking shell command that returns when the state changes" "$brief" \ + "$id: an external wait must sleep in one blocking shell command" + assert_grep "gh pr checks --watch" "$brief" "$id: the CI wait primitive is missing" + assert_grep "a \`timeout\` of at most 2700 seconds" "$brief" "$id: the Pi ceiling is missing" + assert_grep "its maximum \`timeout\` of 600000 ms" "$brief" "$id: the Claude Code ceiling is missing" + assert_grep "empty \`write_stdin\` polls of up to 300000 ms" "$brief" "$id: the Codex ceiling is missing" + assert_grep "is the sanctioned foreground wait" "$brief" \ + "$id: the wait a Claude Code worker may use is not named" + assert_grep "reattach with \`no-mistakes axi run --wait\` instead, and never send the same \`respond\` again" "$brief" \ + "$id: a timed-out respond must reattach with axi run, never resend its answer" + assert_grep "Do not poll or list the inbox while waiting; a waiting instruction rings." "$brief" \ + "$id: polling the inbox while waiting is not forbidden" + assert_grep "natural checkpoint" "$brief" "$id: the flag dropped the natural-checkpoint inbox check" + done + brief="$home/data/brief-wait-ship/brief.md" + assert_grep "issue the same foreground call again" "$brief" \ + "the no-mistakes DOD must reattach with the same foreground call" + assert_no_grep "background the drive call" "$brief" "the no-mistakes DOD still backgrounds the drive call" + + FM_SECONDMATE_CHARTER='Supervise the alpha domain.' \ + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" brief-wait-sm --secondmate --no-projects >/dev/null 2>&1 \ + || fail "fm-brief.sh secondmate scaffold exited non-zero" + brief="$home/data/brief-wait-sm/brief.md" + assert_grep "Do not poll or list the inbox while waiting; a waiting instruction rings." "$brief" \ + "secondmate: polling the inbox while waiting is not forbidden" + assert_grep "natural checkpoint" "$brief" "secondmate: the flag dropped the natural-checkpoint inbox check" + pass "fm-brief: workers end the turn on a decision, wait in one bounded shell command, and never poll" +} + +# Without config/wait-no-turns the scaffold matches the pre-flag brief and drive text. +test_wait_no_turns_absent_keeps_the_previous_brief() { + local home brief + home="$TMP_ROOT/wait-off" + mkdir -p "$home/data" + [ ! -e "$home/config/wait-no-turns" ] + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" brief-wait-off some-proj --mode no-mistakes >/dev/null 2>&1 \ + || fail "fm-brief.sh ship scaffold exited non-zero" + brief="$home/data/brief-wait-off/brief.md" + assert_no_grep "end your turn at once" "$brief" "an absent flag still added the waiting section" + assert_grep "natural checkpoint" "$brief" "an absent flag dropped the unprompted inbox check" + assert_no_grep "Do not poll or list the inbox while waiting" "$brief" "an absent flag still added the no-poll inbox line" + assert_grep "background the drive call" "$brief" "an absent flag replaced the backgrounded drive text" + assert_no_grep "issue the same foreground call again" "$brief" \ + "an absent flag still asked for the foreground reattach" + pass "fm-brief: without config/wait-no-turns the brief and drive text stay as they were" +} + test_worker_role_scope() { local kind home brief home="$TMP_ROOT/worker-role" @@ -1352,6 +1421,8 @@ test_ship_and_scout_teach_validation_round_pause test_scout_and_secondmate_load_decision_hold_policy test_scout_and_secondmate_scaffold test_scout_lavish_line_follows_presentation_floor +test_workers_wait_without_spending_turns +test_wait_no_turns_absent_keeps_the_previous_brief test_home_brief_include_is_appended_last test_ship_branch_prefix_defaults_to_legacy_fm test_ship_branch_prefix_override_is_consistent_across_modes diff --git a/tests/fm-pending-reply.test.sh b/tests/fm-pending-reply.test.sh index 350741fca2a..777d1c97c57 100755 --- a/tests/fm-pending-reply.test.sh +++ b/tests/fm-pending-reply.test.sh @@ -201,6 +201,75 @@ test_completed_turn_no_report_triggers_one_recovery() { pass "completed turn with no report triggers exactly one recovery" } +# A mate waiting on its own open decision is never poked by the recovery; the +# recovery stays unattempted and runs once the decision closes. +test_recovery_waits_while_the_mate_has_an_open_decision() { + local home state corr hook_log + home=$(setup_parent decision-wait) + state="$home/state" + hook_log="$TMP_ROOT/decision-wait-hook.log" + : > "$hook_log" + export FM_PENDING_REPLY_NOW=2500 + mkdir -p "$home/config" + : > "$home/config/wait-no-turns" + FM_CONFIG_OVERRIDE="$home/config" + # Invoked indirectly through FM_PENDING_REPLY_SEND_HOOK. + # shellcheck disable=SC2329 + decision_wait_hook() { + printf '%s\n' "$1" >> "$hook_log" + } + export -f decision_wait_hook + export FM_PENDING_REPLY_SEND_HOOK=decision_wait_hook + + corr=$(fm_pending_reply_create "$home" "$state" "hibit" "status of phase 8") + fm_pending_reply_mark_delivered "$state" "$corr" + fm_pending_reply_observe_busy "$state" "$corr" busy + fm_pending_reply_observe_busy "$state" "$corr" idle + printf 'needs-decision [key=scope]: narrow or wide?\n' >> "$state/hibit.status" + if fm_pending_reply_send_recovery "$state" "$corr" 2>/dev/null; then + fail "recovery must wait while the mate waits on its own decision" + fi + [ ! -s "$hook_log" ] || fail "recovery poked a mate waiting on its decision" + [ "$(phase_of "$state" "$corr")" = awaiting_report ] \ + || fail "a deferred recovery must stay unattempted, got $(phase_of "$state" "$corr")" + + printf 'resolved [key=scope]: answered: narrow\n' >> "$state/hibit.status" + fm_pending_reply_send_recovery "$state" "$corr" || fail "recovery should send once the decision closes" + [ "$(wc -l < "$hook_log" | tr -d ' ')" = 1 ] || fail "expected exactly one recovery send" + unset FM_PENDING_REPLY_SEND_HOOK + unset FM_CONFIG_OVERRIDE + pass "recovery never pokes a mate waiting on its own decision, and runs once it closes" +} + +# Without the flag, an open decision does not hold the recovery. +test_recovery_sends_during_an_open_decision_without_the_flag() { + local home state corr hook_log + home=$(setup_parent decision-wait-off) + state="$home/state" + hook_log="$TMP_ROOT/decision-wait-off-hook.log" + : > "$hook_log" + mkdir -p "$home/config" + FM_CONFIG_OVERRIDE="$home/config" + export FM_PENDING_REPLY_NOW=2500 + # shellcheck disable=SC2329 + decision_wait_off_hook() { + printf '%s\n' "$1" >> "$hook_log" + } + export -f decision_wait_off_hook + export FM_PENDING_REPLY_SEND_HOOK=decision_wait_off_hook + corr=$(fm_pending_reply_create "$home" "$state" "hibit" "status of phase 8") + fm_pending_reply_mark_delivered "$state" "$corr" + fm_pending_reply_observe_busy "$state" "$corr" busy + fm_pending_reply_observe_busy "$state" "$corr" idle + printf 'needs-decision [key=scope]: narrow or wide?\n' >> "$state/hibit.status" + fm_pending_reply_send_recovery "$state" "$corr" \ + || fail "recovery should send while a decision is open when the flag is absent" + [ "$(wc -l < "$hook_log" | tr -d ' ')" = 1 ] || fail "expected the recovery to send" + unset FM_PENDING_REPLY_SEND_HOOK + unset FM_CONFIG_OVERRIDE + pass "recovery sends during an open decision when config/wait-no-turns is absent" +} + test_recovery_grace_measures_from_turn_completion() { local home state corr hook_log lines home=$(setup_parent grace-from-completion) @@ -1927,6 +1996,8 @@ test_escalated_undelivered_correlation_stays_retryable() { test_normal_correlated_reply_resolves_once test_completed_turn_no_report_triggers_one_recovery +test_recovery_waits_while_the_mate_has_an_open_decision +test_recovery_sends_during_an_open_decision_without_the_flag test_recovery_grace_measures_from_turn_completion test_recovery_fresh_status_read_resolves_before_firing test_partial_resolve_write_blocks_firing diff --git a/tests/fm-remote-reply.test.sh b/tests/fm-remote-reply.test.sh index 47f75a75593..d6b00c7cead 100755 --- a/tests/fm-remote-reply.test.sh +++ b/tests/fm-remote-reply.test.sh @@ -694,6 +694,9 @@ pass "source-line identity survives commit failure and cursor-loss recapture" # the reserved key over. # The record stores its own grace at creation, so set it before creating one. export FM_PENDING_REPLY_GRACE_SECS=0 +# Answer the mate's earlier decisions and blocker first: a recovery repost waits +# while the mate has one of its own open (tests/fm-pending-reply.test.sh). +printf 'resolved [key=%s]: answered\n' rough-cut-version ctl default >> "$PARENT/state/ios.status" ESCALATED_CORR=$(fm_pending_reply_create "$PARENT" "$PARENT/state" ios 'confirm the notarization') [ -n "$ESCALATED_CORR" ] || fail "could not create the pending-reply record to escalate" fm_pending_reply_mark_delivered "$PARENT/state" "$ESCALATED_CORR" \ diff --git a/tests/fm-send-inbox.test.sh b/tests/fm-send-inbox.test.sh index c0c61f62ae5..2367771311d 100644 --- a/tests/fm-send-inbox.test.sh +++ b/tests/fm-send-inbox.test.sh @@ -14,7 +14,8 @@ # 4. The composer pre-check is advisory: visibly pending text skips the ring # with a notice, and the steer is still durably sent (exit 0). # 5. A failed doorbell is still a sent steer (exit 0, record durable): the -# watcher's re-ring ladder owns delivery from the record on. +# watcher's re-ring ladder owns delivery from the record on. A +# fire-and-forget record whose ring did not land is owed one retry ring. # 6. Carve-outs keep the typed plane: a leading "/" (any harness), a leading # "$" to codex, an explicit backend target, and the --key path. # 7. A marked secondmate steer carries its marker + corr token in the record @@ -241,6 +242,56 @@ test_failed_ring_is_still_sent() { pass "fm-send inbox: a failed doorbell is still a durably sent steer" } +# Contract: a fire-and-forget record stays outside the re-ring ladder, so a +# ring that did not land at enqueue is owed exactly one retry by the watcher. +test_fire_and_forget_unlanded_ring_owes_one_retry() { + local dir err rc + dir=$(setup_case faf-retry) + mkdir -p "$dir/home/config" + : > "$dir/home/config/wait-no-turns" + err="$dir/send.err" + # The stub lists only window fm-t1, so the secondmate takes it over. + rm -f "$dir/home/state/t1.meta" + fm_write_secondmate_meta "$dir/home/state/domain.meta" "$dir/home" "sess:fm-t1" alpha claude + run_send "$dir" "$err" FM_FAKE_TMUX_COMPOSER=pending -- \ + fm-domain --fire-and-forget 0123456789abcdef "reconcile your books"; rc=$? + expect_code 0 "$rc" "a skipped fire-and-forget ring is still a sent steer" + [ "$(cat "$dir/home/state/domain.inbox/.retry-ring" 2>/dev/null)" = 001.msg ] \ + || fail "a skipped fire-and-forget ring did not owe its one retry" + assert_contains "$(cat "$err")" "the watcher will ring it once more" \ + "the skip notice should promise exactly one retry" + + run_send "$dir" "$err" -- fm-domain --fire-and-forget 1123456789abcdef "reconcile again"; rc=$? + expect_code 0 "$rc" "a rung fire-and-forget steer should succeed" + [ "$(cat "$dir/home/state/domain.inbox/.retry-ring" 2>/dev/null)" = 001.msg ] \ + || fail "a ring that landed must not owe a retry for its own record" + + dir=$(setup_case ordinary-no-retry) + err="$dir/send.err" + run_send "$dir" "$err" FM_FAKE_TMUX_COMPOSER=pending -- t1 "ordinary steer" + [ ! -e "$dir/home/state/t1.inbox/.retry-ring" ] \ + || fail "an ordinary record rides the ladder and must not owe a separate retry" + pass "fm-send inbox: a fire-and-forget ring that did not land owes one retry ring" +} + +# Without the flag a skipped fire-and-forget ring is not owed a retry. +test_fire_and_forget_retry_stays_off_without_the_flag() { + local dir err rc + dir=$(setup_case faf-retry-off) + err="$dir/send.err" + [ ! -e "$dir/home/config/wait-no-turns" ] + rm -f "$dir/home/state/t1.meta" + fm_write_secondmate_meta "$dir/home/state/domain.meta" "$dir/home" "sess:fm-t1" alpha claude + run_send "$dir" "$err" FM_FAKE_TMUX_COMPOSER=pending -- \ + fm-domain --fire-and-forget 0123456789abcdef "reconcile your books"; rc=$? + expect_code 0 "$rc" "a skipped fire-and-forget ring is still a sent steer" + [ ! -e "$dir/home/state/domain.inbox/.retry-ring" ] \ + || fail "an absent flag still owed a fire-and-forget retry" + assert_contains "$(cat "$err")" "the watcher will re-ring" \ + "an absent flag should keep the ordinary re-ring notice" + pass "fm-send inbox: without config/wait-no-turns a fire-and-forget ring is not retried" +} + test_harness_invocations_stay_typed() { local dir err typed # A slash command must reach the harness's own parser, on any harness. @@ -459,6 +510,8 @@ test_multiline_steer_is_legal test_resend_enqueues_new_sequence test_pending_composer_skips_ring_advisorily test_failed_ring_is_still_sent +test_fire_and_forget_unlanded_ring_owes_one_retry +test_fire_and_forget_retry_stays_off_without_the_flag test_harness_invocations_stay_typed test_explicit_target_stays_typed test_key_path_never_touches_inbox diff --git a/tests/fm-task-inbox.test.sh b/tests/fm-task-inbox.test.sh index 7a188c76422..3c9c8dce4d4 100644 --- a/tests/fm-task-inbox.test.sh +++ b/tests/fm-task-inbox.test.sh @@ -26,6 +26,9 @@ # 6. Dead panes: the doorbell line is a shell no-op when executed by a bare # shell, the ring skips an agent the backend classifies dead, and the # watcher surfaces such a record exactly once instead of re-ringing. +# 7. A fire-and-forget record stays outside the ladder, but one whose first +# ring did not land gets exactly one retry ring and never escalates. The +# retry waits while the worker has an open decision of its own. set -u # shellcheck source=tests/wake-helpers.sh @@ -79,6 +82,10 @@ case "${1:-}" in if [ -n "${FM_ACK_RECORD:-}" ] && [ -f "$FM_ACK_RECORD" ]; then mv "$FM_ACK_RECORD" "${FM_ACK_RECORD%/*}/handled/" fi + # A concurrent fire-and-forget send marking its newer record mid-ring. + if [ -n "${FM_RING_MARKS_RETRY:-}" ]; then + printf '%s\n' "${FM_RING_MARKS_RETRY##*/}" > "${FM_RING_MARKS_RETRY%/*}/.retry-ring" + fi fi exit 0 ;; display-message) @@ -548,6 +555,59 @@ test_fire_and_forget_records_never_enter_the_ladder() { pass "inbox: fire-and-forget records stay durable and outside the ladder" } +test_fire_and_forget_retry_is_owed_once() { + local state fire tracked action + state="$TMP_ROOT/faf-retry/state"; mkdir -p "$state" "$TMP_ROOT/faf-retry/config" + : > "$TMP_ROOT/faf-retry/config/wait-no-turns" + export FM_CONFIG_OVERRIDE="$TMP_ROOT/faf-retry/config" + fire=$(inbox_lib "$state" fm_task_inbox_write "$state" t1 "one-shot steer" fire-and-forget) + age_path "$fire" + inbox_lib "$state" fm_task_inbox_mark_retry "$state" t1 "$fire" + action=$(FM_TASK_INBOX_GRACE_SECS=3600 inbox_lib "$state" fm_task_inbox_due_action "$state" t1) + [ "$action" = quiet ] || fail "a retry inside grace should be quiet, got: $action" + age_path "$state/t1.inbox/.retry-ring" + action=$(FM_TASK_INBOX_GRACE_SECS=60 inbox_lib "$state" fm_task_inbox_due_action "$state" t1) + [ "$action" = "retry $fire" ] || fail "an aged retry mark should be due its ring, got: $action" + # An ordinary record's ladder rings the same inbox, so the retry waits behind it. + tracked=$(inbox_lib "$state" fm_task_inbox_write "$state" t1 "tracked steer") + age_path "$tracked" + action=$(FM_TASK_INBOX_GRACE_SECS=60 inbox_lib "$state" fm_task_inbox_due_action "$state" t1) + [ "$action" = "ring $tracked" ] || fail "a pending ordinary record should own the ring, got: $action" + mv "$tracked" "$state/t1.inbox/handled/" + action=$(FM_TASK_INBOX_GRACE_SECS=60 inbox_lib "$state" fm_task_inbox_due_action "$state" t1) + [ "$action" = "retry $fire" ] || fail "the retry should resume once the ordinary record is handled, got: $action" + # Once spent, the record is quiet for good: no second retry and no escalation. + inbox_lib "$state" fm_task_inbox_clear_retry "$state" t1 "$fire" + action=$(FM_TASK_INBOX_GRACE_SECS=0 FM_TASK_INBOX_RING_MAX=0 \ + inbox_lib "$state" fm_task_inbox_due_action "$state" t1) + [ "$action" = quiet ] || fail "a spent retry rang or escalated again: $action" + # An acknowledged record drops its mark. + inbox_lib "$state" fm_task_inbox_mark_retry "$state" t1 "$fire" + age_path "$state/t1.inbox/.retry-ring" + mv "$fire" "$state/t1.inbox/handled/" + action=$(FM_TASK_INBOX_GRACE_SECS=60 inbox_lib "$state" fm_task_inbox_due_action "$state" t1) + [ "$action" = quiet ] || fail "an acknowledged record's retry should be dropped, got: $action" + [ ! -e "$state/t1.inbox/.retry-ring" ] || fail "an acknowledged record kept its retry mark" + unset FM_CONFIG_OVERRIDE + pass "inbox: a fire-and-forget record whose ring did not land is owed exactly one retry" +} + +# A retry mark is ignored while config/wait-no-turns is absent. +test_fire_and_forget_retry_is_quiet_without_the_flag() { + local state fire action + state="$TMP_ROOT/faf-retry-off/state"; mkdir -p "$state" "$TMP_ROOT/faf-retry-off/config" + export FM_CONFIG_OVERRIDE="$TMP_ROOT/faf-retry-off/config" + fire=$(inbox_lib "$state" fm_task_inbox_write "$state" t1 "one-shot steer" fire-and-forget) + age_path "$fire" + inbox_lib "$state" fm_task_inbox_mark_retry "$state" t1 "$fire" + age_path "$state/t1.inbox/.retry-ring" + action=$(FM_TASK_INBOX_GRACE_SECS=60 inbox_lib "$state" fm_task_inbox_due_action "$state" t1) + [ "$action" = quiet ] || fail "an absent flag still owed a retry ring, got: $action" + [ -e "$state/t1.inbox/.retry-ring" ] || fail "an absent flag removed a retry mark it should have left" + unset FM_CONFIG_OVERRIDE + pass "inbox: without config/wait-no-turns a fire-and-forget retry mark stays quiet" +} + test_ring_ladder_policy() { local state rec action state="$TMP_ROOT/ladder/state"; mkdir -p "$state" @@ -725,6 +785,101 @@ test_watcher_surfaces_unwritable_ladder() { pass "watcher: unwritable ladder bookkeeping surfaces a stale wake after the doorbell" } +test_watcher_pays_fire_and_forget_retry_once() { + local dir state out log pid fire rings i=0 + dir=$(setup_watch_case faf-retry) + mkdir -p "$dir/config" + : > "$dir/config/wait-no-turns" + state="$dir/state"; out="$dir/watch.out"; log="$dir/send.log"; : > "$log" + fire=$(inbox_lib "$state" fm_task_inbox_write "$state" t1 "one-shot steer" fire-and-forget) + age_path "$fire" + inbox_lib "$state" fm_task_inbox_mark_retry "$state" t1 "$fire" + age_path "$state/t1.inbox/.retry-ring" + watch_bg "$state" "$dir/fakebin" "$out" \ + FM_CONFIG_OVERRIDE="$dir/config" \ + FM_SEND_LOG="$log" FM_FAKE_TMUX_CAPTURE="$(idle_capture "$dir")" \ + FM_TASK_INBOX_RING_MAX=1 + pid=$! + while [ "$i" -lt 100 ]; do + grep -qF 'Firstmate instruction waiting' "$log" 2>/dev/null && break + kill -0 "$pid" 2>/dev/null || break + sleep 0.1 + i=$((i + 1)) + done + sleep 3 + kill -0 "$pid" 2>/dev/null \ + || fail "a fire-and-forget retry must not wake firstmate (watcher exited):"$'\n'"$(cat "$out")" + kill "$pid" 2>/dev/null; wait "$pid" 2>/dev/null + rings=$(grep -cF 'Firstmate instruction waiting' "$log" || true) + [ "$rings" = 1 ] || fail "expected exactly one retry ring, got $rings:"$'\n'"$(cat "$log")" + [ ! -s "$state/.wake-queue" ] || fail "a fire-and-forget retry queued a wake:"$'\n'"$(cat "$state/.wake-queue")" + [ ! -e "$state/t1.inbox/.retry-ring" ] || fail "the watcher did not spend the retry mark" + [ ! -e "$state/t1.inbox/.ring-state" ] || fail "a fire-and-forget retry entered the re-ring ladder" + [ -f "$fire" ] || fail "the retry ring removed the durable record" + pass "watcher: a fire-and-forget record's owed retry rings exactly once and never escalates" +} + +# One watcher inbox check against an idle pane, through the production watcher +# functions, so a status log the case writes is not also read as a wake. +steer_check_once() { # + PATH="$1/fakebin:$PATH" FM_STATE_OVERRIDE="$1/state" FM_SEND_LOG="$1/send.log" \ + FM_FAKE_TMUX_CAPTURE="$(idle_capture "$1")" FM_TASK_INBOX_GRACE_SECS=1 \ + bash -c '. "$1" && inbox_steer_check sess:fm-t1 t1' _ "$WATCH" >/dev/null 2>&1 +} + +test_watcher_holds_retry_while_the_worker_decides() { + local dir state log fire rings + dir=$(setup_watch_case faf-retry-decision) + mkdir -p "$dir/config" + : > "$dir/config/wait-no-turns" + export FM_CONFIG_OVERRIDE="$dir/config" + state="$dir/state"; log="$dir/send.log"; : > "$log" + fire=$(inbox_lib "$state" fm_task_inbox_write "$state" t1 "one-shot steer" fire-and-forget) + age_path "$fire" + inbox_lib "$state" fm_task_inbox_mark_retry "$state" t1 "$fire" + age_path "$state/t1.inbox/.retry-ring" + printf 'needs-decision [key=pick]: ship alpha or beta?\n' > "$state/t1.status" + steer_check_once "$dir" + steer_check_once "$dir" + [ ! -s "$log" ] || fail "the retry rang a worker waiting on its own decision:"$'\n'"$(cat "$log")" + [ -e "$state/t1.inbox/.retry-ring" ] || fail "the held retry lost its mark" + + printf 'resolved [key=pick]: alpha\n' >> "$state/t1.status" + steer_check_once "$dir" + steer_check_once "$dir" + rings=$(grep -cF 'Firstmate instruction waiting' "$log" || true) + [ "$rings" = 1 ] || fail "expected exactly one retry ring once the decision closed, got $rings:"$'\n'"$(cat "$log")" + [ ! -e "$state/t1.inbox/.retry-ring" ] || fail "the watcher did not spend the retry mark" + unset FM_CONFIG_OVERRIDE + pass "watcher: a fire-and-forget retry waits out the worker's own decision, then rings once" +} + +test_watcher_retry_keeps_a_newer_mark() { + local dir state log fire newer rings + dir=$(setup_watch_case faf-retry-newer) + mkdir -p "$dir/config" + : > "$dir/config/wait-no-turns" + export FM_CONFIG_OVERRIDE="$dir/config" + state="$dir/state"; log="$dir/send.log"; : > "$log" + fire=$(inbox_lib "$state" fm_task_inbox_write "$state" t1 "one-shot steer" fire-and-forget) + age_path "$fire" + inbox_lib "$state" fm_task_inbox_mark_retry "$state" t1 "$fire" + age_path "$state/t1.inbox/.retry-ring" + newer=$(inbox_lib "$state" fm_task_inbox_write "$state" t1 "newer steer" fire-and-forget) + FM_RING_MARKS_RETRY="$newer" steer_check_once "$dir" + rings=$(grep -cF 'Firstmate instruction waiting' "$log" || true) + [ "$rings" = 1 ] || fail "expected the owed retry to ring once, got $rings:"$'\n'"$(cat "$log")" + [ "$(cat "$state/t1.inbox/.retry-ring" 2>/dev/null)" = "${newer##*/}" ] \ + || fail "the spent retry removed a newer record's mark written during its ring" + age_path "$state/t1.inbox/.retry-ring" + steer_check_once "$dir" + rings=$(grep -cF 'Firstmate instruction waiting' "$log" || true) + [ "$rings" = 2 ] || fail "the newer record's retry did not ring, got $rings:"$'\n'"$(cat "$log")" + [ ! -e "$state/t1.inbox/.retry-ring" ] || fail "the watcher did not spend the newer retry mark" + unset FM_CONFIG_OVERRIDE + pass "watcher: spending a retry keeps a newer record's mark written during its ring" +} + test_watcher_escalates_once_after_budget() { local dir state out log pid rec rings dir=$(setup_watch_case escalate) @@ -812,12 +967,17 @@ test_concurrent_writers_never_clobber test_writer_retries_after_a_vanished_lock_collision test_ladder_writes_ignore_vanished_inbox test_fire_and_forget_records_never_enter_the_ladder +test_fire_and_forget_retry_is_owed_once +test_fire_and_forget_retry_is_quiet_without_the_flag test_ring_ladder_policy test_watcher_rerings_idle_pane_quietly test_watcher_waits_on_busy_pane test_watcher_quiet_on_healthy_inbox test_watcher_ack_silences_unwritable_ladder test_watcher_surfaces_unwritable_ladder +test_watcher_pays_fire_and_forget_retry_once +test_watcher_holds_retry_while_the_worker_decides +test_watcher_retry_keeps_a_newer_mark test_watcher_escalates_once_after_budget test_watcher_dead_pane_escalates_once_without_ringing test_watcher_dead_pane_ignores_stale_busy_state From aedb7bbf3b038672cef8b700df2b65ff4491dbfd Mon Sep 17 00:00:00 2001 From: slnkjthien <215876738+slnkjthien@users.noreply.github.com> Date: Wed, 30 Sep 2026 22:22:33 -0400 Subject: [PATCH 06/33] fix(bin): close Gerrit-landed backlog items with the change URL as a note (#6140) * fix(bin): record Gerrit change URLs as close notes Teardown's backlog_done_args hands every ship's recorded pr= URL to fm_backlog_done as --pr, and tasks-axi refuses any --pr that is not a canonical GitHub or Forgejo pull request. A Gerrit change URL therefore left the item In flight after cleanup, and the pending backlog-close record replayed into the same refusal at every session start. fm_backlog_done now rewrites a --pr whose value fm_pr_url_parse reads as a Gerrit change into --note "Gerrit change ". The mapping sits at the tasks-axi call rather than in the pending-close record, so records already written with --pr replay to a close unchanged. The captain-held retain path records the URL in its deliverable line and skips the update --pr it cannot make. * no-mistakes(review): Note retained Gerrit change URL when captain answers early * no-mistakes(document): Document Gerrit change URL handling in captain-hold retention --- bin/fm-backlog-transition-lib.sh | 39 +++++++-- bin/fm-captain-hold.sh | 7 +- docs/captain-hold-lifecycle.md | 2 + tests/fm-backlog-atomicity.test.sh | 40 +++++++++ tests/fm-captain-hold-lifecycle.test.sh | 105 ++++++++++++++++++++++++ tests/fm-teardown.test.sh | 46 +++++++++++ 6 files changed, 233 insertions(+), 6 deletions(-) diff --git a/bin/fm-backlog-transition-lib.sh b/bin/fm-backlog-transition-lib.sh index d7dc67bee53..7d73826034f 100644 --- a/bin/fm-backlog-transition-lib.sh +++ b/bin/fm-backlog-transition-lib.sh @@ -78,6 +78,13 @@ FM_BACKLOG_CLOSE_REPLAY_RESULT= # library does not source fm-tasks-axi-lib.sh does not apply. # shellcheck source=bin/fm-timeout-lib.sh disable=SC1091 . "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/fm-timeout-lib.sh" +# fm-pr-lib.sh owns which URL is a Gerrit change. It is functions and empty +# globals only, so it is sourced once rather than re-initialising a caller's +# parsed identity. +if ! declare -F fm_pr_url_parse >/dev/null 2>&1; then + # shellcheck source=bin/fm-pr-lib.sh disable=SC1091 + . "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/fm-pr-lib.sh" +fi # Latched when a row read hits its bound. fm_backlog_row_show runs inside a # command substitution, so the subshell can READ this latch but cannot set it; @@ -509,16 +516,34 @@ fm_backlog_start() { # fm_backlog_mutate "$1" start "$2" } +# tasks-axi takes a --pr link only as a canonical GitHub or Forgejo pull request +# and refuses anything else, so a Gerrit change URL is recorded on the row as a +# note instead. The subshell keeps the parse from overwriting a caller's +# FM_PR_* identity. +fm_backlog_pr_is_gerrit_change() { # + ( fm_pr_url_parse "$1" && [ "$FM_PR_PROVIDER" = gerrit ] ) +} + fm_backlog_done() { # [flag...] - local data=$1 id=$2 + local data=$1 id=$2 arg previous_arg='' + local -a done_args=() shift 2 - fm_backlog_mutate "$data" "done" "$id" "$@" + for arg in "$@"; do + if [ "$previous_arg" = --pr ] && fm_backlog_pr_is_gerrit_change "$arg"; then + done_args[${#done_args[@]}-1]=--note + done_args+=("Gerrit change $arg") + else + done_args+=("$arg") + fi + previous_arg=$arg + done + fm_backlog_mutate "$data" "done" "$id" "${done_args[@]+"${done_args[@]}"}" } fm_backlog_row_artifact_supported() { local id=$1 flag=${2:-} value=${3:-} case "$flag" in - --pr) return 0 ;; + --pr) ! fm_backlog_pr_is_gerrit_change "$value" ;; --report) [ "$value" = "data/$id/report.md" ] ;; *) return 1 ;; esac @@ -550,8 +575,12 @@ fm_backlog_retain() { # [flag...] fi ;; --pr) - deliverable="${deliverable:+$deliverable; }PR $arg" - row_args=(--pr "$arg") + if fm_backlog_row_artifact_supported "$id" --pr "$arg"; then + deliverable="${deliverable:+$deliverable; }PR $arg" + row_args=(--pr "$arg") + else + deliverable="${deliverable:+$deliverable; }Gerrit change $arg" + fi ;; --note) deliverable="${deliverable:+$deliverable; }$arg" ;; esac diff --git a/bin/fm-captain-hold.sh b/bin/fm-captain-hold.sh index 880926494c2..8d54030703c 100755 --- a/bin/fm-captain-hold.sh +++ b/bin/fm-captain-hold.sh @@ -948,6 +948,7 @@ report_retained_artifact_failure() { # apply_pending_retained_artifact() { # local id=$1 marker local -a args=() + RETAINED_CLOSE_ARGS=() marker=$(fm_backlog_close_marker_path "$STATE" "$id") || return 1 [ -e "$marker" ] || [ -L "$marker" ] || return 0 fm_backlog_close_marker_validate "$marker" "$DATA" "$id" "$STATE" \ @@ -956,6 +957,10 @@ apply_pending_retained_artifact() { # args=("${FM_BACKLOG_CLOSE_VALIDATED_ARGS[@]+"${FM_BACKLOG_CLOSE_VALIDATED_ARGS[@]}"}") case "${args[0]-}" in --pr|--report) + if [ "${args[0]}" = --pr ] && fm_backlog_pr_is_gerrit_change "${args[1]-}"; then + RETAINED_CLOSE_ARGS=(--note "Gerrit change ${args[1]}") + return 0 + fi fm_backlog_row_artifact_supported "$id" "${args[@]}" || return 0 fm_backlog_mutate "$DATA" update "$id" "${args[@]}" \ || { report_retained_artifact_failure "$id" "$marker"; return 1; } @@ -968,7 +973,7 @@ close_answered() { # tasks_axi unhold "$1" >/dev/null else apply_pending_retained_artifact "$1" || return 1 - tasks_axi "done" "$1" >/dev/null + tasks_axi "done" "$1" "${RETAINED_CLOSE_ARGS[@]+"${RETAINED_CLOSE_ARGS[@]}"}" >/dev/null fi } diff --git a/docs/captain-hold-lifecycle.md b/docs/captain-hold-lifecycle.md index 7b0c3ddfe86..7a186a4e6af 100644 --- a/docs/captain-hold-lifecycle.md +++ b/docs/captain-hold-lifecycle.md @@ -140,6 +140,7 @@ After cleanup, and still under the task's own lock, teardown does three things: - It records one `Deliverable of the finished work: ...` line at the end of the task body. - It copies a supported pull request or canonical `data//report.md` into the row's structured artifact fields. + A Gerrit change URL is not a pull request tasks-axi accepts, so it appears only in the deliverable line. - It runs `tasks-axi reopen`. The row returns to Queued with its hold intact. @@ -152,6 +153,7 @@ That record carries the retention intent as a `mode=retain` line. An interrupted cleanup therefore replays the retention at the next session start through the same record, validator, and lock as an ordinary close, and never closes the row. If the captain answers before replay, `answer` validates that record and copies any supported retained pull request or report into the row before closing it. +A retained Gerrit change URL is instead recorded as a `Gerrit change ` note on that close. Replay then retires the record. ### Known retained-delivery gaps diff --git a/tests/fm-backlog-atomicity.test.sh b/tests/fm-backlog-atomicity.test.sh index 1bb88a45538..55d62044bed 100755 --- a/tests/fm-backlog-atomicity.test.sh +++ b/tests/fm-backlog-atomicity.test.sh @@ -2030,6 +2030,45 @@ test_recovery_replays_a_close_an_interrupted_cleanup_left_open() { pass "session start finishes a close an interrupted cleanup recorded but never landed" } +test_recovery_replays_a_gerrit_close_with_its_change_url_as_a_note() { + local case_dir id out real_tasks_axi gerrit_url=https://gerrit.example.com/c/project/+/12345 + id=atomic-heal-gerrit-b9 + case_dir=$(make_home heal-pending-gerrit-close) + add_item "$case_dir" "$id" + start_item "$case_dir" "$id" + # The record a pre-fix teardown left: the Gerrit change URL as a --pr link. + printf 'id=%s\ndata=%s\nspawn_gen=spawn-heal-gerrit\narg=--pr\narg=%s\n' \ + "$id" "$(home_of "$case_dir")/data" "$gerrit_url" \ + > "$(home_of "$case_dir")/state/$id.backlog-close" + # Pin the refusal tasks-axi applies to a --pr link that is not a canonical + # GitHub pull request, so this case keeps reproducing whatever the installed + # release accepts. + real_tasks_axi=$(command -v tasks-axi) + cat > "$case_dir/fakebin/tasks-axi" </dev/null \ + || fail "the replayed Gerrit close did not record its change URL as a note" + assert_absent "$(home_of "$case_dir")/state/$id.backlog-close" \ + "a replayed Gerrit close left its record behind" + pass "session start replays a recorded Gerrit close with its change URL as a note" +} + test_recovery_backfills_a_recorded_link_on_an_already_done_item() { local case_dir id marker out id=atomic-heal-done-backfill-b9 @@ -3056,6 +3095,7 @@ test_recovery_marks_an_owned_record_in_flight test_recovery_rejects_an_internal_worker_record_symlink test_recovery_ignores_a_symlinked_worker_record test_recovery_replays_a_close_an_interrupted_cleanup_left_open +test_recovery_replays_a_gerrit_close_with_its_change_url_as_a_note test_recovery_backfills_a_recorded_link_on_an_already_done_item test_recovery_preserves_a_close_when_the_backlog_cannot_be_read test_recovery_retry_preserves_incomplete_cleanup_warning diff --git a/tests/fm-captain-hold-lifecycle.test.sh b/tests/fm-captain-hold-lifecycle.test.sh index 5ecd51fe9ba..a20a8c5de52 100755 --- a/tests/fm-captain-hold-lifecycle.test.sh +++ b/tests/fm-captain-hold-lifecycle.test.sh @@ -3017,6 +3017,63 @@ SH pass "an answer before cleanup replay preserves the retained report" } +test_answer_before_cleanup_replay_notes_a_retained_gerrit_change() { + local home id repo wt rc show real_tasks_axi gerrit_url=https://gerrit.example.com/c/project/+/12345 + home=$(make_home answer-before-replay-gerrit) + id=sample-answer-before-replay-gerrit + repo="$home/projects/sample" + wt="$home/projects/$id" + fm_git_worktree "$repo" "$wt" fm/answer-before-replay-gerrit + tasks_in "$home" add "$id" "Ship the held Gerrit change" --kind ship \ + --repo sample --start >/dev/null || fail "could not create the held Gerrit answer fixture" + fm_write_meta "$home/state/$id.meta" \ + "window=firstmate:fm-$id" "endpoint_task_id=$id" "worktree=$wt" \ + "project=$repo" "harness=codex" "kind=ship" "mode=no-mistakes" \ + "pr=$gerrit_url" "spawn_gen=fixture-$id" + printf 'done: change landed\n' > "$home/state/$id.status" + run_captain "$home" hold "$id" --reason "captain must choose the follow-up" >/dev/null \ + || fail "could not hold the landed Gerrit task for the captain" + real_tasks_axi=$(command -v tasks-axi) + cat > "$home/fakebin/tasks-axi" < "$home/fakebin/treehouse" <<'SH' +#!/usr/bin/env bash +exit 1 +SH + chmod +x "$home/fakebin/treehouse" + + set +e + PATH="$home/fakebin:$PATH" FM_ROOT_OVERRIDE="$ROOT" FM_HOME="$home" \ + FM_STATE_OVERRIDE="$home/state" FM_DATA_OVERRIDE="$home/data" \ + FM_CONFIG_OVERRIDE="$home/config" "$TEARDOWN" "$id" --force \ + > "$home/teardown.out" 2> "$home/teardown.err" + rc=$? + set -e + [ "$rc" -ne 0 ] || fail "cleanup succeeded despite the failed worktree return" + assert_present "$home/state/$id.backlog-close" \ + "the interrupted cleanup lost its retained-artifact record" + + printf 'Proceed with the landed change.\n' > "$home/answer.txt" + run_captain "$home" answer "$id" --decision-file "$home/answer.txt" >/dev/null \ + || fail "the captain could not answer a Gerrit task before cleanup replay" + show=$(tasks_in "$home" show "$id" --full) || fail "the answered Gerrit row is gone" + assert_contains "$show" "state: done" "the answer did not close the Gerrit row" + assert_contains "$show" "Gerrit change $gerrit_url" \ + "the answer dropped the retained Gerrit change URL" + pass "an answer before cleanup replay notes the retained Gerrit change" +} + test_unusable_pending_close_record_names_its_reason() { local home id wt rc err marker home=$(make_home unusable-pending-close-reason) @@ -3195,6 +3252,52 @@ EOF pass "cleanup retains captain calls in the configured backlog" } +test_teardown_retains_a_gerrit_captain_call_with_its_change_url() { + local home id repo wt show real_tasks_axi gerrit_url=https://gerrit.example.com/c/project/+/12345 + home=$(make_home teardown-held-gerrit) + id=sample-held-gerrit + repo="$home/projects/sample" + wt="$home/projects/$id" + fm_git_worktree "$repo" "$wt" fm/held-gerrit + tasks_in "$home" add "$id" "Ship the held Gerrit change" --kind ship \ + --repo sample --start >/dev/null || fail "could not create the held Gerrit fixture" + fm_write_meta "$home/state/$id.meta" \ + "window=firstmate:fm-$id" "endpoint_task_id=$id" "worktree=$wt" \ + "project=$repo" "harness=codex" "kind=ship" "mode=no-mistakes" \ + "pr=$gerrit_url" "spawn_gen=fixture-$id" + printf 'done: change landed\n' > "$home/state/$id.status" + run_captain "$home" hold "$id" --reason "captain must choose the follow-up" >/dev/null \ + || fail "could not hold the landed Gerrit task for the captain" + # Pin the refusal tasks-axi applies to a --pr link that is not a canonical + # GitHub pull request, so this case keeps reproducing whatever the installed + # release accepts. + real_tasks_axi=$(command -v tasks-axi) + cat > "$home/fakebin/tasks-axi" < "$home/teardown.out" 2> "$home/teardown.err" \ + || fail "cleanup of a captain-held Gerrit task failed: $(cat "$home/teardown.err")" + show=$(tasks_in "$home" show "$id" --full) || fail "the captain-held Gerrit row is gone after cleanup" + assert_contains "$show" "state: queued" "the held Gerrit row still reads as worked on" + assert_contains "$show" "hold_kind: captain" "cleanup dropped the captain hold" + assert_contains "$show" "Deliverable of the finished work: Gerrit change $gerrit_url" \ + "the Gerrit change URL was not recorded on the still-open row" + assert_absent "$home/state/$id.backlog-close" \ + "successful cleanup left its pending transition record behind" + pass "cleanup keeps a captain-held Gerrit task open and records its change URL" +} + test_merge_approval_releases_before_zero_done_retention() { local home id archive repo wt pr show home=$(make_home zero-done-retention) @@ -4066,9 +4169,11 @@ test_teardown_never_closes_a_captain_held_task test_retained_row_artifacts_survive_captain_answers test_interrupted_cleanup_keeps_the_captain_call_recoverable test_answer_before_cleanup_replay_preserves_the_retained_report +test_answer_before_cleanup_replay_notes_a_retained_gerrit_change test_unusable_pending_close_record_names_its_reason test_relocated_report_does_not_wedge_an_answer_before_replay test_teardown_retains_captain_calls_in_a_relocated_backlog +test_teardown_retains_a_gerrit_captain_call_with_its_change_url test_merge_approval_releases_before_zero_done_retention test_pr_merge_entrypoint_refuses_a_captain_held_task test_local_merge_entrypoint_refuses_a_captain_held_task diff --git a/tests/fm-teardown.test.sh b/tests/fm-teardown.test.sh index 53a6bd726cd..a53d66b70b2 100755 --- a/tests/fm-teardown.test.sh +++ b/tests/fm-teardown.test.sh @@ -727,6 +727,51 @@ test_teardown_closes_the_backlog_item_itself() { pass "teardown closes its own backlog item before reporting success" } +test_teardown_closes_a_gerrit_task_with_its_change_url_as_a_note() { + local case_dir out real_tasks_axi gerrit_url=https://gerrit.example.com/c/project/+/12345 + case_dir=$(make_case tasks-axi-close-gerrit) + write_meta "$case_dir" no-mistakes ship + printf 'pr=%s\n' "$gerrit_url" >> "$case_dir/state/task-x1.meta" + seed_backlog_in_flight "$case_dir" + # Pin the refusal tasks-axi applies to a --pr link that is not a canonical + # GitHub pull request, so this case keeps reproducing whatever the installed + # release accepts. + real_tasks_axi=$(command -v tasks-axi) + cat > "$case_dir/fakebin/tasks-axi" <&1) || fail "teardown of a landed Gerrit task failed: $out" + [ "$(backlog_row_state "$case_dir")" = "done" ] \ + || fail "teardown left a landed Gerrit task's backlog item at $(backlog_row_state "$case_dir"): $out" + tasks-axi show task-x1 --file "$case_dir/data/backlog.md" --full \ + | grep -F "body: \"Gerrit change $gerrit_url\"" >/dev/null \ + || fail "closed Gerrit backlog item did not record its change URL as a note" + assert_absent "$case_dir/state/task-x1.backlog-close" \ + "a landed Gerrit close left its pending-close record behind" + + case_dir=$(make_case tasks-axi-close-github-under-refusal) + write_meta "$case_dir" no-mistakes ship + printf '%s\n' 'pr=https://github.com/example/repo/pull/7' >> "$case_dir/state/task-x1.meta" + seed_backlog_in_flight "$case_dir" + cp "$TMP_ROOT/tasks-axi-close-gerrit/fakebin/tasks-axi" "$case_dir/fakebin/tasks-axi" + out=$(run_teardown "$case_dir" 2>&1) || fail "teardown of a landed GitHub task failed: $out" + tasks-axi show task-x1 --file "$case_dir/data/backlog.md" \ + | grep -F 'links: "pr:https://github.com/example/repo/pull/7"' >/dev/null \ + || fail "a GitHub pull request no longer closed as the item's pr link" + pass "teardown closes a landed Gerrit task with its change URL as a note and a GitHub task with --pr" +} + test_teardown_manual_backend_leaves_the_backlog_to_the_operator() { local case_dir out backlog_path case_dir=$(make_case tasks-axi-manual-optout) @@ -4252,6 +4297,7 @@ test_forced_secondmate_own_missing_adapter_sibling_refuses_before_child_cleanup test_retained_sources_still_reach_the_ordinary_refusal test_local_only_fork_remote_allows test_teardown_closes_the_backlog_item_itself +test_teardown_closes_a_gerrit_task_with_its_change_url_as_a_note test_teardown_manual_backend_leaves_the_backlog_to_the_operator test_local_only_truly_unpushed_refuses test_local_only_merged_to_local_main_allows From 549e07f37fd73aa01d74cd126b1111c99175abed Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Wed, 30 Sep 2026 21:25:18 -0700 Subject: [PATCH 07/33] fix: reduce remote-job and supervision polling churn (#6255) * perf(remote): separate active job sampling from dispatcher cadence * no-mistakes(document): Link remote wait timing to its authoritative contract * no-mistakes(ci): Fixed ci-1 with two narrowly scoped SC2030 annotations documenting intentional subshell-local legacy and active cadence overrides in tests/fm-remote-job.test.sh. Runtime behavior is unchanged. Reproduced the lint failure before the fix; afterward ShellCheck 0.11.0 with source following, Bash syntax validation, the complete remote-job behavior suite, and git diff --check all passed * perf(supervision): reduce park, delta and dispatcher polling * no-mistakes(document): Clarify poll latency contracts and authoritative documentation pointers --- bin/fm-remote-delta-read.sh | 7 ++- bin/fm-remote-job-lib.sh | 15 +++++- bin/fm-remote-job-worker.sh | 13 ++--- bin/fm-supervision-host.sh | 36 ++++++++------ docs/remote-secondmates.md | 4 +- docs/supervision-host.md | 2 +- tests/fm-remote-job.test.sh | 80 +++++++++++++++++++++++++++++++ tests/fm-remote-reply.test.sh | 30 ++++++++++++ tests/fm-supervision-host.test.sh | 27 +++++++++++ 9 files changed, 190 insertions(+), 24 deletions(-) diff --git a/bin/fm-remote-delta-read.sh b/bin/fm-remote-delta-read.sh index d4c26bd6697..8450628d568 100755 --- a/bin/fm-remote-delta-read.sh +++ b/bin/fm-remote-delta-read.sh @@ -10,6 +10,11 @@ # the source. A shortened or changed prefix returns a structured continuity-break # result instead of silently rebasing the cursor. # +# An unchanged snapshot is retried after FM_REMOTE_DELTA_POLL_SECONDS (default +# 0.5 seconds). A complete line is visible on the next sample, and the window +# deadline can overshoot by that interval plus snapshot and scheduling work. +# The wait remains an ordinary child sleep; signal handling is unchanged. +# # Exit 75 means the wait window closed with no complete line. SIGTERM exits the # same way after cleanup. The remote job worker preempts this read-only poll to # unblock any queued command other than another reply long-poll, then publishes @@ -19,7 +24,7 @@ set -eu FM_HOME=${FM_HOME:?FM_HOME is required} MAX_BYTES=${FM_REMOTE_DELTA_MAX_BYTES:-65536} -POLL_SECONDS=${FM_REMOTE_DELTA_POLL_SECONDS:-0.2} +POLL_SECONDS=${FM_REMOTE_DELTA_POLL_SECONDS:-0.5} die() { printf 'error: %s\n' "$1" >&2; exit 1; } usage() { sed -n '2,11p' "$0" | sed 's/^# \{0,1\}//'; exit 2; } diff --git a/bin/fm-remote-job-lib.sh b/bin/fm-remote-job-lib.sh index 68d3b62c064..a60daf41d01 100755 --- a/bin/fm-remote-job-lib.sh +++ b/bin/fm-remote-job-lib.sh @@ -55,6 +55,18 @@ # Abandoned .stage.* staging litter older than # FM_REMOTE_JOB_STAGE_REAP_SECONDS is reaped by the worker's stale sweep. # +# Result consumers and active-command monitors sample every 0.25 seconds by +# default; the dispatcher's post-activity burst still samples every 0.05 seconds. +# FM_REMOTE_JOB_ACTIVE_POLL_SECONDS overrides the active/result interval; an +# explicitly supplied FM_REMOTE_JOB_POLL_SECONDS remains the legacy fallback +# for both intervals. Resolve the active default before filling the dispatcher +# default, and retain it when the library is sourced again. +# Once-per-second cancellation, preemption, and disconnect checks can overshoot +# their due time by one sampling interval plus work/scheduling time, as can the +# active command's timeout check. Completion and result collection can each add +# one interval. Sleeps stay ordinary child processes: existing signal handlers +# and the separate cancellation/preemption TERM-to-KILL grace are unchanged. +# # The worker accepts only a tracked, non-symlink executable named fm-*.sh below # its configured FM_ROOT/bin. Every child receives env -i with the composed # PATH, HOME, FM_HOME, FM_ROOT_OVERRIDE, and FM_REMOTE_JOB_ACTIVE=1. The PATH @@ -88,6 +100,7 @@ FM_REMOTE_JOB_MAX_BYTES=${FM_REMOTE_JOB_MAX_BYTES:-1048576} FM_REMOTE_JOB_QUEUE_TIMEOUT=${FM_REMOTE_JOB_QUEUE_TIMEOUT:-360} FM_REMOTE_JOB_TIMEOUT=${FM_REMOTE_JOB_TIMEOUT:-360} FM_REMOTE_JOB_WAIT_GRACE=${FM_REMOTE_JOB_WAIT_GRACE:-30} +FM_REMOTE_JOB_ACTIVE_POLL_SECONDS=${FM_REMOTE_JOB_ACTIVE_POLL_SECONDS:-${FM_REMOTE_JOB_POLL_SECONDS:-0.25}} FM_REMOTE_JOB_POLL_SECONDS=${FM_REMOTE_JOB_POLL_SECONDS:-0.05} FM_REMOTE_JOB_REAP_SECONDS=${FM_REMOTE_JOB_REAP_SECONDS:-3600} FM_REMOTE_JOB_STAGE_REAP_SECONDS=${FM_REMOTE_JOB_STAGE_REAP_SECONDS:-600} @@ -733,7 +746,7 @@ fm_remote_job_wait() { # ; honors FM_REMOTE_JOB_DISCONNECT_PR return 1 fi fi - sleep "$FM_REMOTE_JOB_POLL_SECONDS" + sleep "$FM_REMOTE_JOB_ACTIVE_POLL_SECONDS" done } diff --git a/bin/fm-remote-job-worker.sh b/bin/fm-remote-job-worker.sh index 8973f5d6dae..73191029ff0 100755 --- a/bin/fm-remote-job-worker.sh +++ b/bin/fm-remote-job-worker.sh @@ -23,11 +23,12 @@ # worker's orphan recovery. # # The serving loop does not busy-poll an idle queue. After a lane starts or is -# reaped it rescans every FM_REMOTE_JOB_POLL_SECONDS for 20 passes, so a home +# reaped it rescans every FM_REMOTE_JOB_POLL_SECONDS for four passes, so a home # whose lane just finished starts its next job promptly; otherwise it sleeps -# one second between passes. That bound is how long newly staged or cancelled -# work, a lane that died, an orphaned claim, or an expired queue deadline can -# wait for the next pass, and it refreshes the readiness heartbeat about once +# one second between passes. Work arriving after the four-pass burst may wait +# for that quiet scan. Newly staged or cancelled work, a lane that died, an +# orphaned claim, or an expired queue deadline can wait that interval plus +# scan work and scheduling time. It refreshes the readiness heartbeat about once # per second, far inside the probe's 10-second freshness bound. The stale # sweep, whose state preparation also re-applies the queue directories' 0700 # modes, runs at startup and then at most every 60 seconds, never more rarely @@ -61,7 +62,7 @@ FM_REMOTE_JOB_ORPHAN_GRACE_SECONDS=$(worker_bounded_setting "${FM_REMOTE_JOB_ORP FM_REMOTE_JOB_SUPERVISOR_MAX_RESTARTS=$(worker_bounded_setting "${FM_REMOTE_JOB_SUPERVISOR_MAX_RESTARTS:-}" 20) FM_REMOTE_JOB_SUPERVISOR_MAX_BACKOFF_SECONDS=$(worker_bounded_setting "${FM_REMOTE_JOB_SUPERVISOR_MAX_BACKOFF_SECONDS:-}" 5) FM_REMOTE_JOB_SUPERVISOR_HEALTHY_SECONDS=$(worker_bounded_setting "${FM_REMOTE_JOB_SUPERVISOR_HEALTHY_SECONDS:-}" 10) -WORKER_FAST_PASSES=20 +WORKER_FAST_PASSES=4 WORKER_IDLE_WAIT_SECONDS=1 WORKER_SWEEP_SECONDS=60 @@ -739,7 +740,7 @@ worker_run_with_timeout() { # [args...] fi next_check=$((SECONDS + 1)) fi - sleep "$FM_REMOTE_JOB_POLL_SECONDS" + sleep "$FM_REMOTE_JOB_ACTIVE_POLL_SECONDS" done wait "$group_pid" 2>/dev/null rc=$? diff --git a/bin/fm-supervision-host.sh b/bin/fm-supervision-host.sh index c0dd9905075..90fd54836e9 100755 --- a/bin/fm-supervision-host.sh +++ b/bin/fm-supervision-host.sh @@ -155,10 +155,15 @@ # a new engine conversation after this many turns; every main session start # also opens a new one), FM_SUPERVISION_HOST_READY_TIMEOUT (25: how long a # successor cycle may take to verify), FM_SUPERVISION_HOST_POLL (1). +# Park duration uses Bash's process-relative SECONDS counter (including Bash +# 3.2), while durable timestamps still use epoch time. This is not a portable +# monotonic-clock guarantee. Arm exit probes use ordinary 0.5-second child +# sleeps within the unchanged POLL-cadence maintenance and boundary checks; +# close observation and a shell-only caught signal may wait that interval plus +# work/scheduling time. No stop-signal disposition or cleanup bound changes. # FM_TEST_SUPERVISION_HOST_CLOCK names a file holding the park's elapsed -# seconds, which the park and turn boundary checks read in place of the wall -# clock only when FM_TEST_SEAM=1; tests/lib.sh arms the marker for isolated -# suites. +# seconds, which the park and turn boundary checks read in place of SECONDS +# only when FM_TEST_SEAM=1; tests/lib.sh arms the marker for isolated suites. set -u SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -228,6 +233,7 @@ HEALTH_FILE="$STATE/.supervision-host-health" MIRROR_FEED="$STATE/.supervision-host-mirror" HOST_PID=$$ +HOST_STARTED_SECONDS=$SECONDS HOST_STARTED=$(date +%s) GEN="host-$HOST_PID-$HOST_STARTED" TURN_SEQ=0 @@ -442,22 +448,24 @@ start_arm() { # [--restart]; sets the started pi STARTED_ARM_OUT=$out } -park_elapsed() { +park_elapsed() { # Sets PARK_ELAPSED without a production clock/helper fork. if [ "${FM_TEST_SEAM:-}" = 1 ] && [ -n "${FM_TEST_SUPERVISION_HOST_CLOCK:-}" ]; then - numeric_or "$(cat "$FM_TEST_SUPERVISION_HOST_CLOCK" 2>/dev/null)" 0 + PARK_ELAPSED=$(numeric_or "$(cat "$FM_TEST_SUPERVISION_HOST_CLOCK" 2>/dev/null)" 0) return fi - printf '%s\n' $(( $(date +%s) - HOST_STARTED )) + PARK_ELAPSED=$((SECONDS - HOST_STARTED_SECONDS)) } boundary_reached() { - [ "$(park_elapsed)" -ge "$PARK_SECONDS" ] + park_elapsed + [ "$PARK_ELAPSED" -ge "$PARK_SECONDS" ] } # True when an engine turn started now could still be running at the turn # limit (the boundary unless the owner set a later one). turn_crosses_boundary() { - [ $(( $(park_elapsed) + TURN_TIMEOUT + ENGINE_GRACE )) -ge "$PARK_LIMIT" ] + park_elapsed + [ $((PARK_ELAPSED + TURN_TIMEOUT + ENGINE_GRACE)) -ge "$PARK_LIMIT" ] } # End the park at the boundary: stop the current and successor arms and this @@ -471,7 +479,8 @@ boundary_exit() { SUCCESSOR_PID= SUCCESSOR_OUT= "$SCRIPT_DIR/fm-watch-arm.sh" --stop >/dev/null 2>&1 || true - log_line "boundary after $(park_elapsed)s" + park_elapsed + log_line "boundary after ${PARK_ELAPSED}s" emit 'supervision-host: cycle boundary - the host ended its park at its bound; drain, acknowledge, and end the turn, and the next park starts on its own' exit 0 } @@ -496,12 +505,11 @@ await_close() { refresh_process "$ARM_PID" [ "$READY_PENDING" -eq 0 ] || stream_ready_line boundary_reached && return 1 - # The arm's exit is probed at a tenth of a second between POLL-cadence - # checks: the close is read as soon as the arm dies instead of up to POLL - # seconds late, while refresh keeps its per-second cadence. - i=$((POLL * 10)) + # Probe the arm's exit twice a second between POLL-cadence checks, without + # changing the outer identity refresh, readiness, or boundary cadence. + i=$((POLL * 2)) while [ "$i" -gt 0 ] && fm_pid_alive "$ARM_PID"; do - sleep 0.1 + sleep 0.5 i=$((i - 1)) done done diff --git a/docs/remote-secondmates.md b/docs/remote-secondmates.md index eb7537cf84b..e13eebacf9f 100644 --- a/docs/remote-secondmates.md +++ b/docs/remote-secondmates.md @@ -82,7 +82,8 @@ On macOS the worker is `dev.firstmate.remote-job`, an Aqua-scoped LaunchAgent at After that bootstrap, every non-doctor `fm-on.sh` target runs through that worker in the remote account's GUI session. It never runs in the SSH process or a Herdr pane. Linux uses the same queue and worker protocol without the Aqua-session requirement. -When idle, the worker checks for newly staged work about once per second; after a lane starts or finishes it checks more frequently for a short period. +The [`fm-remote-job-worker.sh` header](../bin/fm-remote-job-worker.sh) owns dispatch cadence and the quiet-scan latency for work arriving after its post-activity burst. +Active-command and result waits use a separate sampling interval; the [`fm-remote-job-lib.sh` header](../bin/fm-remote-job-lib.sh) owns its defaults, overrides, and completion, cancellation, and timeout latency contract. ### Job lanes and preemption @@ -512,6 +513,7 @@ A process-event source takes these steps: - It does not carry blank separators. The listener holds its claim across an empty wait and across a delta it re-arms, so a line appended during either is collected without waiting for the next supervision cycle. +The [`fm-remote-delta-read.sh` header](../bin/fm-remote-delta-read.sh) owns snapshot sampling and its line-visibility and wait-window latency contract. It stops when that registration is retired, the registered command changes, or the home's owner lease lapses. `bin/fm-procevent.sh` owns the generic relisten rule, and `bin/fm-procevent-remote-reply.sh` owns this adapter's answer. diff --git a/docs/supervision-host.md b/docs/supervision-host.md index 583925c1a10..6faad8fb2b1 100644 --- a/docs/supervision-host.md +++ b/docs/supervision-host.md @@ -46,7 +46,7 @@ Until they land, their current behavior stays as described in their own owners. | Component | Owner | Role | |---|---|---| -| The loop | `bin/fm-supervision-host.sh` | Its header owns the per-close order, the park boundary, ownership checks, predecessor cleanup, state files, and tunables. | +| The loop | `bin/fm-supervision-host.sh` | Its header owns the per-close order, the park boundary and elapsed clock, arm-exit sampling and signal-observation latency, ownership checks, predecessor cleanup, state files, and tunables. | | The arm owners | Each primary's existing arm owner | Runs the host for a home that runs it and delivers a handed-back wake to main; see [Arm owners](#arm-owners). | | The engine | `bin/fm-supervision-engine-lib.sh` | Owns the home gate, including the default on Claude and the opt-out, the verified-engine list, and one bounded engine turn, including the reap of engine tool processes that outlive it. | | Row eligibility and the offer rule | `bin/fm-branch-dispatch.mjs` | The command entry to `.pi/extensions/lib/fm-branch-dispatch.ts`, so the host and the Pi extension compute branch-claimable rows, their task scope, and whether the branch may take a close (`branchOfferForWake`) from one owner; it also renders the wake message with the same away-posture tail, or the dialog mirror at its head. | diff --git a/tests/fm-remote-job.test.sh b/tests/fm-remote-job.test.sh index b7e9061bc5b..a7a5b96873a 100755 --- a/tests/fm-remote-job.test.sh +++ b/tests/fm-remote-job.test.sh @@ -107,6 +107,85 @@ git -C "$REMOTE_ROOT" config user.name Test git -C "$REMOTE_ROOT" add AGENTS.md bin git -C "$REMOTE_ROOT" commit -qm 'remote job fixture' +# Observe the actual sleep executable boundary for the result consumer, a +# top-level command lane, and the dispatcher. Re-source the public library as +# callers may do; its own dispatcher default must not become a legacy override. +poll_cadence_case() ( + local label=$1 legacy=$2 active=$3 expected=$4 dispatch=$5 poll_dir pid='' i + poll_dir="$TMP_ROOT/poll-$label" + mkdir -p "$poll_dir/bin" + cat > "$poll_dir/bin/sleep" <<'SH' +#!/bin/bash +printf '%s\n' "$1" >> "$FM_POLL_SLEEP_LOG" +exec /bin/sleep "$@" +SH + chmod +x "$poll_dir/bin/sleep" + trap '[ -z "$pid" ] || { kill -TERM "$pid" 2>/dev/null || true; wait "$pid" 2>/dev/null || true; }' EXIT + unset FM_REMOTE_JOB_POLL_SECONDS FM_REMOTE_JOB_ACTIVE_POLL_SECONDS + # shellcheck disable=SC2030 # The legacy override is local to this cadence fixture. + [ -z "$legacy" ] || export FM_REMOTE_JOB_POLL_SECONDS="$legacy" + # shellcheck disable=SC2030 # The active override is local to this cadence fixture. + [ -z "$active" ] || export FM_REMOTE_JOB_ACTIVE_POLL_SECONDS="$active" + export FM_REMOTE_JOB_STATE_ROOT="$poll_dir/state" FM_ROOT_OVERRIDE="$REMOTE_ROOT" + # shellcheck disable=SC2030 # Each cadence fixture owns its subshell's bounds. + export FM_REMOTE_JOB_QUEUE_TIMEOUT=60 FM_REMOTE_JOB_TIMEOUT=30 + # shellcheck disable=SC2030 # The recording executable is local to this fixture. + export PATH="$poll_dir/bin:$PATH" FM_POLL_SLEEP_LOG="$poll_dir/sleeps" + # shellcheck source=bin/fm-remote-job-lib.sh + . "$ROOT/bin/fm-remote-job-lib.sh" + # shellcheck source=bin/fm-remote-job-lib.sh + . "$ROOT/bin/fm-remote-job-lib.sh" + fm_remote_job_stage "$ACCOUNT_HOME" "$REMOTE_ROOT" "$REMOTE_HOME" \ + fm-delay-job.sh 0.8 "$poll_dir/ran" /dev/null || fail "$FM_REMOTE_JOB_ERROR" + # Publish a real bounded result after the caller has entered its wait, without + # a lane's own samples contaminating this consumer-only executable log. + ( + /bin/sleep 0.8 + : > "$FM_REMOTE_JOB_JOBS/$FM_REMOTE_JOB_ID/stdout" + : > "$FM_REMOTE_JOB_JOBS/$FM_REMOTE_JOB_ID/stderr" + printf '0\n' > "$FM_REMOTE_JOB_JOBS/$FM_REMOTE_JOB_ID/exit" + fm_remote_job_write_state "$FM_REMOTE_JOB_JOBS/$FM_REMOTE_JOB_ID" 'done' + ) & + pid=$! + fm_remote_job_wait "$ACCOUNT_HOME" "$FM_REMOTE_JOB_ID" || fail "$FM_REMOTE_JOB_ERROR" + wait "$pid" || fail "$label result producer failed" + pid='' + [ "$FM_REMOTE_JOB_EXIT" -eq 0 ] || fail "$label result consumer lost the exit status" + grep -qx "$expected" "$FM_POLL_SLEEP_LOG" || fail "$label consumer never sampled at $expected seconds" + [ "$(sort -u "$FM_POLL_SLEEP_LOG")" = "$expected" ] || fail "$label consumer used another cadence" + + : > "$FM_POLL_SLEEP_LOG" + fm_remote_job_stage "$ACCOUNT_HOME" "$REMOTE_ROOT" "$REMOTE_HOME" \ + fm-delay-job.sh 0.8 "$poll_dir/ran" /dev/null || fail "$FM_REMOTE_JOB_ERROR" + HOME="$ACCOUNT_HOME" "$BASH" "$REMOTE_ROOT/bin/fm-remote-job-worker.sh" --lane "$FM_REMOTE_JOB_ID" & + pid=$! + wait "$pid" || fail "$label command lane failed" + pid='' + [ -e "$poll_dir/ran" ] || fail "$label lane did not execute its command" + [ "$(fm_remote_job_read_state "$FM_REMOTE_JOB_JOBS/$FM_REMOTE_JOB_ID")" = 'done' ] || fail "$label lane did not publish completion" + grep -qx "$expected" "$FM_POLL_SLEEP_LOG" || fail "$label lane never sampled at $expected seconds" + if [ "$expected" != 0.05 ]; then + ! grep -qx 0.05 "$FM_POLL_SLEEP_LOG" || fail "$label lane still sampled at the dispatcher default" + fi + + : > "$FM_POLL_SLEEP_LOG" + HOME="$ACCOUNT_HOME" "$BASH" "$REMOTE_ROOT/bin/fm-remote-job-worker.sh" > "$poll_dir/worker.log" 2>&1 & + pid=$! + for ((i = 0; i < 200; i++)); do + grep -qx 1 "$FM_POLL_SLEEP_LOG" && break + /bin/sleep 0.05 + done + grep -qx 1 "$FM_POLL_SLEEP_LOG" || fail "$label dispatcher never reached its one-second quiet wait" + [ "$(grep -cx "$dispatch" "$FM_POLL_SLEEP_LOG")" -eq 4 ] || fail "$label dispatcher did not limit its fast burst to four $dispatch-second waits" + kill -TERM "$pid" || fail "$label dispatcher stopped unexpectedly" + wait "$pid" 2>/dev/null || true + pid='' + pass "$label: result and command samples use $expected seconds; dispatcher uses four $dispatch-second waits then one second" +) +poll_cadence_case default '' '' 0.25 0.05 || exit 1 +poll_cadence_case legacy 0.07 '' 0.07 0.07 || exit 1 +poll_cadence_case active 0.07 0.12 0.12 0.07 || exit 1 + DEFAULT_STATE="$TMP_ROOT/default-timeout-jobs" DEFAULT_BOUNDS=$( unset FM_REMOTE_JOB_QUEUE_TIMEOUT @@ -957,6 +1036,7 @@ fi exec '$(command -v sleep)' "\$@" SH chmod +x "$STALL_BIN/sleep" +# shellcheck disable=SC2031 # Cadence fixture PATH changes stayed in their subshells. HOME="$STALL_HOME" FM_ROOT_OVERRIDE="$REMOTE_ROOT" FM_REMOTE_JOB_STATE_ROOT="$STALL_STATE" \ FM_REMOTE_JOB_PLATFORM_OVERRIDE=Linux PATH="$STALL_BIN:$PATH" \ "$REMOTE_ROOT/bin/fm-remote-job-worker.sh" --serve \ diff --git a/tests/fm-remote-reply.test.sh b/tests/fm-remote-reply.test.sh index d6b00c7cead..dfdc212c1b9 100755 --- a/tests/fm-remote-reply.test.sh +++ b/tests/fm-remote-reply.test.sh @@ -122,6 +122,36 @@ sha256_file() { fi } +# Drive the real delta-reader executable across its unchanged-file wait. +# The recording sleep appends a complete line after the initial empty snapshot, +# so the next snapshot must deliver it without consuming or modifying the log. +delta_cadence_case() { + local label=$1 override=$2 expected=$3 dir log empty_hash + dir="$TMP_ROOT/delta-$label" + mkdir -p "$dir/bin" "$dir/home/state" + log="$dir/home/state/replies.status" + : > "$log" + empty_hash=$(sha256_file "$log") + cat > "$dir/bin/sleep" <<'SH' +#!/bin/bash +printf '%s\n' "$1" >> "$FM_DELTA_SLEEP_LOG" +printf 'cadence-delivered\n' >> "$FM_DELTA_APPEND_LOG" +exec /bin/sleep "$@" +SH + chmod +x "$dir/bin/sleep" + FM_HOME="$dir/home" PATH="$dir/bin:$PATH" FM_REMOTE_DELTA_POLL_SECONDS="$override" \ + FM_DELTA_SLEEP_LOG="$dir/sleeps" FM_DELTA_APPEND_LOG="$log" \ + "$BASH" "$ROOT/bin/fm-remote-delta-read.sh" state/replies.status 0 "$empty_hash" 30 \ + > "$dir/result" || fail "$label delta reader failed" + [ "$(cat "$dir/sleeps")" = "$expected" ] || fail "$label delta reader did not wait $expected seconds" + assert_grep 'status=delta' "$dir/result" "$label delta reader did not publish a delta" + assert_grep 'cadence-delivered' "$dir/result" "$label delta reader lost the appended complete line" + [ "$(cat "$log")" = cadence-delivered ] || fail "$label delta reader changed its source log" + pass "$label delta reader waits $expected seconds then delivers a non-destructive complete-line delta" +} +delta_cadence_case default '' 0.5 +delta_cadence_case override 0.07 0.07 + ADAPTER="$ROOT/bin/fm-procevent-remote-reply.sh" SID=$(remote_env "$ADAPTER" source-id ios) out=$(remote_env "$ADAPTER" arm ios) diff --git a/tests/fm-supervision-host.test.sh b/tests/fm-supervision-host.test.sh index 13509f74d4f..3ee445b9309 100755 --- a/tests/fm-supervision-host.test.sh +++ b/tests/fm-supervision-host.test.sh @@ -2037,6 +2037,32 @@ test_restarted_host_stops_what_a_killed_predecessor_left() { pass "host: a restarted host stops, by recorded identity, the cycle a killed predecessor left running" } +test_park_exit_probe_uses_half_second_child_sleeps() { + local home host_pid + home=$(make_home park-cadence attended) + cat > "$home/fakebin/sleep" <<'SH' +#!/bin/bash +pid='' +if [ -f "$FM_HOME/probe-host" ]; then + IFS= read -r pid < "$FM_HOME/probe-host" || true + if [ "$PPID" = "$pid" ]; then + printf '%s\n' "$1" >> "$FM_HOME/park-sleeps" + fi +fi +exec /bin/sleep "$@" +SH + chmod +x "$home/fakebin/sleep" + start_host "$home" + wait_until 150 watcher_live "$home" || fail "park-cadence: no watcher started" + host_pid=$(awk -F '\t' '$1 == "host" { print $2; exit }' "$home/state/.supervision-host") + [ -n "$host_pid" ] || fail "park-cadence: no recorded host" + printf '%s\n' "$host_pid" > "$home/probe-host" + wait_until 100 test -s "$home/park-sleeps" || fail "park-cadence: no child sleep observed" + [ "$(sort -u "$home/park-sleeps")" = 0.5 ] || fail "park-cadence: exit probing did not use half-second sleeps" + stop_home_processes "$home" + pass "host: parked child-exit sampling uses ordinary half-second sleeps" +} + test_park_boundary_ends_the_park_before_the_hook_timeout() { local home token home=$(make_home boundary attended) @@ -2595,6 +2621,7 @@ test_superseded_host_leaves_the_owner_untouched() { pass "host: a host under a superseded auto-arm generation stands down without touching the owner" } +test_park_exit_probe_uses_half_second_child_sleeps test_report_surface_enforces_actor_turn_and_scope test_report_after_the_return_is_queued_for_main test_dispatch_entry_scopes_rows_and_renders_the_away_tail From 589ccec821bf6310ce888e2a702e4fc9258eb1e8 Mon Sep 17 00:00:00 2001 From: guanchengh-lgtm Date: Thu, 1 Oct 2026 18:26:41 +0800 Subject: [PATCH 08/33] fix(bin): load backend sibling libraries when sourced under zsh (#6221) * fix(bin): load backend sibling libraries under zsh fm_backend_source kept each backend's sibling list in one space-separated string and iterated it unquoted. zsh does not word-split an unquoted expansion, so the readability check saw the whole list as one path and refused every backend with more than one sibling. Hold the list in the function's positional parameters instead, which needs no word splitting in Bash 3.2, Bash 5, or zsh. The existing zsh case in tests/fm-backend.test.sh covers it wherever zsh is installed. * test: run the Calm mod suite on stock Bash 3.2 The suite injected shell values into its generated Node scripts with the ${value@Q} transformation, which needs Bash 4.4. Stock macOS Bash 3.2 reports a bad substitution, so every case failed before it asserted anything. Build each JavaScript string literal with JSON.stringify through a small helper instead, which works on any Bash and is a valid literal for any value. * no-mistakes(review): fix(bin): rename zsh-special path local in fm_backend_source * test: narrow the zsh backend claim to name matching Under zsh the adapters locate their siblings through BASH_SOURCE, so a successful fm_backend_source is not a full load. Assert only what the contract states, and pass js_string values after -- so node never reads a leading-dash value as its own option. --------- Co-authored-by: Nova Agent B --- bin/fm-backend.sh | 21 +++++++++++---------- tests/fm-backend.test.sh | 21 ++++++++++++++++++--- tests/fm-calm-claude-mod.test.sh | 29 ++++++++++++++++++----------- 3 files changed, 47 insertions(+), 24 deletions(-) diff --git a/bin/fm-backend.sh b/bin/fm-backend.sh index f4fdde29436..798bb5c599a 100644 --- a/bin/fm-backend.sh +++ b/bin/fm-backend.sh @@ -622,34 +622,35 @@ fm_backend_source_readable() { # } fm_backend_source() { # - local name=$1 adapter rel path siblings + local name=$1 adapter rel sibling fm_backend_validate "$name" || return 1 adapter="$FM_BACKEND_LIB_DIR/backends/$name.sh" + # The sibling list rides in the positional parameters: zsh does not + # word-split an unquoted expansion, so a space-separated string is one path. case "$name" in tmux) - siblings="fm-tmux-lib.sh fm-composer-lib.sh fm-cursor-lib.sh fm-session-lock-lib.sh fm-agent-process-lib.sh fm-gemini-lib.sh" + set -- fm-tmux-lib.sh fm-composer-lib.sh fm-cursor-lib.sh fm-session-lock-lib.sh fm-agent-process-lib.sh fm-gemini-lib.sh ;; herdr) - siblings="fm-composer-lib.sh fm-transition-lib.sh fm-agent-process-lib.sh fm-session-lock-lib.sh fm-gemini-lib.sh" + set -- fm-composer-lib.sh fm-transition-lib.sh fm-agent-process-lib.sh fm-session-lock-lib.sh fm-gemini-lib.sh ;; zellij) - siblings="fm-backend-hometag-lib.sh fm-composer-lib.sh" + set -- fm-backend-hometag-lib.sh fm-composer-lib.sh ;; orca) - siblings="fm-composer-lib.sh" + set -- fm-composer-lib.sh ;; cmux) - siblings="fm-backend-hometag-lib.sh fm-composer-lib.sh" + set -- fm-backend-hometag-lib.sh fm-composer-lib.sh ;; *) return 1 ;; esac fm_backend_source_readable "$adapter" || return 1 - # shellcheck disable=SC2086 # sibling names are a fixed space-separated list - for rel in $siblings; do - path="$FM_BACKEND_LIB_DIR/$rel" - fm_backend_source_readable "$path" || return 1 + for rel in "$@"; do + sibling="$FM_BACKEND_LIB_DIR/$rel" + fm_backend_source_readable "$sibling" || return 1 done case "$name" in tmux) diff --git a/tests/fm-backend.test.sh b/tests/fm-backend.test.sh index 96d00b10303..a9030018d8c 100755 --- a/tests/fm-backend.test.sh +++ b/tests/fm-backend.test.sh @@ -502,17 +502,32 @@ test_backend_validate_refuses_unknown() { } test_backend_source_shell_portable() { - local out status + local out status stub probe # zsh does not word-split unquoted expansions; sourcing fm-backend.sh from # an interactive zsh session must still recognize known backend names. + # The claim is name matching and the sibling precheck only: the adapters + # find their own siblings through BASH_SOURCE, so zsh is not a full load. if command -v zsh >/dev/null 2>&1; then - zsh -c "cd '$ROOT' && source bin/fm-backend.sh && fm_backend_source herdr && whence -w fm_backend_herdr_capture >/dev/null" 2>/dev/null \ - || fail "zsh: fm_backend_source herdr should load the adapter when sourced" + zsh -c "cd '$ROOT' && source bin/fm-backend.sh && fm_backend_source herdr" >/dev/null 2>&1 \ + || fail "zsh: fm_backend_source herdr should accept the known backend name and find its sibling libraries" out=$(zsh -c "cd '$ROOT' && source bin/fm-backend.sh && fm_backend_source bogus" 2>&1) \ && fail "zsh: fm_backend_source bogus should fail" assert_contains "$out" "unknown backend 'bogus'" \ "zsh: fm_backend_source did not reject bogus with the expected error" pass "zsh: fm_backend_source recognizes known backends and rejects unknown ones" + + # zsh ties the lowercase `path` array to PATH; a backend loaded while + # fm_backend_source clobbers PATH cannot resolve external commands. + stub="$TMP_ROOT/zsh-source-path" + probe="$stub/probe" + mkdir -p "$stub/backends" + printf 'command -v dirname > "%s"\n' "$probe" > "$stub/backends/orca.sh" + : > "$stub/fm-composer-lib.sh" + zsh -c "cd '$ROOT' && source bin/fm-backend.sh && FM_BACKEND_LIB_DIR='$stub' && fm_backend_source orca" >/dev/null 2>&1 \ + || fail "zsh: fm_backend_source orca should load a stub adapter" + [ -s "$probe" ] \ + || fail "zsh: fm_backend_source clobbered PATH while loading a backend adapter" + pass "zsh: fm_backend_source keeps PATH intact while loading a backend adapter" else pass "zsh: shell-portable backend matching skipped (zsh not found)" fi diff --git a/tests/fm-calm-claude-mod.test.sh b/tests/fm-calm-claude-mod.test.sh index 69ab66558e0..ce5dcf8a869 100644 --- a/tests/fm-calm-claude-mod.test.sh +++ b/tests/fm-calm-claude-mod.test.sh @@ -33,6 +33,13 @@ run_node() { # node --input-type=module <"$1" } +# js_string : a JavaScript string literal for a shell value, for the +# generated scripts below. ${value@Q} would need Bash 4.4 and yields shell +# quoting; stock macOS Bash 3.2 reports a bad substitution. +js_string() { # + node -e 'process.stdout.write(JSON.stringify(process.argv[1]))' -- "$1" +} + test_plugin_shape() { local link resolved autoload link="$ROOT/.agents/skills/firstmate-calm" @@ -49,7 +56,7 @@ test_plugin_shape() { [ ! -e "$MOD/SKILL.md" ] || fail "the mod carries a SKILL.md and would load as a skill on every harness" cat >"$TMP_ROOT/shape.mjs" <"$TMP_ROOT/sprite.mjs" <"$TMP_ROOT/raster.mjs" < { if (!condition) throw new Error(message); }; for (let length = 0; length <= 80; length += 1) { const bytes = new Uint8Array(randomBytes(length)); @@ -235,8 +242,8 @@ test_presentation_policy() { local out cat >"$TMP_ROOT/policy.mjs" < { if (!condition) throw new Error(message); }; const plugin = "/repo/.claude/mods/firstmate-calm"; check(policy.calmPreferencePath({}, plugin) === "/repo/config/calm", "plugin-root fallback"); @@ -473,8 +480,8 @@ test_classifier_parity_with_shell_owner() { cat >"$TMP_ROOT/classify.mjs" <"$TMP_ROOT/doorbells.mjs" < Date: Thu, 1 Oct 2026 03:30:57 -0700 Subject: [PATCH 09/33] fix(bin): exclude a remote mate's own parent channel from self-home status scans (#5263) * fix(bin): exclude a remote mate's own parent channel from self-home scans A remote secondmate home's outbound parent channel lives at state/parent-replies.status inside its own state dir, so the watcher's signal scan enumerated it as a task status file and the open-decisions fold classified it as a phantom task named parent-replies: every parent-channel append spun a spurious signal wake and a phantom open decision in the mate's own home. fm-parent-channel-lib.sh gains fm_parent_channel_outbound_status, which resolves the channel into the mate's own state dir for the remote route only, and fm-classify-lib.sh's status_scan_parent_channel_exclude wraps it for the fleet-wide scans. The watcher's scan_signals and heartbeat fail-safe backstop, the whole-file and incremental open-decisions folds, the presentation snapshot, and the unread-surface scan now skip exactly that resolved path. The exclusion is home-shape-aware: a parent-replies.status in a main home or a local mate is an ordinary task log and keeps waking and folding, and every other status file is untouched. * no-mistakes(review): exclude a remote mate's parent channel from the daemon heartbeat scan * no-mistakes(document): Document remote mate parent-channel scan exclusion * ci: retrigger portable serial 4 * no-mistakes(ci): CI check 'Behavior portable serial 7' failed in tests/fm-contributions.test.sh ('reservation poll failed'). CI stderr showed bin/fm-contributions.sh:345 arithmetic 'DEADLINE - 6\n90077104: syntax error in expression': the fixture's fake date returned a torn two-line clock value. Root cause: the fake forge wrapper in wrap_forge advances the shared controllable clock via a non-atomic read-modify-write ('$(cat $FORGE/clock) + 6' with truncate-in-place '> $FORGE/clock') while concurrent background gh calls run and the fake date reads the same file; an interleaved truncate+write publishes a half-written value (CI's torn '6\n90077104', tail of 1790077104) or an emptied-read value ('6'), which either breaks the poll's arithmetic (nonzero exit -> 'reservation poll failed') or defeats the 15-second reservation defer. This is a pre-existing test-fixture race, not caused by the PR's diff (base..target touches no contributions code; the same commit passed this shard in run 35711207830 earlier the same day). Fixed the flaky fixture at its root: clock_bump() now writes each new value to a per-process mktemp file in the same directory and publishes it with mv (atomic rename), so concurrent forge callers and the fake date always read one complete old-or-new clock; fault patterns and deltas are unchanged. Verified: minimal 3-way concurrency repro shows the old wrapper corrupting (12/32/38 outcomes incl. empty-read) while the rename-based wrapper never corrupts (20/20 clean); the full tests/fm-contributions.test.sh passes twice (all 38 assertions ok, incl. the reservation, budget-exhaustion, genuine-failure, shared-once, and latency tests); 10 isolated reservation runs pass; shellcheck rc=0; worktree contains only this one-file change * no-mistakes(document): drop stale file-set copy in daemon catch-all comment --- bin/fm-classify-lib.sh | 38 +- bin/fm-parent-channel-lib.sh | 22 + bin/fm-supervise-daemon.sh | 19 +- bin/fm-test-run.sh | 1 + bin/fm-watch.sh | 13 +- docs/secondmate-parent-channel.md | 3 + tests/fm-contributions.test.sh | 21 +- .../fm-parent-channel-scan-exclusion.test.sh | 414 ++++++++++++++++++ 8 files changed, 512 insertions(+), 19 deletions(-) create mode 100755 tests/fm-parent-channel-scan-exclusion.test.sh diff --git a/bin/fm-classify-lib.sh b/bin/fm-classify-lib.sh index 11cc7f24cfb..7482e5a9df6 100755 --- a/bin/fm-classify-lib.sh +++ b/bin/fm-classify-lib.sh @@ -1057,6 +1057,28 @@ EOF printf '%s' "$verb" } +# The status file inside that is this home's outbound parent channel +# rather than a self-home task status log, printed; empty when there is none. +# Only a remote mate home resolves one - its state/parent-replies.status is the +# parent channel (bin/fm-parent-channel-lib.sh owns that resolution, sourced +# lazily here because that library sources this one at its top level, so a +# top-level source would be circular). A main home, a local mate - whose +# channel lives in the parent home - or an unusable identity or binding keeps +# every file, so ordinary task logs fold and wake exactly as before. The home +# is the directory containing , the /state layout every caller of +# these fleet-wide scans shares; a state dir outside such a home excludes +# nothing. Callers compare the resolved path, never the file name, so a +# parent-replies.status in any other home shape stays an ordinary task log. +status_scan_parent_channel_exclude() { # + local state=$1 exclude + if ! command -v fm_parent_channel_outbound_status >/dev/null 2>&1; then + # shellcheck source=bin/fm-parent-channel-lib.sh + . "$_FM_CLASSIFY_LIB_DIR/fm-parent-channel-lib.sh" + fi + exclude=$(fm_parent_channel_outbound_status "$(dirname "$state")" "$state") || return 0 + printf '%s\n' "$exclude" +} + # Fleet-wide wrapper around status_open_decisions: scans every task's status # log under and prefixes each still-open decision with its owning task # id, so a per-wake or per-session surface can print the consolidated open set @@ -1065,9 +1087,11 @@ EOF # one "\t\t\t" line per open decision, in glob (task id) # order; prints nothing when none are open. scan_open_decisions() { # - local state=$1 f task open line + local state=$1 f task open line exclude + exclude=$(status_scan_parent_channel_exclude "$state") for f in "$state"/*.status; do [ -e "$f" ] || continue + [ "$f" = "$exclude" ] && continue task=$(basename "$f"); task="${task%.status}" open=$(status_open_decisions "$f") || continue [ -n "$open" ] || continue @@ -1358,9 +1382,11 @@ status_open_decisions_incremental() { # [] # the whole-file status_open_decisions, so a fleet-wide per-drain scan stays # bounded by new appends rather than total lifetime log size across every task. scan_open_decisions_incremental() { # - local state=$1 f task open line + local state=$1 f task open line exclude + exclude=$(status_scan_parent_channel_exclude "$state") for f in "$state"/*.status; do [ -e "$f" ] || continue + [ "$f" = "$exclude" ] && continue task=$(basename "$f"); task="${task%.status}" open=$(status_open_decisions_incremental "$f") || continue [ -n "$open" ] || continue @@ -1375,9 +1401,11 @@ EOF } status_presentation_snapshot() { # - local state=$1 f task size ident + local state=$1 f task size ident exclude + exclude=$(status_scan_parent_channel_exclude "$state") for f in "$state"/*.status; do [ -e "$f" ] || continue + [ "$f" = "$exclude" ] && continue [ -f "$f" ] && [ -r "$f" ] && [ ! -L "$f" ] || continue task=$(basename "$f"); task="${task%.status}" size=$(_fm_status_file_size "$f") || return 1 @@ -1989,9 +2017,11 @@ status_line_is_unread_surface() { # # Prints nothing when none are unread. Directory scan rejects status symlinks # the same way scan_open_decisions does. scan_unread_surface_lines() { # - local state=$1 f task lines line + local state=$1 f task lines line exclude + exclude=$(status_scan_parent_channel_exclude "$state") for f in "$state"/*.status; do [ -e "$f" ] || continue + [ "$f" = "$exclude" ] && continue task=$(basename "$f"); task="${task%.status}" lines=$(status_new_lines_since_cursor "$f") || return 1 [ -n "$lines" ] || continue diff --git a/bin/fm-parent-channel-lib.sh b/bin/fm-parent-channel-lib.sh index 24718258317..160d580ac02 100644 --- a/bin/fm-parent-channel-lib.sh +++ b/bin/fm-parent-channel-lib.sh @@ -123,6 +123,28 @@ fm_parent_channel_destination() { # esac } +# The outbound parent-channel status path that lives INSIDE , printed, +# when is a remote mate; non-zero for a main home, a local mate, or an +# unusable identity or binding. Only the remote route resolves the channel into +# the mate's own state dir, so parent-replies.status there is the mate's parent +# channel rather than a self-home task status file: a home's own status scans +# and decision folds exclude exactly this resolved path (the same special case +# fm-pending-reply-lib.sh's wrong-home detection applies). A local mate's +# channel lives in the parent home's state/.status, which the parent's +# scans must keep classifying, so only the remote route resolves here. +fm_parent_channel_outbound_status() { # + local home=$1 state=$2 destination rc=0 + destination=$(fm_parent_channel_destination "$home" "$state") || rc=$? + [ "$rc" -eq 0 ] || return 1 + # The substitution above ran the resolver in a subshell, so its route global + # died with it; resolve once more in this shell (stdout discarded, the same + # shape fm-pending-reply-lib.sh's wrong-home detection uses) so the route + # check reads the resolver's own verdict rather than re-deriving it. + fm_parent_channel_destination "$home" "$state" >/dev/null || return 1 + [ "$FM_PARENT_CHANNEL_ROUTE" = remote ] || return 1 + printf '%s\n' "$destination" +} + # Fold onto one bounded line, so a note copied from a child ledger or a # hold reason cannot break the channel's line framing. fm_parent_channel_clean_note() { # diff --git a/bin/fm-supervise-daemon.sh b/bin/fm-supervise-daemon.sh index 7a7191807df..a2a7664f4fb 100755 --- a/bin/fm-supervise-daemon.sh +++ b/bin/fm-supervise-daemon.sh @@ -65,9 +65,10 @@ # undelivered past FM_MAX_DEFER_SECS, the daemon retries a normal flush and # writes state/.subsuper-inject-wedged and attempts a configurable active # alert if submit still cannot be confirmed. -# - Cheap heartbeat catch-all: every HEARTBEAT_SCAN_SECS the daemon greps all -# state/*.status for a captain-relevant line the per-wake classifier might -# have missed (e.g. a status verb outside CAPTAIN_RE) and escalates it. +# - Cheap heartbeat catch-all: every HEARTBEAT_SCAN_SECS the daemon greps the +# state dir's task status logs for a captain-relevant line the per-wake +# classifier might have missed (e.g. a status verb outside CAPTAIN_RE) and +# escalates it. # # The robustness shell from the prior always-inject version is preserved: # single-instance lock (portable helper, no flock dependency), crash-loop @@ -1184,8 +1185,8 @@ _oldest_line_age() { # -> seconds since the oldest buffered item first ar # re-peek; gone -> clear; still declaring the wait, on an idle OR a busy pane # -> escalate a recheck digest naming which human the wait is on, and reset # the window (repeating bounded re-surface, never a wedge). -# 3) heartbeat scan: every HEARTBEAT_SCAN_SECS, grep state/*.status for a -# captain-relevant line the per-wake classifier missed and escalate it. +# 3) heartbeat scan: every HEARTBEAT_SCAN_SECS, run the catch-all status scan in +# the block below and escalate what it finds; that block owns its file set. housekeeping() { # local state=$1 now due f key task win marker age last max_defer oldest pause_secs marker_epoch until bounded_until pause_reason now=$(_now) @@ -1339,11 +1340,17 @@ housekeeping() { # # because the event this backstop most needs to catch is precisely one a # later routine append has already moved past; fm-classify-lib.sh's span # read decides relevance, and the classified-through offset is the dedup. + # A remote mate's own parent channel is not a self-home task status log, + # so it is excluded here exactly as in the watcher's twin backstop + # (fm-watch.sh heartbeat_scan_finds_actionable); the home-shape-aware + # resolution lives in status_scan_parent_channel_exclude. if [ "$(_file_age "$state/.subsuper-last-scan")" -ge "${FM_HEARTBEAT_SCAN_SECS:-$HEARTBEAT_SCAN_SECS_DEFAULT}" ]; then _now > "$state/.subsuper-last-scan" - local event record rest endpoint ident rc + local event record rest endpoint ident rc exclude + exclude=$(status_scan_parent_channel_exclude "$state") for f in "$state"/*.status; do [ -e "$f" ] || [ -L "$f" ] || continue + [ "$f" = "$exclude" ] && continue task=$(basename "$f"); task="${task%.status}" record=$(status_span_first_actionable_record "$f" \ "$(status_seen_offset "$state" "$task")") diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index dfff544fbdf..4702f403f7f 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -309,6 +309,7 @@ family_for_basename() { ;; fm-daemon.test.sh|fm-guard-stale-banner.test.sh|fm-pi-watch-extension.test.sh|\ fm-session-lock-ancestry.test.sh|fm-cursor-primary.test.sh|\ + fm-parent-channel-scan-exclusion.test.sh|\ fm-supervision-events.test.sh|fm-turnend-guard.test.sh|fm-wake-daemon-lifecycle-e2e.test.sh|\ fm-wake-drain-unread-status.test.sh|\ fm-tool-update-check.test.sh|\ diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index 35c5a9f1b75..6e51f76770a 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -1990,8 +1990,13 @@ age_of() { # seconds since file mtime; "due immediately" if missing # The caller records reported state only after surfacing or intentional absorption, # and commits a status classification position only after a successful span read. scan_signals() { - local f sig sf + local f sig sf exclude + # A remote mate's own parent channel is not a self-home task status log; the + # home-shape-aware exclusion and its precedent live in + # status_scan_parent_channel_exclude (fm-classify-lib.sh). + exclude=$(status_scan_parent_channel_exclude "$STATE") for f in "$STATE"/*.status "$STATE"/*.turn-ended; do + [ "$f" = "$exclude" ] && continue if [ ! -e "$f" ]; then case "$f" in *.status) [ -L "$f" ] || continue ;; *) continue ;; esac fi @@ -2255,10 +2260,14 @@ EOF # is absorbed; it surfaces only an event the per-wake path absorbed by mistake - # the fail-safe backstop. heartbeat_scan_finds_actionable() { - local f task record rest endpoint ident rc found=1 sig marker + local f task record rest endpoint ident rc found=1 sig marker exclude + # Same self-home exclusion as scan_signals: a remote mate's parent channel + # must not come back through the heartbeat fail-safe backstop. + exclude=$(status_scan_parent_channel_exclude "$STATE") FM_HEARTBEAT_SURFACE_ENDPOINTS='' for f in "$STATE"/*.status; do [ -e "$f" ] || [ -L "$f" ] || continue + [ "$f" = "$exclude" ] && continue task=$(basename "$f"); task="${task%.status}" record=$(status_span_first_actionable_record "$f" "$(hb_surfaced_offset "$task")") rc=$? diff --git a/docs/secondmate-parent-channel.md b/docs/secondmate-parent-channel.md index a15a163107c..614a26fca83 100644 --- a/docs/secondmate-parent-channel.md +++ b/docs/secondmate-parent-channel.md @@ -38,6 +38,8 @@ A duplicate line is harmless and a missed one is not, so the mate may still appe For marked replies, the report helper accepts no caller-selected destination and uses the channel resolver for both local and remote homes; its script header owns the exact invocation contract. The pending-reply guard may restate only the correlated line from a local mate's `state/.status` onto the parent channel, which repairs the common parent-home versus mate-home mixup without accepting arbitrary mate-home sightings as acknowledgement. Other correlated mate-home status lines remain wrong-home evidence, while a remote home's routed `state/parent-replies.status` is already the parent channel and is not classified as wrong-home. +The mate home's own status scans treat that remote channel the same way: `status_scan_parent_channel_exclude` in `bin/fm-classify-lib.sh` resolves the outbound path through the same `bin/fm-parent-channel-lib.sh` binding, and the watcher's signal scan and heartbeat backstop, the away-mode daemon's catch-all scan, and the fleet-wide folds skip exactly that resolved path, never a file name. +The remote reply adapter already mirrors every channel line into the parent home, so folding the channel again here would only spin spurious wakes and a phantom `parent-replies` task, while a `parent-replies.status` in a main home or in a local mate is an ordinary task log that keeps folding and waking. A missed-reply escalation includes the complete first sighting path and line number in readable shell-escaped form. ## What is deliberately not built @@ -55,6 +57,7 @@ A missed-reply escalation includes the complete first sighting path and line num `tests/fm-teardown.test.sh` covers teardown delivering a child's final line and refusing when the channel cannot be written. `tests/fm-brief.test.sh` pins the charter's channel rule. `tests/fm-pending-reply.test.sh` covers helper-selected local routing, remote-channel classification, same-basename restatement before false escalation, readable wrong-home diagnostics, and the rule that arbitrary mate-home sightings never acknowledge a reply. +`tests/fm-parent-channel-scan-exclusion.test.sh` covers the home-shape-aware scan exclusion against real remote, main-home, and local-mate fixtures: the watcher signal scan, both heartbeat backstops, the fleet-wide folds, and the real `fm-wake-drain.sh` end to end. ## Live verification diff --git a/tests/fm-contributions.test.sh b/tests/fm-contributions.test.sh index aa14fe28a0e..a1582c4e2e3 100755 --- a/tests/fm-contributions.test.sh +++ b/tests/fm-contributions.test.sh @@ -606,17 +606,24 @@ set -eu printf '%s\n' "$*" >> "$FORGE/calls" fault=$(cat "$FORGE/fault" 2>/dev/null || true) case "$fault" in latency) sleep "${FORGE_LATENCY:-2}" ;; esac +# Concurrent forge callers each advance one shared clock. Truncating it in +# place races with the other callers and the fake date: an interleaved write +# can publish a half-written value (or the 6 an emptied read computes), and a +# caller then evaluates DEADLINE against torn arithmetic. Publish every new +# value by rename so each reader always sees one complete old-or-new clock. +clock_bump() { + local tmp + tmp=$(mktemp "$FORGE/clock.XXXXXX") + printf '%s\n' "$(( $(cat "$FORGE/clock") + $1 ))" > "$tmp" + mv -f "$tmp" "$FORGE/clock" +} case "$fault:$*" in # Advance once before the parallel read wave; its readers share this clock. - reserve:'api repos/o/r/issues/9') - printf '%s\n' "$(( $(cat "$FORGE/clock") + 6 ))" > "$FORGE/clock" ;; + reserve:'api repos/o/r/issues/9') clock_bump 6 ;; slow-wave:'api repos/o/r/pulls/8') sleep 3 ;; slow-wave:'api repos/o/r/pulls/8/reviews?'*) sleep 6 ;; - exhaust:'api repos/o/r/issues/8/comments?'*) - printf '%s\n' "$(( $(cat "$FORGE/clock") + 100 ))" > "$FORGE/clock" ;; - fail-late:'api repos/o/r/pulls/8/reviews?'*) - printf '%s\n' "$(( $(cat "$FORGE/clock") + 100 ))" > "$FORGE/clock" - printf 'HTTP 502\n' >&2; exit 1 ;; + exhaust:'api repos/o/r/issues/8/comments?'*) clock_bump 100 ;; + fail-late:'api repos/o/r/pulls/8/reviews?'*) clock_bump 100; printf 'HTTP 502\n' >&2; exit 1 ;; fail:'api repos/o/r/pulls/8/reviews?'*) printf 'HTTP 502\n' >&2; exit 1 ;; down:*) printf 'HTTP 502\n' >&2; exit 1 ;; hang:'api repos/o/r/pulls/8') sleep 4 ;; diff --git a/tests/fm-parent-channel-scan-exclusion.test.sh b/tests/fm-parent-channel-scan-exclusion.test.sh new file mode 100755 index 00000000000..7d8a580a4c6 --- /dev/null +++ b/tests/fm-parent-channel-scan-exclusion.test.sh @@ -0,0 +1,414 @@ +#!/usr/bin/env bash +# tests/fm-parent-channel-scan-exclusion.test.sh - a remote mate home's own +# outbound parent channel (state/parent-replies.status, resolved through +# bin/fm-parent-channel-lib.sh) must not be enumerated by the home's own status +# scans: every parent-channel append is mirrored into the parent home by the +# remote reply adapter, so folding or waking on it here spins spurious signal +# wakes and phantom "parent-replies" open decisions. The exclusion must be +# home-shape-aware: a parent-replies.status in a main home, in a local mate, or +# in any other home shape is an ordinary task log and keeps waking and folding. +# +# Covers the watcher scan (scan_signals, the heartbeat fail-safe backstop), the +# away-mode daemon's twin catch-all scan (fm-supervise-daemon.sh housekeeping), +# and fm-classify-lib.sh's fleet-wide folds (whole-file, incremental, +# presentation snapshot, unread surface), each against a real remote mate +# fixture plus the main-home and local-mate negative cases, and the real +# fm-wake-drain.sh end to end. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +TMP_ROOT=$(fm_test_tmproot fm-parent-channel-scan-exclusion) +mkdir -p "$TMP_ROOT" +TMP_ROOT=$(cd "$TMP_ROOT" && pwd -P) + +# The real drain asserts watcher liveness through fm-guard.sh, whose tangle +# check warns when FM_ROOT sits on a feature branch; point it at a fresh +# non-git dir so the banner stays inert in this disposable worktree (the same +# trick tests/wake-helpers.sh installs for the drain suites). +FM_ROOT_OVERRIDE="$(fm_test_tmproot fm-parent-channel-scan-exclusion-root)" +export FM_ROOT_OVERRIDE +mkdir -p "$FM_ROOT_OVERRIDE" + +cleanup() { rm -rf -- "$TMP_ROOT"; } +trap cleanup EXIT + +# seed_remote_mate : build a remote mate home whose state dir carries one +# genuine task log and the outbound parent channel, with one captain-facing +# decision, one reserved-key resolution, and one informational note on the +# channel - exactly the line shapes a mate home publishes mechanically. +seed_remote_mate() { # + local dir=$1 + mkdir -p "$dir/state" + printf '%s\n' mate > "$dir/.fm-secondmate-home" + printf 'schema=fm-secondmate-parent.v1\nroute=remote\nparent_host=remote.example\n' \ + > "$dir/.fm-secondmate-parent" + printf 'needs-decision [key=captain-hold-pr-7-1]: captain hold pr-7: merge the green PR?\n' \ + > "$dir/state/parent-replies.status" + printf 'resolved [key=captain-hold-pr-5-2]: captain chose the staged rollout\n' \ + >> "$dir/state/parent-replies.status" + printf 'note: the release branch is cut\n' >> "$dir/state/parent-replies.status" + printf 'needs-decision [key=api-shape]: pick REST or RPC\n' > "$dir/state/real-task.status" + printf 'note: benchmark results are in\n' >> "$dir/state/real-task.status" +} + +# seed_plain_home : a main home (no secondmate identity marker) whose +# state dir carries a parent-replies.status that merely shares the name. +seed_plain_home() { # + local dir=$1 + mkdir -p "$dir/state" + printf 'needs-decision [key=name-only]: an ordinary task file that shares the name\n' \ + > "$dir/state/parent-replies.status" + printf 'needs-decision [key=other-task]: a genuine sibling task decision\n' \ + > "$dir/state/other-task.status" +} + +# seed_local_mate : a LOCAL mate home - its parent channel +# lives in the parent home's state/.status, so a parent-replies.status in +# its own state dir is an ordinary self-home file. +seed_local_mate() { # + local dir=$1 parent_home=$2 + mkdir -p "$dir/state" + printf '%s\n' mate > "$dir/.fm-secondmate-home" + printf 'schema=fm-secondmate-parent.v1\nroute=local\nparent_home=%s\n' "$parent_home" \ + > "$dir/.fm-secondmate-parent" + printf 'needs-decision [key=local-shape]: still an ordinary self-home file\n' \ + > "$dir/state/parent-replies.status" +} + +REMOTE="$TMP_ROOT/remote-mate" +PLAIN="$TMP_ROOT/main-home" +LOCAL_MATE="$TMP_ROOT/local-mate" +seed_remote_mate "$REMOTE" +seed_plain_home "$PLAIN" +seed_local_mate "$LOCAL_MATE" "$PLAIN" +REMOTE_STATE="$REMOTE/state" +PLAIN_STATE="$PLAIN/state" +LOCAL_STATE="$LOCAL_MATE/state" + +# --- unit: the exclusion predicates ----------------------------------------- + +test_predicate_resolves_only_the_remote_channel() { + local out rc + out=$(FM_TEST_LIB_SOURCED=1 bash -c ' + # shellcheck source=bin/fm-classify-lib.sh + . "$1/bin/fm-classify-lib.sh" + status_scan_parent_channel_exclude "$2" + ' _ "$ROOT" "$REMOTE_STATE") \ + || fail "the remote mate's channel must resolve for exclusion, got rc=$?" + [ "$out" = "$REMOTE_STATE/parent-replies.status" ] \ + || fail "the exclusion must be the resolved channel path, got: $out" + out=$(FM_TEST_LIB_SOURCED=1 bash -c ' + # shellcheck source=bin/fm-classify-lib.sh + . "$1/bin/fm-classify-lib.sh" + status_scan_parent_channel_exclude "$2" + ' _ "$ROOT" "$PLAIN_STATE") + [ -z "$out" ] || fail "a main home must exclude nothing, got: $out" + out=$(FM_TEST_LIB_SOURCED=1 bash -c ' + # shellcheck source=bin/fm-classify-lib.sh + . "$1/bin/fm-classify-lib.sh" + status_scan_parent_channel_exclude "$2" + ' _ "$ROOT" "$LOCAL_STATE") + [ -z "$out" ] || fail "a local mate must exclude nothing, got: $out" + pass "only a remote mate home resolves its own parent channel for exclusion" +} + +test_resolver_predicates_on_home_shape_not_name() { + local out + out=$(FM_TEST_LIB_SOURCED=1 bash -c ' + # shellcheck source=bin/fm-parent-channel-lib.sh + . "$1/bin/fm-parent-channel-lib.sh" + fm_parent_channel_outbound_status "$2" "$3" + ' _ "$ROOT" "$REMOTE" "$REMOTE_STATE") \ + || fail "the remote mate's outbound status must resolve" + [ "$out" = "$REMOTE_STATE/parent-replies.status" ] \ + || fail "the remote route must resolve into the mate's own state dir, got: $out" + FM_TEST_LIB_SOURCED=1 bash -c ' + # shellcheck source=bin/fm-parent-channel-lib.sh + . "$1/bin/fm-parent-channel-lib.sh" + fm_parent_channel_outbound_status "$2" "$3" + ' _ "$ROOT" "$PLAIN" "$PLAIN_STATE" \ + && fail "a main home has no outbound parent-channel status" + FM_TEST_LIB_SOURCED=1 bash -c ' + # shellcheck source=bin/fm-parent-channel-lib.sh + . "$1/bin/fm-parent-channel-lib.sh" + fm_parent_channel_outbound_status "$2" "$3" + ' _ "$ROOT" "$LOCAL_MATE" "$LOCAL_STATE" \ + && fail "a local mate's channel lives in the parent home, not its own state dir" + pass "fm_parent_channel_outbound_status resolves only the remote route" +} + +# --- unit: the fleet-wide folds omit the channel and keep genuine tasks ----- + +test_remote_folds_omit_channel_and_keep_genuine_task() { + local dir out + dir="$TMP_ROOT/folds" + seed_remote_mate "$dir/home" + out=$(FM_TEST_LIB_SOURCED=1 bash -c ' + # shellcheck source=bin/fm-classify-lib.sh + . "$1/bin/fm-classify-lib.sh" + echo "WHOLE:"; scan_open_decisions "$2" + echo "SNAPSHOT:"; status_presentation_snapshot "$2" + echo "UNREAD:"; scan_unread_surface_lines "$2" + ' _ "$ROOT" "$dir/home/state") || fail "the remote-mate fold pass failed" + case "$out" in *parent-replies*) + fail "the channel leaked into the remote mate's folds: $out" ;; + esac + printf '%s\n' "$out" | sed -n '/^WHOLE:/,/^SNAPSHOT:/p' | grep -F 'api-shape' >/dev/null \ + || fail "the genuine task's open decision must still fold: $out" + printf '%s\n' "$out" | sed -n '/^SNAPSHOT:/,/^UNREAD:/p' | grep -F 'real-task' >/dev/null \ + || fail "the genuine task must stay in the presentation snapshot: $out" + printf '%s\n' "$out" | sed -n '/^UNREAD:/,$p' | grep -F 'benchmark results' >/dev/null \ + || fail "the genuine task's note must stay on the unread surface: $out" + pass "a remote mate's folds omit its channel and keep a genuine task" +} + +test_incremental_fold_omits_channel_and_keeps_genuine_task() { + local dir out + dir="$TMP_ROOT/folds-incremental" + seed_remote_mate "$dir/home" + out=$(FM_TEST_LIB_SOURCED=1 bash -c ' + # shellcheck source=bin/fm-classify-lib.sh + . "$1/bin/fm-classify-lib.sh" + scan_open_decisions_incremental "$2" + ' _ "$ROOT" "$dir/home/state") || fail "the incremental fold failed" + case "$out" in *parent-replies*) + fail "the channel leaked into the incremental fold: $out" ;; + esac + printf '%s\n' "$out" | grep -F 'api-shape' >/dev/null \ + || fail "the genuine task's decision must still fold incrementally: $out" + pass "the cursor-backed incremental fold omits a remote mate's channel" +} + +test_channel_lines_never_reach_the_remote_unread_surface() { + local dir out + dir="$TMP_ROOT/unread" + seed_remote_mate "$dir/home" + out=$(FM_TEST_LIB_SOURCED=1 bash -c ' + # shellcheck source=bin/fm-classify-lib.sh + . "$1/bin/fm-classify-lib.sh" + scan_unread_surface_lines "$2" + ' _ "$ROOT" "$dir/home/state") || fail "the unread-surface scan failed" + case "$out" in *parent-replies*|*captain-hold*|*release\ branch*) + fail "channel decision, resolution, or note surfaced as self-home unread status: $out" ;; + esac + pass "the channel's resolution and note lines stay off the remote unread surface" +} + +test_name_shared_file_folds_in_a_main_home() { + local out + out=$(FM_TEST_LIB_SOURCED=1 bash -c ' + # shellcheck source=bin/fm-classify-lib.sh + . "$1/bin/fm-classify-lib.sh" + scan_open_decisions "$2" + ' _ "$ROOT" "$PLAIN_STATE") || fail "the main-home fold failed" + printf '%s\n' "$out" | grep -F 'name-only' >/dev/null \ + || fail "a main home's parent-replies.status must keep folding as an ordinary task: $out" + printf '%s\n' "$out" | grep -F 'other-task' >/dev/null \ + || fail "the sibling task decision must keep folding: $out" + pass "a parent-replies.status in a main home still folds" +} + +test_name_shared_file_folds_in_a_local_mate() { + local out + out=$(FM_TEST_LIB_SOURCED=1 bash -c ' + # shellcheck source=bin/fm-classify-lib.sh + . "$1/bin/fm-classify-lib.sh" + scan_open_decisions "$2" + ' _ "$ROOT" "$LOCAL_STATE") || fail "the local-mate fold failed" + printf '%s\n' "$out" | grep -F 'local-shape' >/dev/null \ + || fail "a local mate's parent-replies.status must keep folding: $out" + pass "a parent-replies.status in a local mate still folds" +} + +# --- unit: the watcher's signal scan and heartbeat backstop ----------------- + +# Source the watcher once with an isolated state/home; its source guard returns +# before the lock/loop, so only the functions load. scan_signals and +# heartbeat_scan_finds_actionable read STATE at call time. FM_ROOT_OVERRIDE +# stays at the inert dir set above; the unit-called functions read STATE, not +# the repo root. +WATCH_STATE="$REMOTE_STATE" +export FM_STATE_OVERRIDE="$WATCH_STATE" +export FM_HOME="$REMOTE" +# Production modules are independently linted canonical roots. Keep this test's +# ShellCheck context local while preserving its unchanged runtime source path. +# shellcheck source=/dev/null +. "$ROOT/bin/fm-watch.sh" + +test_watcher_scan_skips_channel_and_keeps_task_in_remote_mate() { + local out rc + STATE="$REMOTE_STATE" + out=$(scan_signals) || fail "scan_signals failed over the remote mate state" + printf '%s\n' "$out" | cut -f3 | grep -F 'parent-replies.status' >/dev/null \ + && fail "the channel must not produce a signal wake: $out" + printf '%s\n' "$out" | cut -f3 | grep -F 'real-task.status' >/dev/null \ + || fail "the genuine task's status must still wake: $out" + pass "scan_signals skips a remote mate's channel and still reports its tasks" +} + +test_heartbeat_backstop_skips_channel_in_remote_mate() { + local dir rc + dir="$TMP_ROOT/heartbeat" + seed_remote_mate "$dir/home" + # A quiet task log keeps the first pass channel-only: the note: line is + # informational, so only the excluded channel could make the scan actionable. + printf 'note: benchmark results are in\n' > "$dir/home/state/real-task.status" + STATE="$dir/home/state" + heartbeat_scan_finds_actionable; rc=$? + [ "$rc" -eq 1 ] || fail "the channel must not surface through the heartbeat backstop (rc=$rc): $FM_HEARTBEAT_SURFACE_ENDPOINTS" + case "$FM_HEARTBEAT_SURFACE_ENDPOINTS" in + *parent-replies*) fail "the channel leaked into the heartbeat backstop: $FM_HEARTBEAT_SURFACE_ENDPOINTS" ;; + esac + # A genuine task's captain-relevant line must keep reaching the backstop. + printf 'blocked [key=wedge]: the crew is stuck\n' >> "$dir/home/state/real-task.status" + heartbeat_scan_finds_actionable; rc=$? + [ "$rc" -eq 0 ] || fail "a genuine task's decision must surface through the heartbeat backstop" + case "$FM_HEARTBEAT_SURFACE_ENDPOINTS" in + *real-task.status*) ;; + *) fail "the heartbeat backstop must name the genuine task: $FM_HEARTBEAT_SURFACE_ENDPOINTS" ;; + esac + case "$FM_HEARTBEAT_SURFACE_ENDPOINTS" in + *parent-replies*) fail "the channel leaked into the heartbeat backstop: $FM_HEARTBEAT_SURFACE_ENDPOINTS" ;; + esac + pass "the heartbeat backstop skips a remote mate's channel and keeps its tasks" +} + +test_watcher_scan_keeps_name_shared_files_outside_remote_mates() { + local out + STATE="$PLAIN_STATE" + out=$(scan_signals) || fail "scan_signals failed over the main-home state" + printf '%s\n' "$out" | cut -f3 | grep -F 'parent-replies.status' >/dev/null \ + || fail "a main home's parent-replies.status must keep waking: $out" + # shellcheck disable=SC2034 # read by the sourced watcher's scans at call time + STATE="$LOCAL_STATE" + out=$(scan_signals) || fail "scan_signals failed over the local-mate state" + printf '%s\n' "$out" | cut -f3 | grep -F 'parent-replies.status' >/dev/null \ + || fail "a local mate's parent-replies.status must keep waking: $out" + pass "scan_signals keeps parent-replies.status outside remote mate homes" +} + +# --- unit: the away-mode daemon's heartbeat catch-all backstop -------------- + +# The daemon runs the watcher's twin catch-all scan while a home is away, so it +# needs the same exclusion. Source it in a subshell - its BASH_SOURCE guard +# skips the main loop, and the isolation keeps its function table from +# colliding with the watcher already sourced above. +daemon_heartbeat_scan() { # + local home=$1 + rm -f "$home/state/.subsuper-last-scan" + FM_TEST_LIB_SOURCED=1 FM_HOME="$home" FM_STATE_OVERRIDE="$home/state" \ + bash -c ' + # shellcheck source=/dev/null + . "$1/bin/fm-supervise-daemon.sh" + housekeeping "$2" + ' _ "$ROOT" "$home/state" >/dev/null 2>&1 +} + +test_daemon_heartbeat_backstop_skips_channel_in_remote_mate() { + local dir buffer + dir="$TMP_ROOT/daemon-heartbeat" + seed_remote_mate "$dir/home" + # A quiet task log keeps the first pass channel-only, so only the excluded + # channel could put anything in the escalation buffer. + printf 'note: benchmark results are in\n' > "$dir/home/state/real-task.status" + daemon_heartbeat_scan "$dir/home" + buffer=$(cat "$dir/home/state/.subsuper-escalations" 2>/dev/null || true) + case "$buffer" in *parent-replies*|*captain-hold*|*release\ branch*) + fail "the channel leaked into the daemon's catch-all scan: $buffer" ;; + esac + [ -z "$(cat "$dir/home/state/.subsuper-seen-status-parent-replies" 2>/dev/null || true)" ] \ + || fail "the daemon tracked the channel as a phantom parent-replies task" + + # A genuine task's captain-relevant line must keep reaching the backstop. + printf 'blocked [key=wedge]: the crew is stuck\n' >> "$dir/home/state/real-task.status" + daemon_heartbeat_scan "$dir/home" + buffer=$(cat "$dir/home/state/.subsuper-escalations" 2>/dev/null || true) + printf '%s\n' "$buffer" | grep -F 'real-task.status' >/dev/null \ + || fail "a genuine task's decision must still surface through the daemon backstop: $buffer" + case "$buffer" in *parent-replies*) + fail "the channel leaked into the daemon's catch-all scan: $buffer" ;; + esac + pass "the daemon's catch-all scan skips a remote mate's channel and keeps its tasks" +} + +test_daemon_heartbeat_backstop_keeps_name_shared_file_in_a_main_home() { + local dir buffer + dir="$TMP_ROOT/daemon-heartbeat-main" + seed_plain_home "$dir/home" + printf 'blocked [key=name-only]: an ordinary task file that shares the name\n' \ + > "$dir/home/state/parent-replies.status" + daemon_heartbeat_scan "$dir/home" + buffer=$(cat "$dir/home/state/.subsuper-escalations" 2>/dev/null || true) + printf '%s\n' "$buffer" | grep -F 'parent-replies.status' >/dev/null \ + || fail "a main home's parent-replies.status must keep reaching the daemon backstop: $buffer" + pass "the daemon's catch-all scan keeps parent-replies.status outside remote mate homes" +} + +# --- end to end: the real drain over a remote mate home --------------------- + +test_drain_presents_no_channel_content_in_remote_mate() { + local dir out manifest + dir="$TMP_ROOT/drain" + seed_remote_mate "$dir/home" + mkdir -p "$dir/home/data" + FM_STATE_OVERRIDE="$dir/home/state" FM_HOME="$dir/home" \ + "$ROOT/bin/fm-wake-drain.sh" > "$dir/drain.out" \ + || fail "the drain failed over a remote mate home" + out=$(cat "$dir/drain.out") + case "$out" in *parent-replies*) + fail "the drain presented the remote mate's channel: $out" ;; + esac + printf '%s\n' "$out" | grep -F 'api-shape' >/dev/null \ + || fail "the genuine task's open decision must still surface in OPEN DECISIONS: $out" + printf '%s\n' "$out" | grep -F 'benchmark results' >/dev/null \ + || fail "the genuine task's note must still surface under UNREAD STATUS: $out" + # The presentation manifest is rebuilt from the excluded snapshot, so a + # channel row an older watcher recorded must not survive the drain. + manifest=$(cat "$dir/home/state/.status-presentation-cursor" 2>/dev/null || true) + case "$manifest" in *parent-replies*) + fail "the presentation manifest still tracks the channel: $manifest" ;; + esac + pass "the real drain presents no channel content from a remote mate home" +} + +test_drain_ignores_stale_channel_records_from_an_older_watcher() { + local dir out manifest + dir="$TMP_ROOT/drain-stale" + seed_remote_mate "$dir/home" + mkdir -p "$dir/home/data" + # An older watcher folded the channel and tracked it as a task; the fixed + # drain must drop both rather than present or choke on them. + printf 'needs-decision [key=old-phantom]: folded by the unfixed watcher\n' \ + > "$dir/home/state/.parent-replies.open-decisions-cursor" + printf 'parent-replies\tstrong:1:2:3\t99\t0\n' \ + > "$dir/home/state/.status-presentation-cursor" + FM_STATE_OVERRIDE="$dir/home/state" FM_HOME="$dir/home" \ + "$ROOT/bin/fm-wake-drain.sh" > "$dir/drain.out" \ + || fail "the drain failed over stale channel records" + out=$(cat "$dir/drain.out") + case "$out" in *parent-replies*|*old-phantom*) + fail "a stale channel fold resurfaced through the drain: $out" ;; + esac + manifest=$(cat "$dir/home/state/.status-presentation-cursor" 2>/dev/null || true) + case "$manifest" in *parent-replies*) + fail "the stale manifest row survived the drain: $manifest" ;; + esac + pass "stale channel records from an older watcher are dropped, not presented" +} + +test_predicate_resolves_only_the_remote_channel +test_resolver_predicates_on_home_shape_not_name +test_remote_folds_omit_channel_and_keep_genuine_task +test_incremental_fold_omits_channel_and_keeps_genuine_task +test_channel_lines_never_reach_the_remote_unread_surface +test_name_shared_file_folds_in_a_main_home +test_name_shared_file_folds_in_a_local_mate +test_watcher_scan_skips_channel_and_keeps_task_in_remote_mate +test_heartbeat_backstop_skips_channel_in_remote_mate +test_watcher_scan_keeps_name_shared_files_outside_remote_mates +test_daemon_heartbeat_backstop_skips_channel_in_remote_mate +test_daemon_heartbeat_backstop_keeps_name_shared_file_in_a_main_home +test_drain_presents_no_channel_content_in_remote_mate +test_drain_ignores_stale_channel_records_from_an_older_watcher From b5d906129ab50eedb3f006791372872dcc01146e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micka=C3=ABl=20R=C3=A9mond?= Date: Thu, 1 Oct 2026 16:24:22 +0200 Subject: [PATCH 10/33] fix(bin): document accepted contribution verdict actors (#6307) * fix(bin): name the accepted verdict actors in fm-contributions help and refusal * fix(ci): Updated tests/fm-contributions.test.sh to assert exactly captain, fleet, maintainer, and nobody in command-emitted help and refusal output. Three focused regressions passed; all three extra-actor mutations were rejected. ShellCheck, syntax, and diff checks passed. Production code remains unchanged --- bin/fm-contributions.sh | 11 ++++++----- tests/fm-contributions.test.sh | 28 +++++++++++++++++++++++++++- 2 files changed, 33 insertions(+), 6 deletions(-) diff --git a/bin/fm-contributions.sh b/bin/fm-contributions.sh index 4e27e33e8a1..12bfc5fffcf 100755 --- a/bin/fm-contributions.sh +++ b/bin/fm-contributions.sh @@ -5,7 +5,7 @@ # fm-contributions.sh snapshot [--all] # fm-contributions.sh poll # fm-contributions.sh pending -# fm-contributions.sh verdict +# fm-contributions.sh verdict # fm-contributions.sh ack # fm-contributions.sh arm [--if-owned] # @@ -23,9 +23,10 @@ # checks/reviews). Checks are normalized by name, id, started_at, status and # conclusion; projection picks the newest attempt per distinct name. The last # observation's lane names also disclose a lane absent from the next head. -# A verdict records the EXACT judged head, source URL, actor and summary. A -# comment's arrival time never supplies its judged head. Record a prose verdict -# only after its source identifies that head; otherwise leave it unbound and +# A verdict records the EXACT judged head, source URL, actor and summary. The +# actor is exactly one of captain, fleet, maintainer or nobody; any other value +# is refused. A comment's arrival time never supplies its judged head. Record a +# prose verdict only after its source identifies that head; otherwise leave it unbound and # triage its signal. Formal reviews carry GitHub's own commit_id. Neither kind # can grant merge authority. Captain-actor prose requires an existing live hold; # an eligible merge remains a captain call, never an automatic forge action. @@ -473,7 +474,7 @@ case "${1:-}" in else [ "$#" -eq 4 ] || fail 'verdict needs judged-head, source-url, actor and summary' fm_pr_head_valid "$1" || fail 'an exact judged commit is required' - case "$3" in captain|fleet|maintainer|nobody) ;; *) fail 'invalid required actor' ;; esac + case "$3" in captain|fleet|maintainer|nobody) ;; *) fail "invalid required actor '$3'; expected one of: captain, fleet, maintainer, nobody" ;; esac case "$2" in "$url"\#*) ;; *) fail 'verdict source must be a comment or review on this contribution' ;; esac jq --arg head "$1" --arg source "$2" --arg actor "$3" --arg summary "$4" \ '.verdict={head:$head,source:$source,actor:$actor,summary:$summary}' "$TMP/row.json" > "$TMP/update.json" diff --git a/tests/fm-contributions.test.sh b/tests/fm-contributions.test.sh index a1582c4e2e3..ce87baefafb 100755 --- a/tests/fm-contributions.test.sh +++ b/tests/fm-contributions.test.sh @@ -299,6 +299,32 @@ test_verdict_retains_judged_head() { pass 'recorded judgment keeps its exact head and is stale immediately on a published replacement' } +test_verdict_actor_values_are_discoverable() { + local home help out actor + home=$(new_home verdict-actors) + forge_home "$home" + with_home "$home" "$ROOT/bin/fm-pr-check.sh" delivery https://github.com/o/r/pull/8 >/dev/null \ + || fail 'could not register delivery before judging its head' + help=$("$ROOT/bin/fm-contributions.sh" --help) || fail 'verdict help did not print' + out=$(with_home "$home" "$ROOT/bin/fm-contributions.sh" verdict delivery https://github.com/o/r/pull/8 "$HEAD_A" \ + https://github.com/o/r/pull/8#issuecomment-99 bogus 'no such actor' 2>&1) \ + && fail 'an unknown actor was accepted' + [ "$(printf '%s\n' "$help" | sed -n '/^ fm-contributions.sh verdict /p')" = \ + ' fm-contributions.sh verdict ' ] \ + || fail "help usage does not name exactly the accepted actors: $help" + [ "$(printf '%s\n' "$help" | sed -n '/^actor is exactly one of /p')" = \ + 'actor is exactly one of captain, fleet, maintainer or nobody; any other value' ] \ + || fail "help explanation does not name exactly the accepted actors: $help" + [ "$out" = "fm-contributions: invalid required actor 'bogus'; expected one of: captain, fleet, maintainer, nobody" ] \ + || fail "refusal does not name exactly the accepted actors: $out" + for actor in captain fleet maintainer nobody; do + with_home "$home" "$ROOT/bin/fm-contributions.sh" verdict delivery https://github.com/o/r/pull/8 "$HEAD_A" \ + https://github.com/o/r/pull/8#issuecomment-99 "$actor" 'documented actor' >/dev/null \ + || fail "documented actor $actor was refused" + done + pass 'verdict help and refusal name exactly the actors the command accepts' +} + test_observed_replacement_refreshes_verdict() { local home home=$(new_home observed-replacement) @@ -1073,7 +1099,7 @@ test_late_owner_keeps_failure_episode_suppressed() { } failures=0 -for test_name in test_actor_coverage test_stale_verdict test_unchecked_is_not_silence test_newest_check_has_no_verdict test_comment_wake test_review_wake test_inline_wake test_ready_issue_wake test_fresh_issue_requires_maintainer test_missing_lane_remains_missing test_partial_freshness_keeps_measured_rows test_malformed_record_cannot_prove_silence test_issue_timeline_and_exact_ack test_verdict_retains_judged_head test_observed_replacement_refreshes_verdict test_unobserved_head_leaves_verdict_unknown test_away_yolo_is_fleet_work test_away_yolo_cross_home_is_fleet_work test_retired_and_unsupported_coverage test_unsupported_forge_is_not_fleet_work test_held_unsupported_forge_is_not_captain_work test_shared_contribution_signal_wakes_once test_watcher_keeps_diagnostics_separate_from_contribution_wakes test_expired_child_unsupported_forge_stays_unmeasured test_watcher_surfaces_new_contribution_once test_home_summary_coverage test_unreadable_pending_is_not_empty test_record_task_identity_matches_dirname_basename test_read_only_views_create_no_state test_budget_refusal_between_calls test_budget_bounded_call_timeout test_genuine_failure_near_deadline_is_unavailable test_shared_url_observed_once test_terminal_contribution_settles test_late_owner_inherits_terminal_observation test_interrupted_multi_owner_poll_settles_every_owner test_done_task_open_pr_still_observed test_reservation_defers_later_url_when_fifteen_seconds_do_not_remain test_three_second_pr_reads_complete_fresh_in_one_cycle test_slow_read_deadline_kill_is_budget_refusal test_unmeasured_url_does_not_starve_the_tail test_budget_is_cut_down_to_the_watcher_check_bound test_arm_plumbs_a_configured_budget_into_the_check_shim test_unavailable_forge_records_error_and_wakes_once_per_episode test_late_owner_keeps_failure_episode_suppressed; do +for test_name in test_actor_coverage test_stale_verdict test_unchecked_is_not_silence test_newest_check_has_no_verdict test_comment_wake test_review_wake test_inline_wake test_ready_issue_wake test_fresh_issue_requires_maintainer test_missing_lane_remains_missing test_partial_freshness_keeps_measured_rows test_malformed_record_cannot_prove_silence test_issue_timeline_and_exact_ack test_verdict_retains_judged_head test_verdict_actor_values_are_discoverable test_observed_replacement_refreshes_verdict test_unobserved_head_leaves_verdict_unknown test_away_yolo_is_fleet_work test_away_yolo_cross_home_is_fleet_work test_retired_and_unsupported_coverage test_unsupported_forge_is_not_fleet_work test_held_unsupported_forge_is_not_captain_work test_shared_contribution_signal_wakes_once test_watcher_keeps_diagnostics_separate_from_contribution_wakes test_expired_child_unsupported_forge_stays_unmeasured test_watcher_surfaces_new_contribution_once test_home_summary_coverage test_unreadable_pending_is_not_empty test_record_task_identity_matches_dirname_basename test_read_only_views_create_no_state test_budget_refusal_between_calls test_budget_bounded_call_timeout test_genuine_failure_near_deadline_is_unavailable test_shared_url_observed_once test_terminal_contribution_settles test_late_owner_inherits_terminal_observation test_interrupted_multi_owner_poll_settles_every_owner test_done_task_open_pr_still_observed test_reservation_defers_later_url_when_fifteen_seconds_do_not_remain test_three_second_pr_reads_complete_fresh_in_one_cycle test_slow_read_deadline_kill_is_budget_refusal test_unmeasured_url_does_not_starve_the_tail test_budget_is_cut_down_to_the_watcher_check_bound test_arm_plumbs_a_configured_budget_into_the_check_shim test_unavailable_forge_records_error_and_wakes_once_per_episode test_late_owner_keeps_failure_episode_suppressed; do ( "$test_name" ) || failures=$((failures + 1)) done [ "$failures" -eq 0 ] || fail "$failures contribution regressions" From 8f756bbc287c5bdfacc64a7cc09e8516c64fc919 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micka=C3=ABl=20R=C3=A9mond?= Date: Thu, 1 Oct 2026 16:24:35 +0200 Subject: [PATCH 11/33] fix(bin): recognize clone roots across path spelling differences (#6306) * fix(bin): recognise a clone root git names with different path spelling fm-fleet-sync compared git's --show-toplevel with pwd -P as strings, so a clone root that git recorded with different casing (case-insensitive volume) was skipped as not a clone root and never refreshed. Compare filesystem identity instead, which also covers symlink spelling. * fix(document): Remove stale clone-root comparison comment --- bin/fm-fleet-sync.sh | 8 +++++--- tests/fm-fleet-sync.test.sh | 37 +++++++++++++++++++++++++++++++++++-- 2 files changed, 40 insertions(+), 5 deletions(-) diff --git a/bin/fm-fleet-sync.sh b/bin/fm-fleet-sync.sh index f8cc3054591..91b76f78555 100755 --- a/bin/fm-fleet-sync.sh +++ b/bin/fm-fleet-sync.sh @@ -319,10 +319,12 @@ sync_project() { echo "$label: skipped: not a git repo" return 0 fi - # Both sides are physical paths (git resolves --show-toplevel through symlinks), - # so a symlinked clone dir still compares equal to its own root. + # Compare filesystem identity, not spelling: the question is whether git's root + # and $PROJ are the same directory, and a string compare of the two paths also + # fails when they merely differ in case (case-insensitive volume) or in how a + # symlink is spelled. proj_abs=$(cd "$PROJ" && pwd -P) || proj_abs="" - if [ "$proj_top" != "$proj_abs" ]; then + if [ -z "$proj_abs" ] || ! [ "$proj_top" -ef "$proj_abs" ]; then echo "$label: skipped: not a clone root (git would act on $proj_top)" return 0 fi diff --git a/tests/fm-fleet-sync.test.sh b/tests/fm-fleet-sync.test.sh index 68c32f50d30..0bee1668c3d 100755 --- a/tests/fm-fleet-sync.test.sh +++ b/tests/fm-fleet-sync.test.sh @@ -683,8 +683,8 @@ test_symlinked_clone_still_syncs() { home=$(new_home) clone=$(build_pair "$home" sigma) advance_origin "$home" sigma C1 - # A symlinked clone dir is a real clone root; the guard compares resolved paths, - # so it must not be mistaken for a directory nested in someone else's repo. + # A symlinked clone dir is a real clone root and must not be mistaken for a + # directory nested in someone else's repo. mv "$clone" "$home/real-sigma" ln -s "$home/real-sigma" "$clone" @@ -694,6 +694,38 @@ test_symlinked_clone_still_syncs() { pass "the clone-root guard accepts a symlinked clone directory" } +test_clone_root_named_by_another_spelling_still_syncs() { + local home clone fakebin alias out + home=$(new_home) + clone=$(build_pair "$home" tau) + advance_origin "$home" tau C1 + fakebin="$home/fb-rootalias"; rm -rf "$fakebin"; mkdir -p "$fakebin" + # git reports the clone's own root through an alias that is the same directory + # but a different string, as it does on a case-insensitive volume when the home + # was recorded with other casing. A symlink stands in for the case difference so + # the test also holds on a case-sensitive filesystem. + alias="$home/root-alias" + ln -s "$clone" "$alias" + cat > "$fakebin/git" <<'SH' +#!/usr/bin/env bash +real=${REAL_GIT_FOR_TEST:?} +case " $* " in + *" rev-parse --show-toplevel "*) printf '%s\n' "${ROOT_ALIAS_FOR_TEST:?}"; exit 0 ;; +esac +exec "$real" "$@" +SH + chmod +x "$fakebin/git" + out="$home/out"; err="$home/err" + + ROOT_ALIAS_FOR_TEST="$alias" run_sync_guarded "$home" "$fakebin" "$out" "$err" tau || true + + assert_contains "$(cat "$out")" "tau: synced" \ + "a clone root that git names with another spelling must still fast-forward" + assert_not_contains "$(cat "$out")" "not a clone root" \ + "the guard must compare the directory itself, not the spelling of its path" + pass "the clone-root guard accepts a root named by a different spelling of the same directory" +} + test_non_signature_fetch_failure_is_not_retried() { local home fakebin clone out err home=$(new_home) @@ -741,3 +773,4 @@ test_non_signature_fetch_failure_is_not_retried test_non_clone_dir_never_syncs_the_enclosing_repo test_non_clone_dir_named_directly_never_syncs_the_enclosing_repo test_symlinked_clone_still_syncs +test_clone_root_named_by_another_spelling_still_syncs From 349e189f310fe477ed88a592592187c1585def55 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Thu, 1 Oct 2026 13:58:15 -0700 Subject: [PATCH 12/33] test: preserve Pi calm transcript captures with Pi 1.0 (#6338) * test(calm): pin Pi's regular TUI mode where pane assertions read scrollback Pi 1.0.0 defaults its TUI to a fullscreen alternate-screen mode whose scrollable transcript is application-owned, so rows that leave the viewport never enter terminal scrollback and tmux capture-pane -S can no longer see them. The Pi Calm e2e launches now pass --tui-mode regular wherever the flag exists so the transcript assertions keep reading real scrollback on both the Pi 1.0.0 line and earlier Pi lines, which have no such flag and render regular-only anyway. * no-mistakes(document): Correct Pi TUI documentation and scrollback rationale --- .../harness-adapters/references/harness/pi.md | 1 - tests/fm-calm-pi-extension.test.sh | 23 ++++++++++++++----- 2 files changed, 17 insertions(+), 7 deletions(-) diff --git a/.agents/skills/harness-adapters/references/harness/pi.md b/.agents/skills/harness-adapters/references/harness/pi.md index c63eb1d5926..b6afb51fad3 100644 --- a/.agents/skills/harness-adapters/references/harness/pi.md +++ b/.agents/skills/harness-adapters/references/harness/pi.md @@ -18,7 +18,6 @@ Verified on 2026-07-27 with Pi and Pi-signed 0.82.0 unless a fact gives another Native Codex sessions may request `ultra` through the native extension flag described by `../../../bin/fm-spawn.sh`; it is separate from Pi's thinking levels. Pi has no permission system, so workers are always autonomous. -Pi's installed `packages/coding-agent/docs/settings.md` UI and display section documents `regular` as the `tuiMode` default and `fullscreen` as experimental. Fullscreen can bury steering messages by rewriting scrollback, so Firstmate avoids it when the installed CLI supports the override. `../../../bin/fm-spawn.sh --help` owns the executable-pinning and version-safe launch mechanics. diff --git a/tests/fm-calm-pi-extension.test.sh b/tests/fm-calm-pi-extension.test.sh index e73c0948c98..287c5de2b0e 100755 --- a/tests/fm-calm-pi-extension.test.sh +++ b/tests/fm-calm-pi-extension.test.sh @@ -53,6 +53,17 @@ wait_for_text() { return 1 } +# Pi 1.0.0 defaults its TUI to a fullscreen alternate-screen mode whose scrollable +# transcript is application-owned: rows that leave the viewport stay reachable +# through Pi's own scroll keys but never enter terminal scrollback, so +# tmux capture-pane -S can no longer see them. Transcript assertions below need +# real terminal scrollback, so each launch pins the regular TUI mode wherever the +# flag exists; versions without the flag retain their existing launch arguments. +PI_TUI_MODE_ARGS= +if pi --help 2>&1 | grep -q -- '--tui-mode'; then + PI_TUI_MODE_ARGS='--tui-mode regular' +fi + find_chrome() { local candidate if [ -n "${FM_CHROME_BIN:-}" ] && [ -x "$FM_CHROME_BIN" ]; then @@ -2289,7 +2300,7 @@ TS fi tmux -L "$TMUX_SOCKET" new-session -d -s "$TMUX_SESSION" -x 160 -y 36 \ - "cd '$project' && env FM_HOME='$home' PI_CODING_AGENT_DIR='$config' FM_OPERATIONAL_INPUT_SCRIPT='$OPERATIONAL_INPUT' PI_OFFLINE=1 pi --approve --no-context-files --no-skills --no-prompt-templates --no-extensions $extensions $session_arg; rc=\$?; printf '\nPI_EXIT=%s\n' \"\$rc\"; sleep 20" + "cd '$project' && env FM_HOME='$home' PI_CODING_AGENT_DIR='$config' FM_OPERATIONAL_INPUT_SCRIPT='$OPERATIONAL_INPUT' PI_OFFLINE=1 pi $PI_TUI_MODE_ARGS --approve --no-context-files --no-skills --no-prompt-templates --no-extensions $extensions $session_arg; rc=\$?; printf '\nPI_EXIT=%s\n' \"\$rc\"; sleep 20" i=0 while [ "$i" -lt 120 ]; do pane=$(tmux -L "$TMUX_SOCKET" capture-pane -p -t "$TMUX_SESSION" -S - 2>/dev/null || true) @@ -2428,7 +2439,7 @@ JS tmux -L "$TMUX_SOCKET" kill-session -t "$TMUX_SESSION" 2>/dev/null || true printf '%s\n' on >"$home/config/calm" tmux -L "$TMUX_SOCKET" new-session -d -s "$TMUX_SESSION" -x 160 -y 36 \ - "cd '$project' && env FM_HOME='$home' PI_CODING_AGENT_DIR='$config' FM_OPERATIONAL_INPUT_SCRIPT='$OPERATIONAL_INPUT' PI_OFFLINE=1 pi --approve --no-context-files --no-skills --no-prompt-templates --no-extensions -e ./.pi/extensions/fm-calm.ts -e ./followup-e2e.ts --session '$exact_session'; rc=\$?; printf '\nPI_EXIT=%s\n' \"\$rc\"; sleep 20" + "cd '$project' && env FM_HOME='$home' PI_CODING_AGENT_DIR='$config' FM_OPERATIONAL_INPUT_SCRIPT='$OPERATIONAL_INPUT' PI_OFFLINE=1 pi $PI_TUI_MODE_ARGS --approve --no-context-files --no-skills --no-prompt-templates --no-extensions -e ./.pi/extensions/fm-calm.ts -e ./followup-e2e.ts --session '$exact_session'; rc=\$?; printf '\nPI_EXIT=%s\n' \"\$rc\"; sleep 20" i=0 while [ "$i" -lt 120 ]; do pane=$(tmux -L "$TMUX_SOCKET" capture-pane -p -t "$TMUX_SESSION" -S - 2>/dev/null || true) @@ -2599,7 +2610,7 @@ TS printf '%s\n' "$calm_state" >"$home/config/calm" mkdir -p "$sessions/$label" tmux -L "$TMUX_SOCKET" new-session -d -s "$TMUX_SESSION" -x 160 -y 36 \ - "cd '$project' && env FM_HOME='$home' PI_CODING_AGENT_DIR='$config' FM_OPERATIONAL_INPUT_SCRIPT='$OPERATIONAL_INPUT' QUEUED_ESCAPE_HELD='$held' QUEUED_ESCAPE_STATUS_LOG='$sessions/$label/status.log' PI_OFFLINE=1 pi --approve --no-context-files --no-skills --no-prompt-templates --no-extensions -e ./.pi/extensions/fm-calm.ts -e ./queued-escape-e2e.ts --session-dir '$sessions/$label'; rc=\$?; printf '\nPI_EXIT=%s\n' \"\$rc\"; sleep 20" + "cd '$project' && env FM_HOME='$home' PI_CODING_AGENT_DIR='$config' FM_OPERATIONAL_INPUT_SCRIPT='$OPERATIONAL_INPUT' QUEUED_ESCAPE_HELD='$held' QUEUED_ESCAPE_STATUS_LOG='$sessions/$label/status.log' PI_OFFLINE=1 pi $PI_TUI_MODE_ARGS --approve --no-context-files --no-skills --no-prompt-templates --no-extensions -e ./.pi/extensions/fm-calm.ts -e ./queued-escape-e2e.ts --session-dir '$sessions/$label'; rc=\$?; printf '\nPI_EXIT=%s\n' \"\$rc\"; sleep 20" wait_for_text "$TMP_ROOT/queued-escape-pane" 'queued-escape-e2e.ts' \ || fail "Pi queued-row $label case did not reach the ready composer" tmux -L "$TMUX_SOCKET" send-keys -t "$TMUX_SESSION" -l "/queued-escape-e2e $label" @@ -2798,7 +2809,7 @@ TS local session_arg=$1 tmux -L "$TMUX_SOCKET" kill-session -t "$TMUX_SESSION" 2>/dev/null || true tmux -L "$TMUX_SOCKET" new-session -d -s "$TMUX_SESSION" -x 100 -y 44 \ - "cd '$project' && env FM_HOME='$home' PI_CODING_AGENT_DIR='$config' PI_OFFLINE=1 pi --approve --no-context-files --no-prompt-templates --no-extensions -e ./.pi/extensions/fm-calm.ts -e ./geometry-provider.ts $session_arg; rc=\$?; printf '\nPI_EXIT=%s\n' \"\$rc\"; sleep 20" + "cd '$project' && env FM_HOME='$home' PI_CODING_AGENT_DIR='$config' PI_OFFLINE=1 pi $PI_TUI_MODE_ARGS --approve --no-context-files --no-prompt-templates --no-extensions -e ./.pi/extensions/fm-calm.ts -e ./geometry-provider.ts $session_arg; rc=\$?; printf '\nPI_EXIT=%s\n' \"\$rc\"; sleep 20" } capture_geometry_viewport() { @@ -4190,7 +4201,7 @@ TS JSON tmux -L "$TMUX_SOCKET" new-session -d -s "$TMUX_SESSION" -x 180 -y 44 \ - "cd '$project' && env FM_HOME='$home' PI_CODING_AGENT_DIR='$config' FM_OPERATIONAL_INPUT_SCRIPT='$OPERATIONAL_INPUT' PI_OFFLINE=1 pi --approve --no-skills --no-prompt-templates --no-context-files --session '$session_file'; rc=\$?; printf '\nPI_EXIT=%s\n' \"\$rc\"; sleep 30" + "cd '$project' && env FM_HOME='$home' PI_CODING_AGENT_DIR='$config' FM_OPERATIONAL_INPUT_SCRIPT='$OPERATIONAL_INPUT' PI_OFFLINE=1 pi $PI_TUI_MODE_ARGS --approve --no-skills --no-prompt-templates --no-context-files --session '$session_file'; rc=\$?; printf '\nPI_EXIT=%s\n' \"\$rc\"; sleep 30" wait_for_text "$default_snapshot" "The deterministic tool example is complete." \ || fail "Pi calm E2E did not reach the restored session transcript" assert_contains "$(cat "$default_snapshot")" "CALM_E2E_OUTPUT" "calm mode was not off by default" @@ -4863,7 +4874,7 @@ JS tmux -L "$TMUX_SOCKET" kill-session -t "$TMUX_SESSION" 2>/dev/null || true tmux -L "$TMUX_SOCKET" new-session -d -s "$TMUX_SESSION" -x 180 -y 44 \ - "cd '$project' && env FM_HOME='$home' PI_CODING_AGENT_DIR='$config' FM_OPERATIONAL_INPUT_SCRIPT='$OPERATIONAL_INPUT' PI_OFFLINE=1 pi --approve --no-skills --no-prompt-templates --no-context-files --session '$session_file'; rc=\$?; printf '\nPI_EXIT=%s\n' \"\$rc\"; sleep 30" + "cd '$project' && env FM_HOME='$home' PI_CODING_AGENT_DIR='$config' FM_OPERATIONAL_INPUT_SCRIPT='$OPERATIONAL_INPUT' PI_OFFLINE=1 pi $PI_TUI_MODE_ARGS --approve --no-skills --no-prompt-templates --no-context-files --session '$session_file'; rc=\$?; printf '\nPI_EXIT=%s\n' \"\$rc\"; sleep 30" wait_for_text "$restarted_snapshot" "CALM_WORKING_E2E_RESPONSE" \ || fail "Pi did not restore the persisted session after restart" assert_not_contains "$(cat "$restarted_snapshot")" "CALM_E2E_OUTPUT" "restart/resume reset Calm and restored a tool row" From 6af83310535b3d48b1d35bcc1597a0b7b98902d7 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micka=C3=ABl=20R=C3=A9mond?= Date: Fri, 2 Oct 2026 00:17:50 +0200 Subject: [PATCH 13/33] fix(bin): preserve hold reasons and reject invalid completion inventories (#6331) * fix(bin): encode captain-hold reasons and reject self-inventory in complete hold now stores a reason with parentheses, line breaks, or percent signs through a reversible percent encoding that every reader decodes, instead of refusing it. hold --origin records the origin on the held task, and complete refuses the origin as its own inventory entry and an entry held for a different origin; holds with no recorded origin are accepted and flagged. * fix(review): Decode marked hold reasons consistently across readers * fix(review): Remove unnecessary lifecycle test dispatch * fix(review): Correct hold origin identity and inventory recovery * fix(review): Record origins before placing backend holds * fix(document): Clarify captain-hold validation and reason reader documentation * fix(ci): Fixed both findings: failed backend holds restore the previous origin, and invalid base64/UTF-8 reasons remain verbatim. Added regressions and documented valid-literal ambiguity. Both failures were reproduced before fixes. Verification: 54 lifecycle tests and 9 wrapper tests passed; 7 Beads-specific cases skipped because tasks-axi is markdown-only. Focused lint and diff checks passed. No pipeline or publication actions performed --- .../skills/captain-hold-lifecycle/SKILL.md | 1 + bin/fm-afk-return.sh | 5 +- bin/fm-captain-hold.sh | 146 ++++- bin/fm-fleet-snapshot.sh | 9 +- bin/fm-hold-reason-lib.sh | 78 +++ bin/fm-session-start.sh | 10 +- bin/fm-tasks-axi.sh | 16 +- bin/fm-test-run.sh | 2 +- docs/captain-hold-lifecycle.md | 20 +- tests/fm-afk-return.test.sh | 1 + tests/fm-captain-hold-lifecycle.test.sh | 522 +++++++++++++++++- 11 files changed, 765 insertions(+), 45 deletions(-) create mode 100644 bin/fm-hold-reason-lib.sh diff --git a/.agents/skills/captain-hold-lifecycle/SKILL.md b/.agents/skills/captain-hold-lifecycle/SKILL.md index eaea0c27411..438f6353b2a 100644 --- a/.agents/skills/captain-hold-lifecycle/SKILL.md +++ b/.agents/skills/captain-hold-lifecycle/SKILL.md @@ -19,6 +19,7 @@ The agent performs the semantic inventory because scripts must not infer captain Every unresolved question that belongs to the captain and is discovered while producing, reading, presenting, or ending an investigation or visual review must be carried by a captain-held task in the authoritative backlog of the home that owns the originating work before that work or review may be treated as complete. For a Lavish board-backed handoff, pass the reply through `bin/fm-procevent-lavish.sh arm --agent-reply-file` before appending the status; the adapter owns version-specific acceptance ordering. Prefer holding the work item the question gates over minting a new row; create a new task only when no work item exists to hold. +The originating investigation or review is never its own inventory entry, so hold a separate task for the call and pass `--origin ` so `complete` can check it. Put the question and its options in the hold reason, and keep one held task per genuine gate: a multi-question review is one held task pointing at its report, not a row per question. Represent that task with exactly one board card that consolidates its questions and options; never fan one task id into duplicate same-key cards. Register or re-hold through `bin/fm-captain-hold.sh hold`, which is idempotent per task id. After inventorying the whole report and review surface, run `bin/fm-captain-hold.sh complete` with every captain-held task id, or with `--none` only when the reviewed surface leaves nothing waiting on the captain. diff --git a/bin/fm-afk-return.sh b/bin/fm-afk-return.sh index 99e6bc86ac7..b162e6bba40 100755 --- a/bin/fm-afk-return.sh +++ b/bin/fm-afk-return.sh @@ -71,6 +71,9 @@ RETURN_GRACE=${FM_GUARD_GRACE:-300} # shellcheck source=bin/fm-afk-contract.sh . "$SCRIPT_DIR/fm-afk-contract.sh" CONTRACT="$SCRIPT_DIR/fm-afk-contract.sh" +# Functions only: decodes the stored hold reasons the catch-up listing shows. +# shellcheck source=bin/fm-hold-reason-lib.sh +. "$SCRIPT_DIR/fm-hold-reason-lib.sh" usage() { sed -n '2,11p' "${BASH_SOURCE[0]}" | sed 's/^# \{0,1\}//' @@ -591,7 +594,7 @@ render_return_brief() { # -decision-` @@ -223,6 +231,9 @@ DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" # shellcheck source=bin/fm-wake-lib.sh # shellcheck disable=SC1091 . "$SCRIPT_DIR/fm-wake-lib.sh" +# shellcheck source=bin/fm-hold-reason-lib.sh +# shellcheck disable=SC1091 +. "$SCRIPT_DIR/fm-hold-reason-lib.sh" # shellcheck source=bin/fm-parent-channel-lib.sh # shellcheck disable=SC1091 . "$SCRIPT_DIR/fm-parent-channel-lib.sh" @@ -792,27 +803,106 @@ write_hold_set_stamp() { # + printf '%s\n' "$1" | sed -n 's/^Captain hold origin: \(.*\)$/\1/p' | head -1 +} + +task_identity() { + local id=$1 + if task_show "$id"; then + id=$(show_field_value "$TASK_SHOW_OUTPUT" id) + validate_slug backend-task-id "$id" + elif ! printf '%s\n' "$TASK_SHOW_OUTPUT" | grep -q '^code: NOT_FOUND$'; then + fail "could not resolve the backend identity of $id" + fi + printf '%s' "$id" +} + +write_hold_origin() { # + local id=$1 body=$2 origin=$3 stamp rest new_body tmp + body=$(decode_shown_value "$body") \ + || fail "could not decode the existing body for $id" + stamp=$(printf '%s\n' "$body" | sed -n 1p) + [ -n "$(body_hold_set_timestamp "$body")" ] \ + || fail "task $id lost its hold-set stamp before its origin was recorded" + rest=$(printf '%s\n' "$body" | sed 1d | awk '!/^Captain hold origin: /' \ + | awk 'NF || started { started = 1; print }') + new_body=$stamp + if [ -n "$origin" ]; then + new_body=$(printf '%s\nCaptain hold origin: %s' "$stamp" "$origin") + fi + if [ -n "$rest" ]; then + new_body=$(printf '%s\n\n%s' "$new_body" "$rest") + fi + tmp=$(umask 077; mktemp "${TMPDIR:-/tmp}/fm-captain-hold-origin.XXXXXX") \ + || fail "cannot stage the hold origin" + if ! printf '%s\n' "$new_body" > "$tmp"; then + rm -f -- "$tmp" + fail "cannot stage the hold origin for $id" + fi + if ! tasks_axi update "$id" --body-file "$tmp" >/dev/null; then + rm -f -- "$tmp" + fail "could not record the hold origin on $id" + fi + rm -f -- "$tmp" +} + +refuse_self_inventory() { + local origin=$1 entry=$2 meta="$STATE/$1.meta" + if list_has_key "$(meta_value "$meta" decision_keys)" "$entry"; then + fail "origin $origin cannot be its own captain-call inventory entry; historical decision_keys in $meta still contains $entry; hold a separate captain task with --origin $origin, replace only $entry in the final decision_keys= line with that task id while preserving all other entries, then re-run complete $origin " + fi + fail "origin $origin cannot be its own captain-call inventory entry; hold a separate captain task for the call and list that task" +} + # Resolve one entry and verify the row it names is durably captain-held. A # resolution failure that is not the read bound keeps resolve_entry's own # status - its stderr already named the entry; 124 means the backend never # answered, which is not the same as an unknown entry and must not be spent -# as absence. On success prints " " so the caller can keep the -# attestation evidence. -verify_entry_durable() { # ; prints " " - local origin=$1 entry=$2 resolved resolve_status=0 +# as absence. The result carries the attestation evidence and whether an +# origin was recorded, so completion can disclose the legacy fallback. +verify_entry_durable() { # ; prints " " + local origin=$1 entry=$2 resolved resolve_status=0 id how stored origin_state=unrecorded origin_id stored_id + # The origin task is never its own captain-call inventory: it is the work the + # calls were found in, so accepting it would let a refused hold look recorded. + if [ -n "$origin" ] && [ "$origin" != "$BINDING_ANY" ] && [ "$entry" = "$origin" ]; then + refuse_self_inventory "$origin" "$entry" + fi resolved=$(resolve_entry "$origin" "$entry") || resolve_status=$? if [ "$resolve_status" -ne 0 ]; then [ "$resolve_status" -ne 124 ] \ || fail "the backlog backend exceeded its read bound resolving $entry" exit "$resolve_status" fi - printf '%s\n' "$resolved" - verify_hold_durable "${resolved%% *}" + id=${resolved%% *} + how=${resolved##* } + verify_hold_durable "$id" + id=$(show_field_value "$TASK_SHOW_OUTPUT" id) + validate_slug backend-task-id "$id" + stored=$(body_hold_origin "$(decode_shown_value "$(show_field "$TASK_SHOW_OUTPUT" body)")") + origin_id=$origin + if [ -n "$origin" ] && [ "$origin" != "$BINDING_ANY" ]; then + origin_id=$(task_identity "$origin") || exit $? + [ "$id" != "$origin_id" ] || refuse_self_inventory "$origin" "$entry" + fi + if [ -n "$stored" ]; then + if [ -n "$origin" ] && [ "$origin" != "$BINDING_ANY" ]; then + stored_id=$(task_identity "$stored") || exit $? + if [ "$stored_id" != "$origin_id" ]; then + fail "captain-held task $id was held for origin $stored, not $origin; hold a task for $origin or list the right one" + fi + fi + origin_state=recorded + fi + printf '%s %s %s\n' "$id" "$how" "$origin_state" } command_hold() { local id=${1:-} title='' reason='' repo='' origin='' until='' show state existing_title body='' hold_kind hold_set occurrence - local existing_hold_kind='' existing_held='' preserve_hold_set=0 + local existing_hold_kind='' existing_held='' preserve_hold_set=0 stored_reason previous_origin='' hold_status=0 [ "$#" -ge 1 ] || { usage >&2; exit 2; } shift while [ "$#" -gt 0 ]; do @@ -827,8 +917,9 @@ command_hold() { shift done validate_slug task-id "$id" - validate_one_line reason "$reason" - case "$reason" in *'('*|*')'*) fail "reason must not contain parentheses (tasks-axi hold contract)" ;; esac + [ -n "$reason" ] || fail "reason must not be empty" + # bin/fm-hold-reason-lib.sh owns the storage constraint and reversible encoding. + stored_reason=$(fm_hold_reason_encode "$reason") || fail "could not encode the hold reason" if [ -n "$origin" ]; then validate_slug origin-id "$origin" fi @@ -889,12 +980,25 @@ command_hold() { task_show_or_fail "$id" "task $id disappeared while recording its hold-set stamp" [ -n "$(body_hold_set_timestamp "$(show_field_value "$show" body)")" ] \ || fail "task $id did not retain its hold-set stamp" + if [ -n "$origin" ]; then + origin=$(task_identity "$origin") || exit $? + previous_origin=$(body_hold_origin "$(show_field_value "$show" body)") + write_hold_origin "$id" "$(show_field "$show" body)" "$origin" || exit $? + fi if [ -n "$until" ]; then - tasks_axi hold "$id" --reason "$reason" --kind captain --until "$until" >/dev/null \ - || fail "could not hold task $id for the captain" + tasks_axi hold "$id" --reason "$stored_reason" --kind captain --until "$until" >/dev/null \ + || hold_status=$? else - tasks_axi hold "$id" --reason "$reason" --kind captain >/dev/null \ - || fail "could not hold task $id for the captain" + tasks_axi hold "$id" --reason "$stored_reason" --kind captain >/dev/null \ + || hold_status=$? + fi + if [ "$hold_status" -ne 0 ]; then + # A refused re-hold must not associate the previous hold or answer with a + # new origin. Restore the old line verbatim, without resolving it again. + if [ -n "$origin" ]; then + write_hold_origin "$id" "$(show_field "$show" body)" "$previous_origin" || exit $? + fi + fail "could not hold task $id for the captain" fi task_show "$id" || fail "task $id disappeared while holding it" show=$TASK_SHOW_OUTPUT @@ -1627,7 +1731,7 @@ reconcile_note() { command_complete() { local origin=${1:-} meta previous='' supplied='' keys='' entry key status_file open has_meta=0 transfer_rc transfers=() resolved - local resolved_how attested_by_prefix='' + local resolved_how attested_by_prefix='' origin_state unrecorded_origin='' [ "$#" -ge 2 ] || { usage >&2; exit 2; } validate_slug origin-id "$origin" shift @@ -1659,8 +1763,13 @@ command_complete() { while IFS= read -r entry; do [ -n "$entry" ] || continue resolved=$(verify_entry_durable "$origin" "$entry") || exit $? + origin_state=${resolved##* } + resolved=${resolved% *} resolved_how=${resolved##* } resolved=${resolved%% *} + if [ "$origin_state" = unrecorded ]; then + unrecorded_origin="${unrecorded_origin}${unrecorded_origin:+ }$resolved" + fi if [ "$resolved_how" = migrated-prefix ]; then attested_by_prefix="${attested_by_prefix}${attested_by_prefix:+ }$entry=$resolved" fi @@ -1702,8 +1811,9 @@ EOF fi fi fi - printf 'complete: %s captain-call inventory reviewed%s%s\n' "$origin" "${keys:+ ($keys)}" \ - "${attested_by_prefix:+ [attested through the configured prefix: $attested_by_prefix]}" + printf 'complete: %s captain-call inventory reviewed%s%s%s\n' "$origin" "${keys:+ ($keys)}" \ + "${attested_by_prefix:+ [attested through the configured prefix: $attested_by_prefix]}" \ + "${unrecorded_origin:+ [no recorded origin on: $unrecorded_origin; not checked against $origin]}" } command_verify() { diff --git a/bin/fm-fleet-snapshot.sh b/bin/fm-fleet-snapshot.sh index 666d03b8d6c..7f830bff272 100755 --- a/bin/fm-fleet-snapshot.sh +++ b/bin/fm-fleet-snapshot.sh @@ -229,6 +229,8 @@ esac . "$SCRIPT_DIR/fm-landed-lib.sh" # FM_LANDED_JQ_DEFS: the shared landed selector # shellcheck source=bin/fm-merge-authority-lib.sh . "$SCRIPT_DIR/fm-merge-authority-lib.sh" +# shellcheck source=bin/fm-hold-reason-lib.sh +. "$SCRIPT_DIR/fm-hold-reason-lib.sh" usage() { cat <<'EOF' @@ -381,13 +383,14 @@ first_pr_url_in_file() { # grep -Eo 'https?://[^[:space:])"]+/pull/[0-9]+' "$1" 2>/dev/null | head -1 } -backlog_json() { # [] - defaults to this home's $BACKLOG +backlog_json() ( # [] - defaults to this home's $BACKLOG local backlog=${1:-$BACKLOG} if [ ! -f "$backlog" ]; then jq -n --arg path "$backlog" '{path:$path,present:false,records:[]}' return 0 fi + set -o pipefail # shellcheck disable=SC2094 jq -Rn --arg path "$backlog" --arg today "$SNAPSHOT_TODAY" --arg now "$SNAPSHOT_NOW" \ --argjson age_days "$FM_SNAPSHOT_UNDATED_HOLD_AGE_DAYS" ' @@ -570,8 +573,8 @@ backlog_json() { # [] - defaults to this home's $BACKLOG | .captain_actionable = (.hold_bucket == "live") else . end) | del(.section,.order) - ' < "$backlog" -} + ' < "$backlog" | fm_hold_reason_decode_stream json +) SNAPSHOT_TASK_DIR= SNAPSHOT_TASK_METAS=() diff --git a/bin/fm-hold-reason-lib.sh b/bin/fm-hold-reason-lib.sh new file mode 100644 index 00000000000..6b6b6a6d9ff --- /dev/null +++ b/bin/fm-hold-reason-lib.sh @@ -0,0 +1,78 @@ +#!/usr/bin/env bash +# fm-hold-reason-lib.sh - the one reversible encoding of a captain-hold reason. +# +# tasks-axi stores a hold reason as one markdown line inside a parenthesised tag, +# so its own `hold` refuses parentheses and line breaks. A decision reason is +# ordinary prose, so bin/fm-captain-hold.sh encodes the reason where it writes +# it and every reader that shows it decodes it again, instead of banning the +# characters. Stored reasons use the reserved fm-hold-v1: prefix followed by +# base64-encoded UTF-8 text. Unmarked reasons are plain text. Readers decode only +# the hold-reason field, once, and keep line breaks in quoted output strings. +# +# Source this file; it defines functions only. + +# fm_hold_reason_encode : print the storable form, no trailing newline. +fm_hold_reason_encode() { + printf '%s' "$1" | perl -MMIME::Base64=encode_base64 -0777 -ne \ + 'print "fm-hold-v1:", encode_base64($_, "")' +} + +# fm_hold_reason_decode_stream [toon|markdown|json]: decode marked reason fields. +fm_hold_reason_decode_stream() { + perl -MJSON::PP -MMIME::Base64=encode_base64,decode_base64 -MEncode=decode,FB_CROAK -e ' + use strict; + use warnings; + binmode STDIN, ":encoding(UTF-8)"; + binmode STDOUT, ":encoding(UTF-8)"; + my $format = shift; + my $json = JSON::PP->new->allow_nonref; + sub decode_reason { + my ($value) = @_; + return $value unless defined($value) && $value =~ /^fm-hold-v1:(.*)\z/s; + my $payload = $1; + my $bytes = decode_base64($payload); + return $value unless encode_base64($bytes, "") eq $payload; + # Historical literals with valid base64 and UTF-8 remain indistinguishable + # from encoded reasons; malformed payloads retain their stored text. + my $decoded = eval { decode("UTF-8", $bytes, FB_CROAK) }; + return $@ ? $value : $decoded; + } + sub decode_field { + my ($raw) = @_; + my $value = $raw =~ /^"/ ? $json->decode($raw) : $raw; + my $decoded = decode_reason($value); + return $decoded eq $value ? $raw : $json->encode($decoded); + } + if ($format eq "json") { + local $/; + my $snapshot = $json->decode(); + for my $record (@{$snapshot->{records}}) { + $record->{hold_reason} = decode_reason($record->{hold_reason}) + if exists $record->{hold_reason}; + } + print $json->encode($snapshot), "\n"; + exit; + } + my ($column, $task); + while (my $line = ) { + if ($format eq "markdown") { + $line =~ s{^([-*] .*\(hold:\s*)(fm-hold-v1:[A-Za-z0-9+/]*={0,2})(\).*)$} + {$1 . decode_field($2) . $3}e; + } elsif ($line =~ /^tasks\[\d+\]\{([^}]*)\}:\n?$/) { + my @names = split /,/, $1; + ($column) = grep { $names[$_] eq "hold_reason" } 0 .. $#names; + $task = 0; + } elsif (defined($column) && $line =~ /^ (.*)\n?$/) { + my @fields = $1 =~ /("(?:[^"\\]|\\.)*"|[^,]+)/g; + $fields[$column] = decode_field($fields[$column]); + $line = " " . join(",", @fields) . "\n"; + } elsif ($task && $line =~ /^ hold_reason: (.*)\n?$/) { + $line = " hold_reason: " . decode_field($1) . "\n"; + } elsif ($line !~ /^ /) { + $column = undef; + $task = $line eq "task:\n"; + } + print $line; + } + ' "${1:-toon}" +} diff --git a/bin/fm-session-start.sh b/bin/fm-session-start.sh index 9ddaadc88ba..67f25704265 100755 --- a/bin/fm-session-start.sh +++ b/bin/fm-session-start.sh @@ -371,6 +371,8 @@ PRIMARY_HARNESS=$("$SCRIPT_DIR/fm-harness.sh" 2>/dev/null || printf unknown) . "$SCRIPT_DIR/fm-wake-lib.sh" # shellcheck source=bin/fm-line-cap-lib.sh . "$SCRIPT_DIR/fm-line-cap-lib.sh" +# shellcheck source=bin/fm-hold-reason-lib.sh +. "$SCRIPT_DIR/fm-hold-reason-lib.sh" # One tasks-axi compatibility verdict per session start. The probe costs three # tasks-axi subprocesses and this digest needs the same answer twice - here for @@ -470,7 +472,7 @@ print_backlog_manual_compact() { } } } - ' "$path" + ' "$path" | fm_hold_reason_decode_stream markdown } # tasks-axi closes every listing with its own help block. This section composes @@ -522,11 +524,11 @@ print_backlog_tasks_axi_compact() { printf 'compact backlog listing (tasks-axi; done rows omitted; every in-flight, held, and blocked row shown in full; ready queued bounded to %s; task bodies omitted)\n' \ "$QUEUED_LIMIT" printf '\nin flight:\n' - printf '%s\n' "$in_flight" | strip_axi_help + printf '%s\n' "$in_flight" | fm_hold_reason_decode_stream | strip_axi_help printf '\nheld (captain- or time-gated; an in-flight item that is also held appears in both groups):\n' - printf '%s\n' "$held" | strip_axi_help + printf '%s\n' "$held" | fm_hold_reason_decode_stream | strip_axi_help printf '\nblocked queued:\n' - printf '%s\n' "$blocked" | strip_axi_help + printf '%s\n' "$blocked" | fm_hold_reason_decode_stream | strip_axi_help printf '\nready queued (dispatchable now):\n' print_ready_queued_bounded "$ready" return 0 diff --git a/bin/fm-tasks-axi.sh b/bin/fm-tasks-axi.sh index b773014a115..7f0dc1822c9 100755 --- a/bin/fm-tasks-axi.sh +++ b/bin/fm-tasks-axi.sh @@ -14,6 +14,10 @@ # stores it verbatim as a link, which lifecycle transitions record relative to # that same root. # +# `show` (including `view`) and `list` decode stored captain-hold reasons +# through bin/fm-hold-reason-lib.sh, which owns the field-only decoding contract. +# Decoded reasons use quoted strings so embedded line breaks remain intact. +# # Why it exists: a bare `tasks-axi` resolves the tracked `.tasks.toml` paths # against its working directory, so from the code root it forks the queue # whenever the home lives elsewhere; docs/configuration.md ("Backlog backend") @@ -46,7 +50,8 @@ # - a markdown `/backlog.md` that is itself a symlink, because the # first write would replace the link with a private copy, exactly the fork # this command exists to prevent. Lifecycle transitions refuse the same file. -# Otherwise the exit status is tasks-axi's own. +# Otherwise the exit status is tasks-axi's own, unless decoding a read fails; +# in that case the decoder's nonzero status is returned. set -u SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -57,6 +62,8 @@ DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" . "$SCRIPT_DIR/fm-tasks-axi-lib.sh" # shellcheck source=bin/fm-backlog-transition-lib.sh disable=SC1091 . "$SCRIPT_DIR/fm-backlog-transition-lib.sh" +# shellcheck source=bin/fm-hold-reason-lib.sh disable=SC1091 +. "$SCRIPT_DIR/fm-hold-reason-lib.sh" usage() { awk ' @@ -137,4 +144,11 @@ else fi cd "$FM_BACKLOG_AXI_ROOT" || fail "cannot enter the backlog root $FM_BACKLOG_AXI_ROOT" +case "${1:-}" in + show|view|list) + set -o pipefail + tasks-axi ${ARGS[@]+"${ARGS[@]}"} | fm_hold_reason_decode_stream + exit $? + ;; +esac exec tasks-axi ${ARGS[@]+"${ARGS[@]}"} diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 4702f403f7f..cd4fd0e33f6 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -1650,7 +1650,7 @@ families_for_changed_path() { bin/fm-lint.sh|bin/fm-lint-workflows.sh|bin/fm-install-shellcheck.sh|\ bin/fm-install-actionlint.sh|\ bin/fm-brief.sh|bin/fm-ensure-agents-md.sh|bin/fm-crew-state.sh|\ - bin/fm-captain-hold.sh|bin/fm-decision-hold.sh|bin/fm-supervision*|bin/fm-transition-lib.sh|\ + bin/fm-captain-hold.sh|bin/fm-hold-reason-lib.sh|bin/fm-decision-hold.sh|bin/fm-supervision*|bin/fm-transition-lib.sh|\ bin/fm-tmux-lib.sh|bin/fm-marker-lib.sh|bin/fm-operational-input.sh|bin/fm-tasks-axi-lib.sh|\ bin/fm-vendor-auth-probe.sh|\ bin/fm-primary-scope-lib.sh|bin/fm-project-mode.sh|bin/fm-forge-detect.sh|bin/fm-promote.sh|\ diff --git a/docs/captain-hold-lifecycle.md b/docs/captain-hold-lifecycle.md index 7a186a4e6af..5c19372265c 100644 --- a/docs/captain-hold-lifecycle.md +++ b/docs/captain-hold-lifecycle.md @@ -51,8 +51,9 @@ It works in this order: 1. It uses an existing task, or creates one when nothing exists to hold. 2. It records the task's UTC hold-set timestamp as the leading line of the task body. -3. It invokes the underlying tasks-axi hold operation. -4. It verifies both records. +3. When `--origin` is supplied, it records the origin on the task, replacing any previous association. +4. It invokes the underlying tasks-axi hold operation. +5. It verifies the hold and timestamp. Publishing the stamp first ensures a snapshot cannot observe a newly captain-held task without the timestamp that defines its age. @@ -62,6 +63,10 @@ Repeat and edge cases: - Re-holding released work starts a new timestamped lifecycle. - A closed task is refused rather than reopened. - `--until` stores the captain's own deferral date through tasks-axi's date gate. +- Before the backend hold runs, `--origin` records the origin the call is held for on its own `Captain hold origin:` body line, which `complete` and `verify` check using backend identities rather than alias spellings. + If that write fails, the backend hold is not attempted. +- The reason may contain parentheses, semicolons, quotes, and line breaks. + [`bin/fm-hold-reason-lib.sh`](../bin/fm-hold-reason-lib.sh) owns the storage encoding and compatibility rules; [`bin/fm-tasks-axi.sh --help`](../bin/fm-tasks-axi.sh) owns the public read commands and output contract. ### Answering a call (`answer`) @@ -103,6 +108,10 @@ A post-teardown visual review can complete against the surviving report and dura `complete` accepts `--none` as an explicit semantic inventory result. `--none` is refused while the origin still has a lifecycle-open keyed status decision. Before recording completion, `complete` verifies every listed task against tasks-axi. +The origin is never its own inventory entry, so a hold that failed cannot be vouched for by the origin row. +For a historical inventory that names its own origin, hold a separate captain task with `--origin`, replace only the invalid entry in the final `decision_keys=` line of the origin metadata with that task id while preserving all other entries, and re-run `complete`. +An entry whose recorded origin differs from the one being completed is refused. +An entry with no recorded origin, such as a hold made before origins were recorded or without `--origin`, is accepted on the durability check alone and named in the output. With a non-empty inventory, `complete` appends a `captain-held [key=]` transfer event for every still-open keyed status decision. The event names the reviewed inventory. @@ -114,7 +123,7 @@ Scout teardown calls the read-only `verify` subcommand after checking for the re `verify` checks three things: - The recorded attestation exists. -- Every recorded inventory entry is still durable: actively captain-held, or carrying a recorded answer. +- Every recorded inventory entry still passes the [completion inventory checks](#recording-a-reviewed-inventory-complete). - No keyed status decision opened after the last `complete`. A keyed status decision opened after the last `complete` makes `verify` fail, and re-running `complete` is the repair. @@ -479,7 +488,7 @@ It then finishes any still-recorded dependency-edge cleanup without rewriting th ## Verification record -The focused end-to-end regression suite is `tests/fm-captain-hold-lifecycle.test.sh`, using only synthetic `sample` identities and decision text. +The focused end-to-end regression suite is `tests/fm-captain-hold-lifecycle.test.sh`, using only synthetic identities and decision text. It proves the behaviors below. The suite does not test the accepted merge-to-cleanup re-hold window or asynchronous queued-forge landing because those events occur after the locally serialized merge command has returned. @@ -528,7 +537,8 @@ The suite does not test the accepted merge-to-cleanup re-hold window or asynchro ### Legacy paths -- Every legacy path works: composed identities through the shim, pre-collapse `decision_keys=` metadata, routed-resolution replay, and a concrete-origin binding. +- Composed identities through the shim, valid pre-collapse `decision_keys=` inventories, routed-resolution replay, and a concrete-origin binding remain supported. + Historical self-inventories require the [documented repair](#recording-a-reviewed-inventory-complete). ### Task-body read-back cases diff --git a/tests/fm-afk-return.test.sh b/tests/fm-afk-return.test.sh index b66ead8c0da..fd589be2965 100755 --- a/tests/fm-afk-return.test.sh +++ b/tests/fm-afk-return.test.sh @@ -35,6 +35,7 @@ install_runner() { # cp "$ROOT/bin/fm-afk-contract.sh" "$dir/bin/" cp "$ROOT/bin/fm-branch-outcome.sh" "$dir/bin/" cp "$ROOT/bin/fm-tasks-axi-lib.sh" "$dir/bin/" + cp "$ROOT/bin/fm-hold-reason-lib.sh" "$dir/bin/" cp "$ROOT/bin/fm-backlog-transition-lib.sh" "$dir/bin/" # The merge-notification marker reader behind the brief's landed section. cp "$ROOT/bin/fm-pr-lib.sh" "$dir/bin/" diff --git a/tests/fm-captain-hold-lifecycle.test.sh b/tests/fm-captain-hold-lifecycle.test.sh index a20a8c5de52..2864a3e30eb 100755 --- a/tests/fm-captain-hold-lifecycle.test.sh +++ b/tests/fm-captain-hold-lifecycle.test.sh @@ -88,6 +88,15 @@ run_captain() { # FM_CONFIG_OVERRIDE="$home/config" "$ROOT/bin/fm-captain-hold.sh" "$@" } +# Completes 's captain-call inventory through a separate held task, because +# the origin task is never accepted as its own inventory entry. +complete_through_sibling() { # + local home=$1 id=$2 + run_captain "$home" hold "$id-call" --title "Sibling captain call for $id" \ + --reason "captain must decide the sibling call" --repo sample --origin "$id" >/dev/null \ + && run_captain "$home" complete "$id" "$id-call" +} + request_reconciles() { # ... local home=$1 source_id=$2 id shift 2 @@ -357,7 +366,7 @@ case "${1:-}" in show) case "${2:-}" in @KNOWN@) ;; - *) printf 'error: no task %s in this backlog\n' "${2:-}" >&2; exit 1 ;; + *) printf 'error: no task %s in this backlog\ncode: NOT_FOUND\n' "${2:-}" >&2; exit 1 ;; esac printf '%s\n' 'task:' printf ' id: %s\n' "$2" @@ -2575,7 +2584,7 @@ test_teardown_never_closes_a_captain_held_task() { run_captain "$home" hold "$id" \ --reason "captain must choose inline or by-reference attachments" >/dev/null \ || fail "could not hold the originating work item for the captain" - run_captain "$home" complete "$id" "$id" >/dev/null \ + complete_through_sibling "$home" "$id" >/dev/null \ || fail "completion gate failed with the origin as its own captain call" run_teardown "$home" "$id" > "$home/teardown.out" 2> "$home/teardown.err" \ @@ -2666,7 +2675,7 @@ test_retained_row_artifacts_survive_captain_answers() { > "$home/data/$retained_id/report.md" run_captain "$home" hold "$retained_id" --reason "captain must choose the report follow-up" \ >/dev/null || fail "could not hold the retained report" - run_captain "$home" complete "$retained_id" "$retained_id" >/dev/null \ + complete_through_sibling "$home" "$retained_id" >/dev/null \ || fail "completion gate failed for the retained report" run_teardown "$home" "$retained_id" > "$home/retained-teardown.out" \ 2> "$home/report-teardown.err" \ @@ -2687,7 +2696,7 @@ test_retained_row_artifacts_survive_captain_answers() { run_captain "$home" hold "$precedence_id" \ --reason "captain must choose the report follow-up" >/dev/null \ || fail "could not hold the report precedence fixture" - run_captain "$home" complete "$precedence_id" "$precedence_id" >/dev/null \ + complete_through_sibling "$home" "$precedence_id" >/dev/null \ || fail "completion gate failed for the report precedence fixture" run_teardown "$home" "$precedence_id" > "$home/precedence-teardown.out" \ 2> "$home/precedence-teardown.err" \ @@ -2815,7 +2824,7 @@ test_retained_row_artifacts_survive_captain_answers() { printf '# Released report\n' > "$home/data/$released_id/report.md" run_captain "$home" hold "$released_id" --reason "captain report release pending" \ >/dev/null || fail "could not hold the released report" - run_captain "$home" complete "$released_id" "$released_id" >/dev/null \ + complete_through_sibling "$home" "$released_id" >/dev/null \ || fail "completion gate failed for the released report" printf 'Release the completed report.\n' > "$home/released-answer.txt" run_captain "$home" answer "$released_id" --release \ @@ -2920,7 +2929,7 @@ test_interrupted_cleanup_keeps_the_captain_call_recoverable() { printf '# Failed cleanup\n\nThe captain call remains open.\n' > "$home/data/$id/report.md" run_captain "$home" hold "$id" --reason "captain must choose after cleanup retry" >/dev/null \ || fail "could not hold the cleanup-failure fixture" - run_captain "$home" complete "$id" "$id" >/dev/null \ + complete_through_sibling "$home" "$id" >/dev/null \ || fail "completion gate failed for the cleanup-failure fixture" cat > "$home/fakebin/treehouse" <<'SH' #!/usr/bin/env bash @@ -2979,7 +2988,7 @@ test_answer_before_cleanup_replay_preserves_the_retained_report() { printf '# Interrupted cleanup\n\nThe captain call remains open.\n' > "$home/data/$id/report.md" run_captain "$home" hold "$id" --reason "captain must choose after interrupted cleanup" \ >/dev/null || fail "could not hold the answer-before-replay fixture" - run_captain "$home" complete "$id" "$id" >/dev/null \ + complete_through_sibling "$home" "$id" >/dev/null \ || fail "completion gate failed for the answer-before-replay fixture" cat > "$home/fakebin/treehouse" <<'SH' #!/usr/bin/env bash @@ -3090,7 +3099,7 @@ test_unusable_pending_close_record_names_its_reason() { printf '# Unusable pending close\n\nThe captain call remains open.\n' > "$home/data/$id/report.md" run_captain "$home" hold "$id" --reason "captain must choose after interrupted cleanup" \ >/dev/null || fail "could not hold the unusable pending-close fixture" - run_captain "$home" complete "$id" "$id" >/dev/null \ + complete_through_sibling "$home" "$id" >/dev/null \ || fail "completion gate failed for the unusable pending-close fixture" cat > "$home/fakebin/treehouse" <<'SH' #!/usr/bin/env bash @@ -3154,7 +3163,12 @@ EOF || fail "could not hold the relocated answer-before-replay fixture" PATH="$home/fakebin:$PATH" FM_HOME="$home" FM_STATE_OVERRIDE="$home/state" \ FM_DATA_OVERRIDE="$data" FM_CONFIG_OVERRIDE="$home/config" \ - "$ROOT/bin/fm-captain-hold.sh" complete "$id" "$id" >/dev/null \ + "$ROOT/bin/fm-captain-hold.sh" hold "$id-call" --title "Sibling captain call" \ + --reason "captain must decide the sibling call" --repo sample --origin "$id" >/dev/null \ + || fail "could not hold the sibling captain call" + PATH="$home/fakebin:$PATH" FM_HOME="$home" FM_STATE_OVERRIDE="$home/state" \ + FM_DATA_OVERRIDE="$data" FM_CONFIG_OVERRIDE="$home/config" \ + "$ROOT/bin/fm-captain-hold.sh" complete "$id" "$id-call" >/dev/null \ || fail "completion gate failed for the relocated answer-before-replay fixture" cat > "$home/fakebin/treehouse" <<'SH' #!/usr/bin/env bash @@ -3231,7 +3245,12 @@ EOF || fail "could not hold the relocated work item" PATH="$home/fakebin:$PATH" FM_HOME="$home" FM_STATE_OVERRIDE="$home/state" \ FM_DATA_OVERRIDE="$data" FM_CONFIG_OVERRIDE="$home/config" \ - "$ROOT/bin/fm-captain-hold.sh" complete "$id" "$id" >/dev/null \ + "$ROOT/bin/fm-captain-hold.sh" hold "$id-call" --title "Sibling captain call" \ + --reason "captain must decide the sibling call" --repo sample --origin "$id" >/dev/null \ + || fail "could not hold the sibling captain call" + PATH="$home/fakebin:$PATH" FM_HOME="$home" FM_STATE_OVERRIDE="$home/state" \ + FM_DATA_OVERRIDE="$data" FM_CONFIG_OVERRIDE="$home/config" \ + "$ROOT/bin/fm-captain-hold.sh" complete "$id" "$id-call" >/dev/null \ || fail "completion gate failed for the relocated captain hold" PATH="$home/fakebin:$PATH" FM_ROOT_OVERRIDE="$ROOT" FM_HOME="$home" \ @@ -4061,7 +4080,7 @@ PM > "$home/data/$scout/report.md" run_captain "$home" hold "$scout" --reason "captain must choose" >/dev/null \ || fail "could not hold the investigation for the captain" - run_captain "$home" complete "$scout" "$scout" >/dev/null \ + complete_through_sibling "$home" "$scout" >/dev/null \ || fail "the completion gate failed with the origin as its own captain call" PERL5LIB="$shim" PERL5OPT=-MFmNoNonrefDefault \ run_teardown "$home" "$scout" > "$home/nonref.out" 2> "$home/nonref.err" \ @@ -4099,7 +4118,7 @@ retain_row_with_body() { # || fail "could not give $id a body carrying non-ASCII characters" run_captain "$home" hold "$id" --reason "captain must choose" >/dev/null \ || fail "could not hold $id for the captain" - run_captain "$home" complete "$id" "$id" >/dev/null \ + complete_through_sibling "$home" "$id" >/dev/null \ || fail "the completion gate failed for $id" run_teardown "$home" "$id" > "$home/$id.out" 2> "$home/$id.err" \ || fail "cleanup of captain-held $id failed: $(cat "$home/$id.err")" @@ -4137,6 +4156,485 @@ test_retained_body_keeps_its_utf8_bytes() { pass "cleanup preserves every byte of a retained body's non-ASCII characters" } +# A refused hold must never read as a recorded one. The gate used to accept the +# origin as its own inventory whenever the origin row looked durable, so a hold +# that failed just before `complete ` left a satisfied gate +# with no captain call recorded. +test_origin_is_never_its_own_inventory_entry() { + local home id + home=$(make_home origin-self-inventory) + id=sample-self-review + mkdir -p "$home/data/$id" + tasks_in "$home" add "$id" "Investigate sample self review" --kind scout --repo sample --start >/dev/null \ + || fail "could not create the investigation fixture" + write_origin_meta "$home" "$id" + printf 'done: report complete\n' > "$home/state/$id.status" + if run_captain "$home" hold "$id" --reason "" >/dev/null 2> "$home/hold.err"; then + fail "hold accepted an empty reason" + fi + if run_captain "$home" complete "$id" "$id" > "$home/self.out" 2> "$home/self.err"; then + fail "complete accepted the origin as its own inventory after a failed hold" + fi + assert_grep "cannot be its own captain-call inventory entry" "$home/self.err" \ + "the refusal does not say why the origin was rejected" + assert_no_grep "decisions_reviewed=1" "$home/state/$id.meta" \ + "the refused completion recorded an inventory attestation" + + # Holding the origin row itself must not let it vouch for itself either. + run_captain "$home" hold "$id" --reason "captain must choose" >/dev/null \ + || fail "could not hold the origin row" + if run_captain "$home" complete "$id" "$id" > "$home/held.out" 2> "$home/held.err"; then + fail "complete accepted a held origin row as its own inventory" + fi + pass "complete refuses the origin as its own captain-call inventory" +} + +# `hold --origin` records which origin a call was held for, and `complete` +# refuses a task held for a different origin. A hold recorded before that +# record existed, or without --origin, still verifies and is flagged. +test_complete_refuses_an_entry_held_for_another_origin() { + local home id other o out + home=$(make_home origin-mismatch) + id=sample-first-review + other=sample-second-review + for o in "$id" "$other"; do + mkdir -p "$home/data/$o" + tasks_in "$home" add "$o" "Investigate $o" --kind scout --repo sample --start >/dev/null \ + || fail "could not create the $o fixture" + write_origin_meta "$home" "$o" + printf 'done: report complete\n' > "$home/state/$o.status" + done + run_captain "$home" hold sample-other-call --title "Call for the second review" \ + --reason "captain must decide" --repo sample --origin "$other" >/dev/null \ + || fail "could not hold the call recorded for the second review" + if run_captain "$home" complete "$id" sample-other-call > "$home/mismatch.out" 2> "$home/mismatch.err"; then + fail "complete accepted an entry held for a different origin" + fi + assert_grep "was held for origin $other, not $id" "$home/mismatch.err" \ + "the refusal does not name both origins" + assert_no_grep "decisions_reviewed=1" "$home/state/$id.meta" \ + "the refused completion recorded an inventory attestation" + + run_captain "$home" hold sample-own-call --title "Call for the first review" \ + --reason "captain must decide" --repo sample --origin "$id" >/dev/null \ + || fail "could not hold the call recorded for the first review" + out=$(run_captain "$home" complete "$id" sample-own-call) \ + || fail "complete refused an entry held for its own origin" + assert_not_contains "$out" "no recorded origin" \ + "an entry with a recorded origin was flagged as unrecorded" + + tasks_in "$home" add sample-old-call "Call held before origins were recorded" --kind captain --repo sample >/dev/null \ + || fail "could not create the older call" + tasks_in "$home" hold sample-old-call --reason "captain must decide" --kind captain >/dev/null \ + || fail "could not hold the older call" + out=$(run_captain "$home" complete "$other" sample-old-call) \ + || fail "complete refused an older hold with no recorded origin" + assert_contains "$out" "no recorded origin on: sample-old-call" \ + "an older hold with no recorded origin was not flagged" + pass "complete refuses an entry held for another origin and flags one with none recorded" +} + +test_hold_origins_precede_backend_holds() { + local home phase timing failure id shown origin until_args=() + for phase in new active released; do + for timing in plain dated; do + home=$(make_home "origin-failure-$phase-$timing") + id=sample-call + for origin in origin-a origin-b; do + tasks_in "$home" add "$origin" "Review $origin" --kind scout --repo sample >/dev/null \ + || fail "could not create $origin" + write_origin_meta "$home" "$origin" + done + if [ "$phase" != new ]; then + run_captain "$home" hold "$id" --title "Separate call" --reason "Choose for A" \ + --origin origin-a >/dev/null || fail "could not establish the original association" + fi + if [ "$phase" = released ]; then + printf 'Release this work.\n' > "$home/answer.txt" + run_captain "$home" answer "$id" --release --decision-file "$home/answer.txt" >/dev/null \ + || fail "could not release the original hold" + fi + cat > "$home/fakebin/tasks-axi" <<'SH' +#!/usr/bin/env bash +if [ "${1:-}" = show ] && [ "${2:-}" = origin-b ] && [ -f "$FM_HOME/fail-lookup" ]; then + : > "$FM_HOME/lookup-refused" + printf 'error: origin read failed\ncode: READ_FAILED\n' >&2 + exit 2 +fi +if [ "${1:-}" = update ] && [ -f "$FM_HOME/fail-write" ]; then + previous='' + for arg in "$@"; do + if [ "$previous" = --body-file ] && grep -qx 'Captain hold origin: origin-b' "$arg"; then + : > "$FM_HOME/write-refused" + exit 9 + fi + previous=$arg + done +fi +if [ "${1:-}" = hold ] && [ "${2:-}" != --help ]; then + "$REAL_TASKS_AXI" show "$2" --full > "$FM_HOME/before-backend-hold" || exit $? + if [ -f "$FM_HOME/fail-hold" ]; then + : > "$FM_HOME/hold-refused" + exit 9 + fi +fi +exec "$REAL_TASKS_AXI" "$@" +SH + chmod +x "$home/fakebin/tasks-axi" + until_args=() + [ "$timing" != dated ] || until_args=(--until 2099-01-01) + for failure in lookup write hold; do + : > "$home/fail-$failure" + if run_captain "$home" hold "$id" --title "Separate call" --reason "Choose for B" \ + --origin origin-b ${until_args[@]+"${until_args[@]}"} > "$home/hold.out" 2> "$home/hold.err"; then + fail "$phase $timing hold succeeded despite an origin $failure failure" + fi + assert_present "$home/$failure-refused" "the failure did not reach the origin $failure" + if [ "$failure" = hold ]; then + assert_present "$home/before-backend-hold" "$phase $timing failure never reached the backend hold" + assert_grep 'Captain hold origin: origin-b' "$home/before-backend-hold" \ + "the failed backend hold did not see the new association" + rm "$home/before-backend-hold" + else + assert_absent "$home/before-backend-hold" "$phase $timing origin $failure failure reached the backend hold" + fi + shown=$(tasks_in "$home" show "$id" --full) + assert_not_contains "$shown" 'Captain hold origin: origin-b' \ + "$phase $timing origin $failure failure published the new association" + if [ "$phase" = active ]; then + assert_contains "$shown" 'held: yes' "an origin $failure failure lifted an existing hold" + else + assert_contains "$shown" 'held: no' "$phase $timing origin $failure failure left the task held" + fi + if [ "$phase" != new ]; then + assert_contains "$shown" 'Captain hold origin: origin-a' \ + "$phase $timing origin $failure failure lost the original association" + fi + rm "$home/fail-$failure" + if [ "$phase" = new ] && run_captain "$home" complete origin-a "$id" \ + > "$home/unrelated.out" 2> "$home/unrelated.err"; then + fail "$timing origin $failure failure satisfied an unrelated inventory" + fi + if run_captain "$home" complete origin-b "$id" > "$home/complete.out" 2> "$home/complete.err"; then + fail "$phase $timing origin $failure failure satisfied completion for B" + fi + printf 'decisions_reviewed=1\ndecision_keys=%s\n' "$id" >> "$home/state/origin-b.meta" + if run_captain "$home" verify origin-b > "$home/verify.out" 2> "$home/verify.err"; then + fail "$phase $timing origin $failure failure verified an inventory for B" + fi + if [ "$phase" != new ]; then + run_captain "$home" complete origin-a "$id" >/dev/null \ + || fail "$phase $timing origin $failure failure invalidated completion for A" + run_captain "$home" verify origin-a >/dev/null \ + || fail "$phase $timing origin $failure failure invalidated verification for A" + fi + done + run_captain "$home" hold "$id" --reason "Choose for B" --origin origin-b \ + ${until_args[@]+"${until_args[@]}"} >/dev/null || fail "$phase $timing successful retry failed" + assert_present "$home/before-backend-hold" "the successful retry did not reach the backend hold" + shown=$(cat "$home/before-backend-hold") + assert_contains "$shown" 'Captain hold origin: origin-b' "the backend hold ran before the new origin was recorded" + assert_not_contains "$shown" 'Captain hold origin: origin-a' "the backend hold ran with the old association" + shown=$(tasks_in "$home" show "$id" --full) + assert_contains "$shown" 'held: yes' "the successful retry did not hold the task" + assert_contains "$shown" 'Captain hold origin: origin-b' "a successful hold lost its association" + assert_not_contains "$shown" 'Captain hold origin: origin-a' "a successful hold retained the old association" + run_captain "$home" complete origin-b "$id" >/dev/null \ + || fail "a successful hold could not complete B" + run_captain "$home" verify origin-b >/dev/null || fail "a successful hold could not verify B" + if run_captain "$home" complete origin-a "$id" >/dev/null 2> "$home/old-origin.err"; then + fail "a successful reassociation still certified A" + fi + done + done + pass "new, active, and released holds require the origin first with and without deferral" +} + +test_historical_self_inventory_has_workable_repair() { + local home origin=sample-review keep=retained-call replacement=repair-call meta before out + home=$(make_home historical-self-inventory) + run_captain "$home" hold "$origin" --title "Old review call" --reason "Choose" >/dev/null \ + || fail "could not create the historical origin" + write_origin_meta "$home" "$origin" + for out in "$keep" "$replacement"; do + run_captain "$home" hold "$out" --title "Call $out" --reason "Choose" --origin "$origin" >/dev/null \ + || fail "could not create $out" + done + meta="$home/state/$origin.meta" + printf 'decisions_reviewed=1\ndecision_keys=%s,%s\n' "$origin" "$keep" >> "$meta" + before=$(cat "$meta") + for out in "$replacement" --none; do + if run_captain "$home" complete "$origin" "$out" > "$home/complete.out" 2> "$home/complete.err"; then + fail "complete accepted the historical self-inventory" + fi + assert_grep "historical decision_keys in $meta still contains $origin" "$home/complete.err" \ + "the historical refusal did not identify the persisted entry" + assert_grep 'replace only' "$home/complete.err" "the refusal omitted the repair instruction" + done + if run_captain "$home" verify "$origin" > "$home/verify.out" 2> "$home/verify.err"; then + fail "verify accepted the historical self-inventory" + fi + assert_grep "historical decision_keys in $meta still contains $origin" "$home/verify.err" \ + "verify omitted the historical repair instruction" + assert_equals "$before" "$(cat "$meta")" "refusing a historical inventory changed it" + sed "s/^decision_keys=$origin,$keep$/decision_keys=$replacement,$keep/" "$meta" > "$meta.repaired" + mv "$meta.repaired" "$meta" + run_captain "$home" complete "$origin" "$replacement" >/dev/null \ + || fail "the documented historical repair did not allow completion" + run_captain "$home" verify "$origin" >/dev/null || fail "the repaired inventory did not verify" + assert_equals "decision_keys=$replacement,$keep" "$(grep '^decision_keys=' "$meta" | tail -1)" \ + "repair lost a sibling inventory entry" + if run_captain "$home" complete "$origin" "$origin" >/dev/null 2> "$home/self.err"; then + fail "repair allowed a new self-inventory" + fi + pass "historical self-inventories name a workable repair that preserves sibling entries" +} + +test_inventory_compares_backend_identities() { + local home origin entry shown before + home=$(make_home backend-identities) + run_captain "$home" hold fm-o --title "Origin" --reason "Choose" >/dev/null \ + || fail "could not create the canonical origin" + tasks_in "$home" add fm-other "Other origin" --kind scout --repo sample >/dev/null \ + || fail "could not create the other origin" + cat > "$home/fakebin/tasks-axi" <<'SH' +#!/usr/bin/env bash +if [ "${1:-}" = show ] && [ "${2:-}" = o ] && [ -f "$FM_HOME/fail-identity" ]; then + printf 'error: origin read failed\ncode: READ_FAILED\n' >&2 + exit 2 +fi +if [ "$#" -ge 2 ]; then + case "$2" in + o|call|other) set -- "$1" "fm-$2" "${@:3}" ;; + esac +fi +exec "$REAL_TASKS_AXI" "$@" +SH + chmod +x "$home/fakebin/tasks-axi" + for origin in fm-o o; do + for entry in fm-o o; do + write_origin_meta "$home" "$origin" + if run_captain "$home" complete "$origin" "$entry" > "$home/self.out" 2> "$home/self.err"; then + fail "complete accepted aliased self-inventory $origin/$entry" + fi + assert_grep 'cannot be its own captain-call inventory entry' "$home/self.err" \ + "the alias refusal did not identify self-inventory" + printf 'decisions_reviewed=1\ndecision_keys=%s\n' "$entry" >> "$home/state/$origin.meta" + if run_captain "$home" verify "$origin" > "$home/verify.out" 2> "$home/verify.err"; then + fail "verify accepted aliased self-inventory $origin/$entry" + fi + assert_grep 'historical decision_keys' "$home/verify.err" "the alias repair diagnostic was missing" + done + write_origin_meta "$home" "$origin" + done + run_captain "$home" hold fm-call --title "Separate call" --reason "Choose" --origin o >/dev/null \ + || fail "could not hold a call using the origin alias" + shown=$(tasks_in "$home" show fm-call --full) + assert_contains "$shown" 'Captain hold origin: fm-o' "hold did not store the backend origin identity" + printf '%s\n' "$shown" | sed -n 's/^ body: //p' | jq -r . \ + | sed 's/^Captain hold origin: fm-o$/Captain hold origin: o/' > "$home/legacy-origin.txt" + tasks_in "$home" update fm-call --body-file "$home/legacy-origin.txt" >/dev/null \ + || fail "could not create a legacy stored alias" + for origin in fm-o o; do + for entry in fm-call call; do + run_captain "$home" complete "$origin" "$entry" >/dev/null \ + || fail "complete refused equivalent origin spellings for $origin/$entry" + run_captain "$home" verify "$origin" >/dev/null \ + || fail "verify refused equivalent origin spellings for $origin/$entry" + done + done + for origin in fm-other other; do + write_origin_meta "$home" "$origin" + if run_captain "$home" complete "$origin" call > "$home/other.out" 2> "$home/other.err"; then + fail "complete accepted another origin through $origin" + fi + assert_grep "was held for origin o, not $origin" "$home/other.err" "the alias mismatch was not identified" + printf 'decisions_reviewed=1\ndecision_keys=call\n' >> "$home/state/$origin.meta" + if run_captain "$home" verify "$origin" >/dev/null 2> "$home/other-verify.err"; then + fail "verify accepted another origin through $origin" + fi + done + : > "$home/fail-identity" + before=$(cat "$home/state/o.meta") + if run_captain "$home" complete o fm-call >/dev/null 2> "$home/read.err"; then + fail "an unreadable backend identity was treated as an absent origin" + fi + assert_grep 'could not resolve the backend identity of o' "$home/read.err" "the identity read failure was hidden" + assert_equals "$before" "$(cat "$home/state/o.meta")" "a failed identity read changed the inventory" + rm "$home/fail-identity" + write_origin_meta "$home" report-only + run_captain "$home" hold report-call --title "Report call" --reason "Choose" --origin report-only >/dev/null \ + || fail "an origin with metadata but no backlog row could not record a call" + run_captain "$home" complete report-only report-call >/dev/null \ + || fail "an origin with metadata but no backlog row could not complete" + run_captain "$home" verify report-only >/dev/null \ + || fail "an origin with metadata but no backlog row could not verify" + pass "completion and verification compare backend identities for entries and current or stored origins" +} + +# tasks-axi refuses parentheses and line breaks in a hold reason and stores the +# rest on one markdown line. The reason is encoded where it is written and +# decoded wherever it is shown, so prose with every awkward character survives. +test_hold_reason_round_trips_awkward_characters() { + local home id reason stored json shown start verb fields out raw rc raw_rc mode + local title legacy body quoted_reason quoted_title quoted_legacy expected_reason until_args=() + local malformed index=0 malformed_reasons=( + 'fm-hold-v1:/w==' 'fm-hold-v1:bm9ydGg=$' 'fm-hold-v1:bm9ydGg' 'fm-hold-v1:Zh==' + ) + home=$(make_home reason-round-trip) + title='Investigate literal %28, "fm-hold-v1:bm9ydGg="' + legacy='Visit https://example.test/%28literal%29 and %0A; fm-hold-v1:bm9ydGg=' + body=$'fm-hold-v1:bm9ydGg=\n hold_reason: "%28"\n' + quoted_title=$(jq -cn --arg value "$title" '$value') + quoted_legacy=$(jq -cn --arg value "$legacy" '$value') + tasks_in "$home" add sample-legacy-call "$title" --kind captain --repo sample >/dev/null \ + || fail "could not create the legacy call" + tasks_in "$home" hold sample-legacy-call --reason "$legacy" --kind captain >/dev/null \ + || fail "could not hold the legacy call" + printf '%s' "$body" > "$home/legacy-body.txt" + tasks_in "$home" update sample-legacy-call --body-file "$home/legacy-body.txt" >/dev/null \ + || fail "could not write the legacy body" + + # Historical literal reasons are persisted input, not encoder output. + for malformed in "${malformed_reasons[@]}"; do + id="sample-malformed-$index" + index=$((index + 1)) + tasks_in "$home" add "$id" "Historical reason $index" --kind captain --repo sample >/dev/null \ + || fail "could not create $id" + tasks_in "$home" hold "$id" --reason "$malformed" --kind captain >/dev/null \ + || fail "could not store the historical literal reason" + for verb in show view; do + out=$(FM_HOME="$home" "$ROOT/bin/fm-tasks-axi.sh" "$verb" "$id" --full) \ + || fail "public $verb failed on historical literal $malformed" + raw=$(tasks_in "$home" "$verb" "$id" --full) + assert_equals "$raw" "$out" "public $verb changed historical literal $malformed" + done + done + + for id in sample-reason-call sample-dated-call; do + until_args=() + reason=$' Pick route (north); say "yes" or \'no\' - 100% sure %28x%29, café\t\\slash\r\nSecond line\n\n' + if [ "$id" = sample-dated-call ]; then + until_args=(--until 2099-01-01) + reason='fm-hold-v1:bm9ydGg=' + fi + quoted_reason=$(jq -cn --arg value "$reason" '$value') + run_captain "$home" hold "$id" --title "$title" --reason "$reason" \ + --repo sample ${until_args[@]+"${until_args[@]}"} >/dev/null \ + || fail "hold refused the reason for $id" + stored=$(grep "^- \[ \] $id " "$home/data/backlog.md") \ + || fail "the held row is not on one backlog line" + assert_contains "$stored" "(hold: fm-hold-v1:" "the persisted reason has no encoding marker" + assert_contains "$stored" "(hold-kind: captain)" "the reason broke the hold-kind tag" + + for verb in show view; do + out=$(FM_HOME="$home" "$ROOT/bin/fm-tasks-axi.sh" "$verb" "$id") \ + || fail "public $verb failed for $id" + shown=$(printf '%s\n' "$out" | sed -n 's/^ hold_reason: //p') + printf '%s\n' "$shown" | jq -e --arg reason "$reason" '. == $reason' >/dev/null \ + || fail "public $verb changed the reason for $id" + assert_contains "$out" " title: $quoted_title" "public $verb changed the title" + done + for fields in hold_reason,body body,hold_reason,hold_until; do + out=$(FM_HOME="$home" "$ROOT/bin/fm-tasks-axi.sh" list --fields "$fields") \ + || fail "public list failed with $fields" + assert_contains "$out" "$quoted_reason" "public list changed the reason with $fields" + assert_contains "$out" "$quoted_title" "public list changed the title with $fields" + raw=$(tasks_in "$home" list --fields "$fields" | grep '^ sample-legacy-call,') + shown=$(printf '%s\n' "$out" | grep '^ sample-legacy-call,') + assert_equals "$raw" "$shown" "public list changed legacy or unrelated fields" + for malformed in "${malformed_reasons[@]}"; do + assert_contains "$out" "$malformed" "public list changed historical literal $malformed" + done + done + json=$(PATH="$home/fakebin:$PATH" FM_HOME="$home" FM_STATE_OVERRIDE="$home/state" \ + FM_DATA_OVERRIDE="$home/data" FM_CONFIG_OVERRIDE="$home/config" \ + "$ROOT/bin/fm-fleet-snapshot.sh" --json) || fail "fleet snapshot failed" + printf '%s' "$json" | jq -e --arg id "$id" --arg reason "$reason" --arg title "$title" \ + '.backlog.records[] | select(.id == $id) | .hold_reason == $reason and .title == $title' >/dev/null \ + || fail "fleet changed the reason or title for $id" + printf '%s' "$json" | jq -e --arg reason "$legacy" --arg title "$title" \ + '.backlog.records[] | select(.id == "sample-legacy-call") | + .hold_reason == $reason and .title == $title and .body_lines[0] == "fm-hold-v1:bm9ydGg="' >/dev/null \ + || fail "fleet changed legacy or unrelated fields" + for malformed in "${malformed_reasons[@]}"; do + printf '%s' "$json" | jq -e --arg reason "$malformed" \ + 'any(.backlog.records[]; .hold_reason == $reason)' >/dev/null \ + || fail "fleet changed historical literal $malformed" + done + done + + out=$(FM_HOME="$home" "$ROOT/bin/fm-tasks-axi.sh" show sample-legacy-call --full) + raw=$(tasks_in "$home" show sample-legacy-call --full) + assert_equals "$raw" "$out" "public show changed legacy or unrelated fields" + out=$(FM_HOME="$home" "$ROOT/bin/fm-tasks-axi.sh" list) + raw=$(tasks_in "$home" list) + assert_equals "$raw" "$out" "public list changed output with no reason column" + for verb in show list; do + out=$(FM_HOME="$home" "$ROOT/bin/fm-tasks-axi.sh" "$verb" --help) + raw=$(tasks_in "$home" "$verb" --help) + assert_equals "$raw" "$out" "public $verb changed help output" + done + out=$(FM_HOME="$home" "$ROOT/bin/fm-tasks-axi.sh" show nonexistent-call 2>&1) + rc=$? + raw=$(tasks_in "$home" show nonexistent-call 2>&1) + raw_rc=$? + [ "$raw_rc" -ne 0 ] || fail "the missing-task fixture unexpectedly exists" + expect_code "$raw_rc" "$rc" "public show missing task" + assert_equals "$raw" "$out" "public show changed a read error" + + expected_reason=$' Pick route (north); say "yes" or \'no\' - 100% sure %28x%29, café\t\\slash\r\nSecond line\n\n' + quoted_reason=$(jq -cn --arg value "$expected_reason" '$value') + for mode in tool manual fallback; do + case "$mode" in + manual) printf 'manual\n' > "$home/config/backlog-backend" ;; + fallback) + rm "$home/config/backlog-backend" + cat > "$home/fakebin/tasks-axi" <<'SH' +#!/usr/bin/env bash +if [ "${1:-}" = list ]; then + printf 'read failed: literal %%28 and fm-hold-v1:bm9ydGg=\n' >&2 + exit 1 +fi +exec "$REAL_TASKS_AXI" "$@" +SH + chmod +x "$home/fakebin/tasks-axi" + ;; + esac + start=$(PATH="$home/fakebin:$PATH" REAL_TASKS_AXI="$TASKS_AXI_BIN" FM_HOME="$home" \ + FM_STATE_OVERRIDE="$home/state" FM_DATA_OVERRIDE="$home/data" FM_CONFIG_OVERRIDE="$home/config" \ + FM_BOOTSTRAP_NETWORK=skip "$ROOT/bin/fm-session-start.sh" 2>&1 || true) + assert_contains "$start" "$quoted_reason" "startup $mode changed the encoded reason" + assert_contains "$start" 'Investigate literal %28' "startup $mode changed the title" + assert_contains "$start" "$legacy" "startup $mode changed the legacy reason" + assert_contains "$start" '"fm-hold-v1:bm9ydGg="' "startup $mode decoded a reason twice" + for malformed in "${malformed_reasons[@]}"; do + assert_contains "$start" "$malformed" "startup $mode changed historical literal $malformed" + done + if [ "$mode" = fallback ]; then + assert_contains "$start" 'read failed: literal %28 and fm-hold-v1:bm9ydGg=' \ + "startup changed unrelated error text" + fi + done + rm "$home/fakebin/tasks-axi" + out=$(PATH="$home/fakebin:$PATH" FM_HOME="$home" FM_STATE_OVERRIDE="$home/state" \ + FM_DATA_OVERRIDE="$home/data" FM_CONFIG_OVERRIDE="$home/config" \ + "$ROOT/bin/fm-afk-return.sh" check 2>&1 || true) + assert_contains "$out" "$quoted_reason" "return brief changed the encoded reason" + assert_contains "$out" "$quoted_title" "return brief changed the title" + assert_contains "$out" "$quoted_legacy" "return brief changed the legacy reason" + for malformed in "${malformed_reasons[@]}"; do + assert_contains "$out" "$malformed" "return brief changed historical literal $malformed" + done + pass "marked hold reasons round-trip through public reads, fleet, startup, and return without changing other fields" +} + +test_hold_reason_round_trips_awkward_characters +test_hold_origins_precede_backend_holds +test_historical_self_inventory_has_workable_repair +test_inventory_compares_backend_identities +test_origin_is_never_its_own_inventory_entry +test_complete_refuses_an_entry_held_for_another_origin test_uninventoried_report_decision_refuses_completion test_hold_decodes_a_bare_scalar_body_without_the_nonref_default test_retained_body_keeps_its_utf8_bytes From 8690c4117a298ee870f0f72f571c72b846b4c5fb Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Thu, 1 Oct 2026 17:08:12 -0700 Subject: [PATCH 14/33] fix: reclaim orphaned watcher arms on the next park (#6335) * fix(bin): take over the watcher cycle a main-only pass-through leaves An attended main-only pass-through leaves a successor watcher cycle running through main's handling turn. The session's next park attached to that cycle instead of owning it, so the successor's arm, orphaned by its host's exit, kept owning the watcher while the new park's arm polled it twice a second until the next close or the park boundary, hours later in a quiet second mate. A second-mate restart hit this every time, since its persist request is a main-only close. The host now records the successor it leaves for main, and the next host's first cycle runs bin/fm-watch-arm.sh --take-over on it: when that arm still owns the healthy watcher, the new arm stops it, reports a reason the cycle delivered first, and otherwise owns a fresh cycle. The stop's own downtime publication is undone over an acknowledged episode when no wake was appended in between, so the handover wakes nobody. * no-mistakes(review): Keep left-arm record until the orphaned arm is gone * no-mistakes(review): Relinquish successor arm only after durably recording it * no-mistakes(review): Relinquish successor only after its record reads back * no-mistakes(document): Clarify successor takeover guarantees and authoritative documentation * no-mistakes(ci): Fixed ci-1: acknowledgement restore now requires the exact taken-over arm/watcher ledger row with signal=TERM, awaited within a short bound. Otherwise takeover proceeds without erasing downtime. Added a self-exit regression confirmed failing before the fix and passing afterward; ordinary takeover tests and the full watcher-arm suite pass. Updated Generation reuse documentation. Syntax, diff checks, and ShellCheck pass with existing SC1091/SC2034 warnings excluded. ci-2 remains unchanged * no-mistakes(document): Clarify watcher take-over recovery and restart limits --- bin/fm-supervision-host.sh | 60 ++++++++++-- bin/fm-wake-lib.sh | 59 +++++++++++ bin/fm-watch-arm.sh | 122 +++++++++++++++++++---- docs/supervision-host.md | 4 +- docs/watcher-continuity.md | 4 + tests/fm-supervision-host.test.sh | 154 +++++++++++++++++++++++++++++ tests/fm-wake-queue.test.sh | 52 ++++++++++ tests/fm-watch-arm.test.sh | 156 ++++++++++++++++++++++++++++++ 8 files changed, 583 insertions(+), 28 deletions(-) diff --git a/bin/fm-supervision-host.sh b/bin/fm-supervision-host.sh index 90fd54836e9..4de9d6c705a 100755 --- a/bin/fm-supervision-host.sh +++ b/bin/fm-supervision-host.sh @@ -54,8 +54,11 @@ # before the close is printed, so supervision continues when the session # drops the handoff. It confirms no handling handoff, so the recovery # marker still reads downtime and the re-arm owner delivers the close to -# main. The watcher singleton lock makes the session's next arm attach to -# that cycle instead of starting a second one; +# main. The host records that successor's arm before relinquishing it +# (detach_successor owns the persistence check and failure path). The +# session's next park without --restart requests a take-over of its cycle +# rather than an ordinary attach; bin/fm-watch-arm.sh's --take-over header owns the +# conditions under which that restores a single owner and the fallback; # - away (an away record exists): every close goes to the engine. # Every turn that starts attended meets that rule again at its start, so a # close accepted away whose turn starts attended (the captain returned in @@ -132,7 +135,12 @@ # left running (recorded with identities, never by name), including the # engine descendants its turn recorded, removes that turn's files, and # releases the branch actor's leases; it releases them again after every -# engine turn. +# engine turn. It also reads the record of a successor a pass-through left for +# main: while that arm still runs under its recorded identity, the first cycle +# without --restart requests a take-over rather than an ordinary attach. +# Activation removes the +# record only once that identity is no longer alive, so a later host retries a +# take-over that left it running. # # STATE (all under state/, owned here): .supervision-host (this host's pid and # the processes it runs), .supervision-host-engine (the engine conversation: @@ -141,7 +149,8 @@ # report scope and the reports it recorded), .supervision-host-prompt and # .supervision-host-wake (the prompt and wake text of the current turn), # .supervision-host-mirror (the dialog-mirror feed while an attended wake is -# rendered), +# rendered), .supervision-host-left (the pid and identity of the successor arm a +# pass-through left running for main, until that arm is gone), # .supervision-host-health (the latch: errors, cooldown, and probe time, keyed # to the main session, engine, and model), and .supervision-host.log (a bounded # ledger of where every close went, with each engine turn's usage and @@ -231,6 +240,7 @@ HOST_LOG="$STATE/.supervision-host.log" ENGINE_PID_FILE="$STATE/.supervision-host.engine-pid" HEALTH_FILE="$STATE/.supervision-host-health" MIRROR_FEED="$STATE/.supervision-host-mirror" +LEFT_RECORD="$STATE/.supervision-host-left" HOST_PID=$$ HOST_STARTED_SECONDS=$SECONDS @@ -252,6 +262,9 @@ ENGINE_SUBSHELL= SUCCESSOR_PID= SUCCESSOR_OUT= ENGINE_RUNNING=0 +# The successor arm a predecessor's pass-through left for main, which the +# first cycle takes over. +LEFT_ARM= # The running turn's result and diagnostics files, removed by the cleanup when # the host is stopped mid-turn. TURN_RESULT= @@ -362,6 +375,18 @@ activate() { done rm -f "$STATE"/.supervision-host-arm.* "$STATE"/.supervision-host-descendants.* "$STATE"/.supervision-host-result.* \ "$STATE"/.supervision-host-errors.* "$STATE"/.supervision-host-readback.* "$TURN_FILE" "$MIRROR_FEED" 2>/dev/null || true + # The successor a pass-through left for main: the first cycle takes it over + # while it still answers to its recorded identity, and its record goes only + # once it does not. + if [ -f "$LEFT_RECORD" ]; then + pid='' identity='' + IFS="$(printf '\t')" read -r pid identity < "$LEFT_RECORD" || true + if fm_pid_alive "$pid" && [ -n "$identity" ] && [ "$(identity_of "$pid")" = "$identity" ]; then + LEFT_ARM=$pid + else + rm -f "$LEFT_RECORD" + fi + fi printf 'host\t%s\t%s\n' "$HOST_PID" "$(identity_of "$HOST_PID")" > "$HOST_RECORD" || return 1 release_branch_leases } @@ -637,13 +662,27 @@ start_successor() { # done } -# Drop the successor from this host's cleanup without stopping it. The shell -# signals background jobs when it exits, and this arm's handler would then -# stop the watcher, so disown it first. The capture file stays tracked so the -# EXIT trap unlinks it; the arm already holds that descriptor and keeps -# waiting on the watcher. +# Record the successor for the next host to take over, then drop it from this +# host's cleanup without stopping it. A successor whose record does not read +# back as a regular file holding exactly its pid and identity stays tracked, +# so the cleanup stops it and main's next turn end arms a fresh cycle; that +# returns 1. The shell signals background jobs when it exits, and this arm's +# handler would then stop the watcher, so disown it first. The capture file +# stays tracked so the EXIT trap unlinks it; the arm already holds that +# descriptor and keeps waiting on the watcher. detach_successor() { + local identity tmp= [ -n "${SUCCESSOR_PID:-}" ] || return 0 + identity=$(identity_of "$SUCCESSOR_PID") + if [ -z "$identity" ] || ! tmp=$(mktemp "$LEFT_RECORD.tmp.XXXXXX" 2>/dev/null) \ + || ! printf '%s\t%s\n' "$SUCCESSOR_PID" "$identity" > "$tmp" 2>/dev/null \ + || ! mv -f "$tmp" "$LEFT_RECORD" 2>/dev/null \ + || [ -L "$LEFT_RECORD" ] || [ ! -f "$LEFT_RECORD" ] \ + || [ "$(cat "$LEFT_RECORD" 2>/dev/null)" != "$SUCCESSOR_PID"$'\t'"$identity" ]; then + [ -z "$tmp" ] || rm -f "$tmp" "$LEFT_RECORD/${tmp##*/}" 2>/dev/null || true + log_line "pass-through successor-unrecorded $(printf '%s\n' "$REASON" | head -n 1)" + return 1 + fi disown "$SUCCESSOR_PID" 2>/dev/null || true forget_process "$SUCCESSOR_PID" SUCCESSOR_PID= @@ -1008,6 +1047,9 @@ log_line "start gen=$GEN primary=$PRIMARY" # The first cycle. if [ "$FIRST_ARM_RESTART" -eq 1 ]; then start_arm "$OWNER_PREDECESSOR" --restart +elif [ -n "$LEFT_ARM" ]; then + log_line "take-over arm=$LEFT_ARM" + start_arm "$OWNER_PREDECESSOR" --take-over "$LEFT_ARM" else start_arm "$OWNER_PREDECESSOR" fi || { echo "watcher: FAILED - the supervision host could not start a watcher cycle"; exit 1; } diff --git a/bin/fm-wake-lib.sh b/bin/fm-wake-lib.sh index 60a9d289090..dfa387079b0 100755 --- a/bin/fm-wake-lib.sh +++ b/bin/fm-wake-lib.sh @@ -971,9 +971,64 @@ _fm_recovery_marker_reopen_announced() { fm_lock_release "$FM_WAKE_QUEUE_LOCK" } +# The handover rule for a watcher stopped by bin/fm-watch-arm.sh --take-over +# (docs/watcher-continuity.md "Generation reuse" owns it). The snapshot reads +# the marker token and the queue's append sequence under both locks before the +# stop; handover-restore puts an acknowledged token back only while that +# sequence is unchanged and the marker reads the fresh pending downtime the +# stopped watcher's own close published. +FM_RECOVERY_HANDOVER_TOKEN= +FM_RECOVERY_HANDOVER_SEQ= +fm_recovery_marker_handover_snapshot() { # + local marker=$1 lock + FM_RECOVERY_HANDOVER_TOKEN= + FM_RECOVERY_HANDOVER_SEQ= + lock="${marker}.lock" + fm_lock_acquire_wait "$FM_WAKE_QUEUE_LOCK" || return 1 + if ! fm_lock_acquire_wait "$lock"; then + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return 1 + fi + if fm_recovery_marker_read "$marker"; then + # shellcheck disable=SC2034 # Read by callers after this function returns. + FM_RECOVERY_HANDOVER_TOKEN=$FM_RECOVERY_MARKER_TOKEN + fi + # shellcheck disable=SC2034 # Read by callers after this function returns. + FM_RECOVERY_HANDOVER_SEQ=$(cat "$STATE/.wake-queue.seq" 2>/dev/null || true) + fm_lock_release "$lock" + fm_lock_release "$FM_WAKE_QUEUE_LOCK" +} + +_fm_recovery_marker_handover_restore() { + local marker=$1 token=$2 seq=$3 lock status=0 + case "$token" in acked:*) ;; *) return 0 ;; esac + lock="${marker}.lock" + fm_lock_acquire_wait "$FM_WAKE_QUEUE_LOCK" || return 1 + if ! fm_lock_acquire_wait "$lock"; then + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return 1 + fi + if [ "$(cat "$STATE/.wake-queue.seq" 2>/dev/null || true)" = "$seq" ] \ + && fm_recovery_marker_read "$marker"; then + case "$FM_RECOVERY_MARKER_TOKEN" in + pending:downtime:*) + if [ "${FM_RECOVERY_MARKER_TOKEN##*:}" != "${token##*:}" ]; then + _fm_recovery_marker_restore_token_locked "$marker" "$token" || status=1 + fi + ;; + esac + fi + fm_lock_release "$lock" + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return "$status" +} + fm_recovery_transition() { local marker=$1 action=$2 target=${3:-} value=${4:-} bound=${5:-} case "$action" in + handover-restore) + _fm_recovery_marker_handover_restore "$marker" "$target" "$value" + ;; publish) _fm_recovery_marker_publish "$marker" "${target:-downtime}" "$bound" ;; @@ -1035,6 +1090,10 @@ fm_recovery_marker_reopen_announced() { fm_recovery_transition "$1" reopen-announced } +fm_recovery_marker_handover_restore() { # + fm_recovery_transition "$1" handover-restore "$2" "$3" +} + # fm_lock_reap_dead_link # Remove a link lock whose owner is dead without a nested mutex. Renaming the # dead owner directory to this process's tombstone elects exactly one reaper, diff --git a/bin/fm-watch-arm.sh b/bin/fm-watch-arm.sh index 1932c0b9414..3a996a0320e 100755 --- a/bin/fm-watch-arm.sh +++ b/bin/fm-watch-arm.sh @@ -68,6 +68,17 @@ # bin/fm-watch.sh`: that pattern matches every firstmate home's watcher # (secondmate homes run the same script) and would kill siblings. # +# --take-over : own the cycle that arm owns, for an owner +# that left a successor cycle running through main's turn and now parks again +# (bin/fm-supervision-host.sh). Only when this home's healthy watcher is that +# arm's own child, it stops that watcher by its locked identity: a cycle that +# delivered a reason before the stop landed reports it exactly as an attached +# arm would, and otherwise this arm owns a fresh cycle as a plain arm does. +# Recovery restoration follows docs/watcher-continuity.md "Generation reuse"; +# an unconfirmed stop leaves downtime for the fresh cycle's recovery check. +# Any other watcher, or one that outlives the stop, +# is attached to exactly as a plain arm attaches. +# # --stop: the same home-scoped stop without re-arming, for an owner that ends # its own supervision cycle on purpose (the supervision host's park boundary, # bin/fm-supervision-host.sh). The stopped watcher publishes downtime exactly @@ -137,9 +148,10 @@ ARM_PID=${BASHPID:-$$} case "$CYCLE_LOG_MAX_BYTES" in ''|*[!0-9]*|0) CYCLE_LOG_MAX_BYTES=262144 ;; esac case "$CYCLE_LOG_KEEP_LINES" in ''|*[!0-9]*|0) CYCLE_LOG_KEEP_LINES=1000 ;; esac -# The lifecycle ledger is diagnostic evidence, not a supervision dependency. -# Writes are bounded and best-effort so an observability failure cannot stall an -# otherwise healthy watcher cycle. +# Lifecycle writes are bounded and best-effort so an observability failure +# cannot stall an otherwise healthy watcher cycle. Take-over also uses the +# owner's row as stop evidence; missing evidence takes the safe recovery path +# (docs/watcher-continuity.md "Generation reuse"). cycle_clean_field() { printf '%s' "$1" | tr '\t\r\n' ' ' | cut -c1-512 } @@ -237,9 +249,10 @@ cycle_log_append() { # A persistent adapter passes the arm pid that just closed. Once this new arm # verifies its watcher, update that predecessor's final record in place so the # one-record-per-cycle ledger captures the actual successor outcome without an -# extra synthetic lifecycle row. +# extra synthetic lifecycle row. A taking-over arm names itself instead, so its +# record of the cycle it took over names the cycle it started. cycle_mark_predecessor_successor() { - local successor=$1 predecessor=${FM_WATCH_PREDECESSOR_ARM_PID:-} i tmp + local successor=$1 predecessor=${2:-${FM_WATCH_PREDECESSOR_ARM_PID:-}} i tmp case "$predecessor" in ''|*[!0-9]*) return 0 ;; esac @@ -324,31 +337,35 @@ fail_unexplained_cycle() { return 1 } -# Close a cycle whose reason line this arm could not read against the bounded -# terminal-delivery ledger the watcher publishes before releasing its lock. -close_unobserved_cycle() { - local i reason clean_identity record_pid record_identity record_reason +# Read the reason the current cycle's watcher recorded in the bounded +# terminal-delivery ledger it publishes before releasing its lock. Sets +# DELIVERED_REASON; fails when no record matches the cycle's pid and identity. +DELIVERED_REASON= +cycle_delivered_reason() { + local i clean_identity record_pid record_identity record_reason + DELIVERED_REASON= clean_identity=$(printf '%s' "$cycle_watcher_identity" | tr '\t\r\n' ' ') i=0 while ! fm_lock_try_acquire "$WATCH_DELIVERY_LOCK"; do - [ "$i" -lt 20 ] || { - fail_unexplained_cycle - return 1 - } + [ "$i" -lt 20 ] || return 1 sleep 0.02 i=$((i + 1)) done - reason= if [ -f "$WATCH_DELIVERY_LOG" ]; then while IFS=$'\t' read -r record_pid record_identity record_reason; do if [ "$record_pid" = "$cycle_watcher_pid" ] && [ "$record_identity" = "$clean_identity" ]; then - reason=$record_reason + DELIVERED_REASON=$record_reason fi done < "$WATCH_DELIVERY_LOG" fi fm_lock_release "$WATCH_DELIVERY_LOCK" - if [ -n "$reason" ]; then - printf '%s\n' "$reason" + [ -n "$DELIVERED_REASON" ] +} + +# Close a cycle whose reason line this arm could not read against that ledger. +close_unobserved_cycle() { + if cycle_delivered_reason; then + printf '%s\n' "$DELIVERED_REASON" return 0 fi fail_unexplained_cycle @@ -461,10 +478,17 @@ handling_successor_generation() { mode=arm handling_generation= handling_watcher_pid= +take_over_arm_pid= case "${1:-}" in ''|arm|--arm) mode=arm ;; --restart) mode=restart ;; --stop) mode=stop ;; + --take-over) + mode=take-over + take_over_arm_pid=${2:-} + case "$take_over_arm_pid" in ''|*[!0-9]*) echo "watcher: invalid take-over arm pid" >&2; exit 2 ;; esac + [ "$#" -eq 2 ] || { echo "watcher: unexpected take-over arguments" >&2; exit 2; } + ;; --handling-delivered) mode=handling-delivered handling_generation=${2:-} @@ -474,7 +498,7 @@ case "${1:-}" in case "$handling_watcher_pid" in ''|*[!0-9]*) echo "watcher: invalid successor watcher pid" >&2; exit 2 ;; esac [ "$#" -eq 4 ] || { echo "watcher: unexpected handling delivery arguments" >&2; exit 2; } ;; - *) echo "usage: $(basename "$0") [--restart | --stop | --handling-delivered GENERATION --watcher-pid PID]" >&2; exit 2 ;; + *) echo "usage: $(basename "$0") [--restart | --stop | --take-over ARM_PID | --handling-delivered GENERATION --watcher-pid PID]" >&2; exit 2 ;; esac if [ "$mode" = handling-delivered ]; then @@ -524,6 +548,67 @@ if [ "$mode" = stop ]; then exit 0 fi +# Stop the watcher the named arm owns, by its locked identity, and wait for it +# to exit (header, --take-over). Returns 3 after printing the reason that cycle +# delivered before the stop landed, 0 once it stopped without delivering, and +# 1 when it was not stopped (its handover state was unreadable, or it outlived +# the stop), which leaves it to the plain attach below. +take_over_cycle() { # + local pid=$1 i owner_signal + cycle_begin "$pid" attached "$2" + fm_recovery_marker_handover_snapshot "$STATE/.watcher-down" || return 1 + if attached_holder_live "$pid"; then + kill -TERM "$pid" 2>/dev/null || true + fi + i=0 + while [ "$i" -lt 50 ] && fm_pid_alive "$pid"; do + sleep 0.1 + i=$((i + 1)) + done + if fm_pid_alive "$pid"; then + return 1 + fi + if cycle_delivered_reason; then + cycle_log_append unknown unknown taken-over-delivered-wake none + printf '%s\n' "$DELIVERED_REASON" + return 3 + fi + # Only the owner can wait on this watcher and distinguish our TERM from a + # self-exit that raced the stop. Give its post-wait ledger append a short bound. + i=0 + owner_signal= + while [ "$i" -lt 50 ]; do + owner_signal=$(awk -F '\t' -v arm="$take_over_arm_pid" -v watcher="$pid" ' + $1 == "arm_pid=" arm && $2 == "watcher_pid=" watcher { signal = $7 } + END { sub(/^signal=/, "", signal); print signal } + ' "$CYCLE_LOG" 2>/dev/null || true) + [ -z "$owner_signal" ] || break + sleep 0.02 + i=$((i + 1)) + done + if [ "$owner_signal" = TERM ]; then + fm_recovery_marker_handover_restore "$STATE/.watcher-down" \ + "$FM_RECOVERY_HANDOVER_TOKEN" "$FM_RECOVERY_HANDOVER_SEQ" || true + cycle_log_append unknown unknown taken-over none + else + cycle_log_append unknown unknown taken-over-unconfirmed-stop none + fi + return 0 +} + +TAKEN_OVER=0 +if [ "$mode" = take-over ]; then + mode=arm + if healthy_watcher \ + && [ "$(ps -o ppid= -p "$HEALTHY_PID" 2>/dev/null | tr -d ' ')" = "$take_over_arm_pid" ]; then + take_over_cycle "$HEALTHY_PID" "$HEALTHY_IDENTITY" + case $? in + 0) TAKEN_OVER=1 ;; + 3) exit 0 ;; + esac + fi +fi + # If a genuinely live+fresh watcher already holds the lock, do not start a second # one - attach to that cycle and wait until it ends so the harness notify fires # then, not as an immediate empty wake. (--restart skips this: it just stopped @@ -672,6 +757,7 @@ while :; do exit 1 fi cycle_mark_predecessor_successor "started:$child" + [ "$TAKEN_OVER" -eq 0 ] || cycle_mark_predecessor_successor "started:$child" "$ARM_PID" if [ -n "$handling_generation" ]; then echo "watcher: started pid=$child (beacon fresh) recovery-generation=$handling_generation" else diff --git a/docs/supervision-host.md b/docs/supervision-host.md index 6faad8fb2b1..3b6e2fb26ce 100644 --- a/docs/supervision-host.md +++ b/docs/supervision-host.md @@ -109,7 +109,9 @@ The host asks the Pi branch's offer rule (`branchOfferForWake`, through `bin/fm- So a close reaches main off Pi exactly when it would on Pi: a check trigger, a decision-owned signal or stale trigger, and a scan that is unsafe or holds nothing for the branch stay main's. On that main-only pass-through the host starts the successor watcher cycle and leaves it running, then prints the close unchanged. It leaves the watcher's recovery marker reading downtime, confirming no handling handoff, because the re-arm owner delivers a close to main only while that marker reads downtime. -The watcher's singleton lock makes the session's next arm attach to that cycle instead of starting a second one. +The session's next park without `--restart` requests a take-over to restore a single host-owned arm; the [host header](../bin/fm-supervision-host.sh) owns successor persistence and cleanup, and the [arm header](../bin/fm-watch-arm.sh) owns take-over eligibility and fallback. +OpenCode and omp still launch the host with `--restart`, which takes precedence over recorded take-over and lacks its acknowledgement-preserving handover; changing that first-cycle path remains a follow-up. +The host-off Claude Stop hook's detached handling successor is also unchanged; see [Claude handling successor](watcher-continuity.md#claude-handling-successor). It also passes the close through unchanged, with no added line, when any of these holds (`fm_supervision_host_attended_ready` in `bin/fm-supervision-engine-lib.sh` owns the list): - The home names no usable engine. diff --git a/docs/watcher-continuity.md b/docs/watcher-continuity.md index c2641b368ba..86f4d4e5e2d 100644 --- a/docs/watcher-continuity.md +++ b/docs/watcher-continuity.md @@ -226,6 +226,9 @@ A downtime republication of a pending episode reuses its generation. A watcher close leaves an announced downtime episode announced, while a successful durable append opens a fresh pending generation so a live watcher can recover the new work. An announced handling episode becomes pending downtime on the same generation because its handling turn may have been interrupted. That handling republication gives a successor exactly one recovery presentation without orphaning the acknowledgement already printed for that generation. +A watcher stopped so an arm can take its cycle over (`bin/fm-watch-arm.sh --take-over`) publishes downtime like any close, but the taking arm restores an acknowledged episode that stop reopened only when the taken-over arm's cycle-ledger row for that exact arm and watcher records the watcher ending by the take-over's TERM and no wake was appended in between. +The taking arm waits within a short bound for that row; a missing row or any other signal leaves downtime for the fresh cycle's ordinary recovery wake, while take-over still proceeds. +Any other episode is left for the next cycle's arm check. ### What an acknowledgement retires @@ -441,6 +444,7 @@ They also prove that a legacy or handoff-phase watcher marker from an absent rep - A watcher close inside the handling window that must leave the printed acknowledgement valid. - A re-arm whose recovery cycle is slowed after confirmation and must still surface rather than read as a watcher that stayed live. - The self-healing moved-generation acknowledgement that consumes its handled rows and names its remedy. +- A take-over that stays quiet after a confirmed TERM, still surfaces queued work and self-exit downtime, and attaches without stopping a cycle the named arm does not own. - The disposable-checkout arm refusal. - The home-gone and state-gone watcher exits. - The test reaper that stops a watcher armed for a temporary home. diff --git a/tests/fm-supervision-host.test.sh b/tests/fm-supervision-host.test.sh index 3ee445b9309..0f64411baae 100755 --- a/tests/fm-supervision-host.test.sh +++ b/tests/fm-supervision-host.test.sh @@ -1335,6 +1335,157 @@ test_successor_close_during_main_turn_is_delivered_at_the_next_turn_end() { pass "host+hook: a successor close that lands during main's turn is delivered at the next turn end" } +# The arm processes running from 's bin, one " " per line. +# A command substitution inside an arm is a forked copy that shows the same +# command line, so a process whose parent is itself an arm is not counted. +home_arms() { # + ps -A -o pid= -o ppid= -o command= 2>/dev/null \ + | awk -v arm="$1/bin/fm-watch-arm.sh" ' + $3 ~ /(^|\/)bash$/ && $4 == arm { ppid[$1] = $2; order[++n] = $1 } + END { for (i = 1; i <= n; i++) if (!(ppid[order[i]] in ppid)) print order[i], ppid[order[i]] }' +} +parent_of() { ps -o ppid= -p "$1" 2>/dev/null | tr -d ' '; } + +# True once the park's own arm owns the home's only watcher cycle: exactly one +# arm runs from the home, it is the host's child, and it is the watcher's parent. +host_owns_the_only_cycle() { # + local home=$1 host watcher arm arms + host=$(awk -F '\t' '$1 == "host" { print $2; exit }' "$home/state/.supervision-host" 2>/dev/null) + watcher=$(cat "$home/state/.watch.lock/pid" 2>/dev/null) + [ -n "$host" ] && [ -n "$watcher" ] && kill -0 "$watcher" 2>/dev/null || return 1 + arms=$(home_arms "$home") + [ "$(printf '%s\n' "$arms" | grep -c .)" -eq 1 ] || return 1 + arm=$(parent_of "$watcher") + [ "$arms" = "$arm $host" ] +} + +# The live leak (2026-10-01): a main-only pass-through leaves its successor +# cycle running through main's handling turn, and the next park - here a +# restarted session's first turn end - attached to that cycle instead of +# owning it. The successor arm, orphaned by its host's exit, kept owning the +# watcher while the new park's arm polled it until the park boundary, hours +# later. The next park now takes that cycle over: one arm, the host's own +# child, owns the watcher, nothing reaches main for the takeover, no downtime +# episode is opened, and the cycle it owns still delivers the next close. +# A main-only pass-through in leaves its successor cycle running, main +# handles and acknowledges the close, and the session restarts. Sets +# LEFT_WATCHER and LEFT_ARM to the successor watcher and the arm that owns it. +LEFT_WATCHER= +LEFT_ARM= +leave_a_cycle_for_main_and_restart() { # + local home=$1 first_session + start_hook_session "$home" + turn_end "$home" + wait_until 150 watcher_live "$home" || fail "takeover: the Stop hook never started a watcher cycle: $(cat "$home/hook.err" 2>/dev/null)" + append_status "$home" 'which export format?' needs-decision + wait_until 250 hook_exited "$home" || fail "takeover: the decision close never reached the Stop hook: $(cat "$home/state/.supervision-host.log")" + assert_re ' pass-through attended main-only signal:' "$home/state/.supervision-host.log" "fixture: the close was not a main-only pass-through" + assert_rewoke_main "$home" "takeover (pass-through)" + LEFT_WATCHER=$(cat "$home/state/.watch.lock/pid") + LEFT_ARM=$(parent_of "$LEFT_WATCHER") + [ -n "$LEFT_ARM" ] && [ "$LEFT_ARM" != 1 ] || fail "fixture: the successor watcher has no arm of its own" + main_drain "$home" >/dev/null + # shellcheck disable=SC2086 # the printed acknowledgement arguments + [ -z "$MAIN_ACK" ] || FM_HOME="$home" "$FAKE_CLAUDE" -c '"$0" "$@" >/dev/null 2>&1' "$ROOT/bin/fm-wake-drain.sh" $MAIN_ACK \ + || fail "takeover: main's acknowledgement failed: $MAIN_ACK" + # The session restarts: the old one ends, and a new one holds the lock. + first_session=$(tail -n 1 "$home/claude-pids") + : > "$home/session.stop" + wait_until 100 sh -c '! kill -0 "$1" 2>/dev/null' _ "$first_session" || fail "fixture: the first session did not end" + rm -f "$home/session.stop" + kill -0 "$LEFT_ARM" 2>/dev/null || fail "fixture: the successor arm did not outlive its session" + start_hook_session "$home" +} + +test_next_park_takes_over_the_cycle_a_pass_through_left_for_main() { + local home left_watcher left_arm + home=$(make_primary_home hook-takeover) + leave_a_cycle_for_main_and_restart "$home" + left_watcher=$LEFT_WATCHER + left_arm=$LEFT_ARM + turn_end "$home" + wait_until 150 host_owns_the_only_cycle "$home" \ + || fail "takeover: the next park did not own the home's only watcher cycle (left arm $left_arm, watcher $left_watcher):"$'\n'"$(home_arms "$home")"$'\n'"$(cat "$home/state/.supervision-host.log")" + ! kill -0 "$left_arm" 2>/dev/null || fail "takeover: the successor arm a pass-through left still runs (pid $left_arm)" + ! kill -0 "$left_watcher" 2>/dev/null || fail "takeover: the successor watcher still runs (pid $left_watcher)" + sleep 2 + ! hook_exited "$home" || fail "takeover: the takeover woke main: $(cat "$home/hook.err")" + host_owns_the_only_cycle "$home" || fail "takeover: the park did not keep the cycle it took over" + assert_re '^acked:' "$home/state/.watcher-down" "takeover: the takeover opened a downtime episode" + assert_no_re 'rearm-resurface' "$home/state/.supervision-host.log" "takeover: the takeover resurfaced a recovery to main" + append_status "$home" 'which region?' needs-decision + wait_until 250 hook_exited "$home" || fail "takeover: the owned cycle did not deliver the next close: $(cat "$home/state/.supervision-host.log")" + assert_rewoke_main "$home" "takeover (next close)" + assert_re '^signal: .*demo.status' "$home/hook.err" "takeover: the next close must carry the watcher's reason line" + pass "host+hook: the next park takes over the cycle a main-only pass-through left, so one arm owns it" +} + +# A park stopped before its take-over stops the left cycle (here held in the +# take-over's handover snapshot by the recovery-marker lock) must not forget +# that cycle's arm: the park the Stop hook runs next still takes it over rather +# than attaching to it beside the orphan. +test_a_park_stopped_mid_take_over_leaves_the_take_over_to_the_next_park() { + local home holder host + home=$(make_primary_home hook-takeover-interrupted) + leave_a_cycle_for_main_and_restart "$home" + FM_STATE_OVERRIDE="$home/state" bash -c ' + . "$1" + fm_lock_acquire_wait "$2" || exit 1 + : > "$3" + while [ ! -e "$4" ]; do sleep 0.1; done + fm_lock_release "$2" + ' _ "$ROOT/bin/fm-wake-lib.sh" "$home/state/.watcher-down.lock" "$home/marker-lock-held" "$home/marker-lock-release" & + holder=$! + wait_until 100 test -e "$home/marker-lock-held" || fail "fixture: could not hold the recovery-marker lock" + turn_end "$home" + wait_until 150 grep -q " take-over arm=$LEFT_ARM\$" "$home/state/.supervision-host.log" \ + || fail "interrupted takeover: the park did not start a take-over of $LEFT_ARM: $(cat "$home/state/.supervision-host.log")" + host=$(awk -F '\t' '$1 == "host" { print $2; exit }' "$home/state/.supervision-host") + sleep 1 + kill -0 "$LEFT_WATCHER" 2>/dev/null || fail "fixture: the take-over stopped the left watcher before the park was stopped" + kill -TERM "$host" 2>/dev/null || fail "fixture: the park host $host was not running" + wait_until 150 sh -c '! kill -0 "$1" 2>/dev/null' _ "$host" || fail "fixture: the park host did not stop" + : > "$home/marker-lock-release" + wait "$holder" 2>/dev/null || true + # The Stop hook runs the next park in place of the one stopped by a signal. + wait_until 150 host_owns_the_only_cycle "$home" \ + || fail "interrupted takeover: the next park did not own the home's only watcher cycle (left arm $LEFT_ARM):"$'\n'"$(home_arms "$home")"$'\n'"$(cat "$home/state/.supervision-host.log")" + ! kill -0 "$LEFT_ARM" 2>/dev/null || fail "interrupted takeover: the left arm still runs (pid $LEFT_ARM)" + pass "host+hook: a park stopped mid take-over leaves the take-over to the next park" +} + +no_home_arms() { [ -z "$(home_arms "$1")" ]; } + +# A successor the host cannot record for the next park's take-over (here the +# record path is a directory the record would land inside) must not be left +# running: the host stops it on exit, the close still reaches main unchanged, +# and main's next turn end owns a fresh cycle with no orphan beside it. +test_unrecorded_successor_is_stopped_rather_than_left_for_main() { + local home + home=$(make_primary_home hook-successor-unrecorded) + mkdir "$home/state/.supervision-host-left" + start_hook_session "$home" + turn_end "$home" + wait_until 150 watcher_live "$home" || fail "unrecorded successor: the Stop hook never started a watcher cycle: $(cat "$home/hook.err" 2>/dev/null)" + append_status "$home" 'which export format?' needs-decision + wait_until 250 hook_exited "$home" || fail "unrecorded successor: the decision close never reached the Stop hook: $(cat "$home/state/.supervision-host.log")" + assert_re ' pass-through attended main-only signal:' "$home/state/.supervision-host.log" "fixture: the close was not a main-only pass-through" + assert_re ' pass-through successor-unrecorded signal:' "$home/state/.supervision-host.log" "unrecorded successor: the failed record was not logged" + assert_rewoke_main "$home" "unrecorded successor (pass-through)" + assert_re '^signal: .*demo.status' "$home/hook.err" "unrecorded successor: the close must carry the watcher's reason line" + wait_until 100 no_home_arms "$home" || fail "unrecorded successor: an arm outlived the host:"$'\n'"$(home_arms "$home")" + rmdir "$home/state/.supervision-host-left" \ + || fail "unrecorded successor: the record left inside the directory was not removed: $(ls -A "$home/state/.supervision-host-left")" + main_drain "$home" >/dev/null + # shellcheck disable=SC2086 # the printed acknowledgement arguments + [ -z "$MAIN_ACK" ] || FM_HOME="$home" "$FAKE_CLAUDE" -c '"$0" "$@" >/dev/null 2>&1' "$ROOT/bin/fm-wake-drain.sh" $MAIN_ACK \ + || fail "unrecorded successor: main's acknowledgement failed: $MAIN_ACK" + turn_end "$home" + wait_until 150 host_owns_the_only_cycle "$home" \ + || fail "unrecorded successor: main's next turn end did not own the home's only watcher cycle:"$'\n'"$(home_arms "$home")"$'\n'"$(cat "$home/state/.supervision-host.log")" + pass "host+hook: a successor that cannot be recorded is stopped, and main's next turn end arms a fresh cycle" +} + # The captain returns after the loop accepted a decision close away but before # its turn starts: the turn meets the attended rule, so the close still reaches # main exactly as the arm printed it instead of being scoped to nothing. @@ -2655,6 +2806,9 @@ test_claude_stop_hook_runs_the_host_without_the_file_and_off_opts_out test_claude_stop_hook_delivers_a_close_that_turns_main_only_at_its_turn test_claude_stop_hook_notifies_when_at_turn_downtime_write_fails test_successor_close_during_main_turn_is_delivered_at_the_next_turn_end +test_next_park_takes_over_the_cycle_a_pass_through_left_for_main +test_a_park_stopped_mid_take_over_leaves_the_take_over_to_the_next_park +test_unrecorded_successor_is_stopped_rather_than_left_for_main test_primary_without_a_verified_mirror_runs_away_only test_attended_wake_carries_the_dialog_mirror test_dialog_bearing_files_are_owner_only diff --git a/tests/fm-wake-queue.test.sh b/tests/fm-wake-queue.test.sh index f5850586f58..40dacd953c7 100755 --- a/tests/fm-wake-queue.test.sh +++ b/tests/fm-wake-queue.test.sh @@ -1909,6 +1909,57 @@ test_legacy_generationless_wake_is_adopted() { # Pin the recovery acknowledgement contract from docs/watcher-continuity.md at # the queue-library boundary. +# A handover (bin/fm-watch-arm.sh --take-over) undoes only the downtime its own +# watcher stop published over an acknowledged episode. A wake appended between +# the snapshot and the stop, or an episode that was still open, is left for the +# next watcher's arm check to surface. +handover_case() { # + FM_STATE_OVERRIDE="$1" bash -c ' + # shellcheck disable=SC1090,SC1091 + . "$1/bin/fm-wake-lib.sh" + marker="$STATE/.watcher-down" + fm_recovery_marker_publish "$marker" downtime || exit 1 + fm_recovery_marker_read "$marker" || exit 1 + case "$2" in + acked) fm_recovery_marker_ack "$marker" "${FM_RECOVERY_MARKER_TOKEN##*:}" || exit 1 ;; + handling) fm_recovery_marker_begin_handling "$marker" || exit 1 ;; + esac + fm_recovery_marker_read "$marker" || exit 1 + printf "before=%s\n" "$FM_RECOVERY_MARKER_TOKEN" + fm_recovery_marker_handover_snapshot "$marker" || exit 1 + [ "$3" = 0 ] || fm_wake_append signal handover "signal: appended during the handover" || exit 1 + # The stopped watcher closes and publishes downtime, as its EXIT cleanup does. + fm_recovery_marker_publish "$marker" downtime || exit 1 + fm_recovery_marker_handover_restore "$marker" "$FM_RECOVERY_HANDOVER_TOKEN" "$FM_RECOVERY_HANDOVER_SEQ" || exit 1 + fm_recovery_marker_read "$marker" || exit 1 + printf "after=%s\n" "$FM_RECOVERY_MARKER_TOKEN" + ' _ "$ROOT" "$2" "$3" +} + +test_handover_restore_undoes_only_its_own_stop() { + local out before after + out=$(handover_case "$(make_case handover-acked)/state" acked 0) || fail "acked handover case failed: $out" + before=$(printf '%s\n' "$out" | sed -n 's/^before=//p') + after=$(printf '%s\n' "$out" | sed -n 's/^after=//p') + case "$before" in acked:downtime:*) ;; *) fail "fixture: the episode was not acknowledged: $out" ;; esac + [ "$after" = "$before" ] || fail "a handover with nothing queued left a downtime episode: $out" + + out=$(handover_case "$(make_case handover-appended)/state" acked 1) || fail "appended handover case failed: $out" + before=$(printf '%s\n' "$out" | sed -n 's/^before=//p') + after=$(printf '%s\n' "$out" | sed -n 's/^after=//p') + case "$after" in + pending:downtime:*) [ "${after##*:}" != "${before##*:}" ] || fail "fixture: no fresh episode opened: $out" ;; + *) fail "a handover hid a wake appended during it: $out" ;; + esac + + out=$(handover_case "$(make_case handover-handling)/state" handling 0) || fail "handling handover case failed: $out" + before=$(printf '%s\n' "$out" | sed -n 's/^before=//p') + after=$(printf '%s\n' "$out" | sed -n 's/^after=//p') + case "$before" in pending:handling:*) ;; *) fail "fixture: the episode was not being handled: $out" ;; esac + [ "$after" = "pending:downtime:${before##*:}" ] || fail "a handover rewrote an episode main had not acknowledged: $out" + pass "a handover undoes only the downtime its own stop published over an acknowledged episode" +} + test_stale_recovery_generation_cannot_touch_a_newer_episode() { local dir state first_err replay_err sequence generation handling_marker local newer_marker newer_sequence newer_generation rc @@ -3417,6 +3468,7 @@ test_branch_actor_without_eligible_snapshot_refuses test_wake_publish_requires_atomic_recovery_evidence test_recovery_mint_and_delivery_log_avoid_sibling_subst test_legacy_generationless_wake_is_adopted +test_handover_restore_undoes_only_its_own_stop test_stale_recovery_generation_cannot_touch_a_newer_episode test_stale_ack_that_consumes_nothing_names_the_current_wake test_branch_stale_ack_that_consumes_nothing_names_its_granted_wake diff --git a/tests/fm-watch-arm.test.sh b/tests/fm-watch-arm.test.sh index a2c13f946f6..0267782497b 100755 --- a/tests/fm-watch-arm.test.sh +++ b/tests/fm-watch-arm.test.sh @@ -1111,6 +1111,159 @@ test_stop_ends_the_home_watcher_and_publishes_downtime() { pass "watch-arm: --stop ends only this home's watcher, publishes downtime, and reports when none runs" } +# --take-over stops only a watcher that the named arm itself owns. The seed +# watcher here is this shell's child, so naming any other process leaves it +# running and the arm attaches to it exactly as a plain arm does. +test_take_over_attaches_to_a_cycle_the_named_arm_does_not_own() { + local dir state fakebin armout other status + dir=$(make_case take-over-not-owner) + state="$dir/state" + fakebin="$dir/fakebin" + armout="$dir/arm.out" + FM_HOME="$dir" start_seed_watcher "$state" "$fakebin" "$dir/watch.out" + sleep 60 & + other=$! + PATH="$fakebin:$PATH" FM_HOME="$dir" FM_STATE_OVERRIDE="$state" "$WATCH_ARM" --take-over 2>/dev/null + status=$? + expect_code 2 "$status" "--take-over without an arm pid must be refused" + PATH="$fakebin:$PATH" FM_HOME="$dir" FM_STATE_OVERRIDE="$state" FM_ARM_ATTACH_POLL=0.1 \ + "$WATCH_ARM" --take-over "$other" > "$armout" & + ARM_PID=$! + wait_for_file_text "$armout" "watcher: attached pid=$SEED_PID" \ + || fail "--take-over of a cycle the named arm does not own did not attach: $(cat "$armout")" + sleep 1 + is_live_non_zombie "$SEED_PID" || fail "--take-over stopped a watcher the named arm does not own" + [ "$(cat "$state/.watch.lock/pid" 2>/dev/null)" = "$SEED_PID" ] || fail "--take-over moved a lock it does not own" + kill -TERM "$ARM_PID" "$SEED_PID" "$other" 2>/dev/null || true + wait_for_exit "$ARM_PID" 50 >/dev/null 2>&1 || true + wait_for_exit "$SEED_PID" 50 >/dev/null 2>&1 || true + wait "$other" 2>/dev/null || true + pass "watch-arm: --take-over attaches to a cycle the named arm does not own and leaves it running" +} + +# --take-over stops the named real arm's watcher, as it would a successor +# left for main, and owns a fresh cycle. The stop must not open recovery over an episode +# main already acknowledged, and must not hide work still queued. +test_take_over_owns_a_fresh_cycle_and_keeps_queued_work_surfacing() { + local dir state fakebin armout status owner + dir=$(make_case take-over-owner) + state="$dir/state" + fakebin="$dir/fakebin" + armout="$dir/arm.out" + + # Main acknowledged everything: the fresh cycle stays quiet. + start_rearm_arm "$dir" "$state" "$fakebin" "$dir/owner.out" "$$" + owner=$ARM_PID + SEED_PID=$(cat "$state/.watch.lock/pid") + append_wake "$state" signal take-over "signal: fixture handled by main" + ack_wakes "$state" >/dev/null || fail "fixture: main could not acknowledge the handled wake" + case "$(cat "$state/.watcher-down" 2>/dev/null)" in acked:*) ;; *) fail "fixture: the episode was not acknowledged" ;; esac + PATH="$fakebin:$PATH" FM_HOME="$dir" FM_STATE_OVERRIDE="$state" \ + FM_POLL=1 FM_SIGNAL_GRACE=0 FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 \ + "$WATCH_ARM" --take-over "$owner" > "$armout" & + ARM_PID=$! + wait_for_file_text "$armout" 'watcher: started pid=' \ + || fail "--take-over did not own a fresh cycle: $(cat "$armout")" + wait_for_exit "$SEED_PID" 50 >/dev/null 2>&1 || true + ! is_live_non_zombie "$SEED_PID" || fail "--take-over left the watcher it took over running" + [ "$(cat "$state/.watch.lock/pid" 2>/dev/null)" != "$SEED_PID" ] || fail "--take-over did not take the lock" + sleep 3 + is_live_non_zombie "$ARM_PID" || fail "the taken-over cycle closed with no new work: $(cat "$armout")" + case "$(cat "$state/.watcher-down" 2>/dev/null)" in + acked:*) ;; + *) fail "the takeover opened a downtime episode: $(cat "$state/.watcher-down" 2>/dev/null)" ;; + esac + grep -q 'reason=taken-over .*successor=started:' "$state/.watch-cycle-exits.log" \ + || fail "the lifecycle ledger does not link the taken-over cycle to the one it started: $(cat "$state/.watch-cycle-exits.log")" + kill -TERM "$ARM_PID" 2>/dev/null || true + wait_for_exit "$ARM_PID" 50 >/dev/null 2>&1 || true + + # A wake still queued for main resurfaces from the cycle the arm took over. + start_rearm_arm "$dir" "$state" "$fakebin" "$dir/owner2.out" "$$" + owner=$ARM_PID + SEED_PID=$(cat "$state/.watch.lock/pid") + append_wake "$state" signal take-over "signal: fixture still queued for main" + PATH="$fakebin:$PATH" FM_HOME="$dir" FM_STATE_OVERRIDE="$state" \ + FM_POLL=1 FM_SIGNAL_GRACE=0 FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 \ + FM_ARM_CONFIRM_TIMEOUT="$REARM_CONFIRM_SECONDS" "$WATCH_ARM" --take-over "$owner" > "$armout" & + ARM_PID=$! + wait_for_exit "$ARM_PID" "$REARM_EXIT_POLLS" + status=$? + expect_code 0 "$status" "a takeover that resurfaces queued work closes cleanly" + grep -q '^check: rearm-resurface' "$armout" \ + || fail "work queued for main did not resurface after the takeover: $(cat "$armout")" + ! is_live_non_zombie "$SEED_PID" || fail "--take-over left the second watcher running" + pass "watch-arm: --take-over owns a fresh cycle without a recovery wake and still surfaces queued work" +} + +# Pause just after handover releases its snapshot locks, then fail the old +# watcher's secondmate tick write so it exits through cleanup before TERM lands. +# The ledger and recovery wake are public output contracts, not source probes. +test_take_over_preserves_downtime_from_watcher_self_exit() { + local dir home state fakebin owner watcher armout acknowledged real_rm real_touch status i + dir=$(make_case take-over-self-exit) + home="$dir/home" + mkdir -p "$home" + state="$dir/state" + fakebin="$dir/fakebin" + armout="$dir/arm.out" + real_touch=$(command -v touch) + printf '#!/usr/bin/env bash\nif [ "$*" = "%s/.secondmate-liveness-tick" ] && [ -e "%s/fail-tick" ]; then exit 1; fi\nexec "%s" "$@"\n' \ + "$state" "$dir" "$real_touch" > "$fakebin/touch" + chmod +x "$fakebin/touch" + FM_SECONDMATE_LIVENESS_SECS=1 start_rearm_arm "$home" "$state" "$fakebin" "$dir/owner.out" "$$" + owner=$ARM_PID + watcher=$(cat "$state/.watch.lock/pid") + append_wake "$state" signal take-over "signal: fixture handled by main" + ack_wakes "$state" >/dev/null || fail "fixture: could not acknowledge wake" + acknowledged=$(cat "$state/.watcher-down") + real_rm=$(command -v rm) + # Only the taking arm gets this shim. Release the real queue lock before + # exposing the barrier: the self-exiting watcher needs it for cleanup. + mkdir -p "$dir/barrier-bin" + printf '#!/usr/bin/env bash\n"%s" "$@"\n' "$real_rm" > "$dir/barrier-bin/rm" + printf 'if [ "$*" = "-f %s/.wake-queue.lock" ]; then\n' "$state" >> "$dir/barrier-bin/rm" + printf ' touch "%s/snapshot-read"\n for ((i=0; i<700; i++)); do\n [ -e "%s/release" ] && break\n sleep 0.05\n done\nfi\n' "$dir" "$dir" >> "$dir/barrier-bin/rm" + chmod +x "$dir/barrier-bin/rm" + PATH="$dir/barrier-bin:$fakebin:$PATH" FM_HOME="$home" FM_STATE_OVERRIDE="$state" \ + FM_POLL=1 FM_SIGNAL_GRACE=0 FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 \ + FM_ARM_CONFIRM_TIMEOUT="$REARM_CONFIRM_SECONDS" \ + "$WATCH_ARM" --take-over "$owner" > "$armout" & + ARM_PID=$! + i=0 + while [ "$i" -lt 200 ] && [ ! -e "$dir/snapshot-read" ]; do + sleep 0.05 + i=$((i + 1)) + done + [ -e "$dir/snapshot-read" ] || fail "takeover never reached its snapshot" + touch "$dir/fail-tick" + wait_for_exit "$owner" "$REARM_EXIT_POLLS" + status=$? + expect_code 1 "$status" "old watcher must fail through its own cleanup" + ! is_live_non_zombie "$watcher" || fail "old watcher did not self-exit" + grep -q "arm_pid=$owner watcher_pid=$watcher.*exit_code=1 signal=none" "$state/.watch-cycle-exits.log" \ + || fail "owner did not record its watcher's non-signal failure" + [ ! -s "$state/.wake-queue" ] || fail "self-exit unexpectedly queued a wake" + ! grep -q "^$watcher " "$state/.watch-deliveries.log" 2>/dev/null \ + || fail "self-exit unexpectedly delivered a wake" + case "$(cat "$state/.watcher-down")" in pending:downtime:*) ;; *) fail "self-exit did not publish downtime" ;; esac + rm -f "$dir/fail-tick" + touch "$dir/release" + i=0 + while [ "$i" -lt "$REARM_REPORT_POLLS" ]; do + grep -qE '^watcher: started pid=|^check: rearm-resurface' "$armout" && break + is_live_non_zombie "$ARM_PID" || break + sleep 0.05 + i=$((i + 1)) + done + [ "$(cat "$state/.watcher-down")" != "$acknowledged" ] || fail "takeover restored the old acknowledgement" + wait_for_exit "$ARM_PID" "$REARM_EXIT_POLLS" + status=$? + expect_code 0 "$status" "fresh takeover must surface recovery cleanly" + assert_contains "$(cat "$armout")" 'check: rearm-resurface' "self-exit downtime must surface as recovery" + pass "watch-arm: takeover preserves self-exit downtime and surfaces a recovery wake" +} + test_downtime_marker_does_not_follow_symlink() { local dir home state fakebin armout watcher_pid sentinel dir=$(make_case downtime-marker-symlink) @@ -1342,3 +1495,6 @@ test_handling_window_close_keeps_the_acknowledgement_valid test_moved_generation_acknowledgement_is_self_healing test_downtime_marker_does_not_follow_symlink test_stop_ends_the_home_watcher_and_publishes_downtime +test_take_over_attaches_to_a_cycle_the_named_arm_does_not_own +test_take_over_owns_a_fresh_cycle_and_keeps_queued_work_surfacing +test_take_over_preserves_downtime_from_watcher_self_exit From 241d4617d6c160471d7da8ffe637b59f8f4a7af9 Mon Sep 17 00:00:00 2001 From: Tiago Date: Fri, 2 Oct 2026 03:22:32 -0300 Subject: [PATCH 15/33] fix(bin): restore downtime on supervision-host hand-back when the successor already closed (#6355) * fix(bin): restore supervision host hand-back continuity * no-mistakes(review): Scope host hand-back failure fallback to lost pending:handling * no-mistakes(review): Remove stray scratch test copy tests/.tmp-rest.test.sh * no-mistakes(test): Initialise successor globals so early hand-back survives set -u * no-mistakes(document): Document host hand-back downtime failure and Claude lost-handback notice * no-mistakes(ci): I fixed the Greptile finding. The rule that must hold: when the supervision host hands back an actionable wake, its rewake is refused, and no watcher is healthy, the hand-back still has to reach main as a delivered notice. That must be true whether the recovery marker is `pending:handling` or `announced:handling`. Only one place applies this check: the lost hand-back fallback in `bin/fm-claude-stop-autoarm.sh`. **Fix:** that check now accepts both `pending:handling:*` and `announced:handling:*` tokens (a one-line change). Nothing else in the fallback changed: - Refusals on any other marker, such as an already acknowledged one, still exit 0 silently and open no failure episode. - The notice is still sent once per episode, and repeats are recorded as `failed-suppressed`. **Tests:** - `tests/fm-claude-stop-autoarm.test.sh`: the lost hand-back test now runs as a shared helper with two variants, one writing a `pending:handling` marker and a new one writing `announced:handling` (`test_host_lost_announced_handback_notifies_once_per_episode`). - `tests/fm-supervision-host.test.sh`: the end-to-end test where downtime restoration fails is now a shared helper too, with a new `announced` variant (`test_claude_stop_hook_notifies_when_closed_announced_successor_downtime_restore_fails`). It moves the handling episode to `announced` before the host hands back. Without the fix the hook would exit 0 here; the test requires exit 2, `outcome=failed` and a delivered failure notice. **Verification (all under nice -n 10):** - The full `tests/fm-claude-stop-autoarm.test.sh` suite passed (rc=0), including both lost hand-back variants and the benign-refusal test. - In `tests/fm-supervision-host.test.sh` I ran only the four hand-back test functions, all passing (rc=0). The suite can't run single functions, so I used a temporary copy with a trimmed test list and deleted it afterwards; `git status` shows only the 3 intended files changed. - shellcheck is clean on all three changed files. I did not run `bin/fm-lint.sh`. - I did not run the new tests against the unfixed code; the claim that they fail without the fix comes from reading the old check --- bin/fm-claude-stop-autoarm.sh | 16 ++++ bin/fm-supervision-host.sh | 13 ++- docs/supervision-host.md | 5 +- docs/watcher-continuity.md | 2 +- tests/fm-claude-stop-autoarm.test.sh | 71 +++++++++++++++ tests/fm-supervision-host.test.sh | 128 ++++++++++++++++++++++++++- 6 files changed, 227 insertions(+), 8 deletions(-) diff --git a/bin/fm-claude-stop-autoarm.sh b/bin/fm-claude-stop-autoarm.sh index 69785abfc30..f42e0846cb6 100755 --- a/bin/fm-claude-stop-autoarm.sh +++ b/bin/fm-claude-stop-autoarm.sh @@ -537,6 +537,22 @@ if [ "$ACTIONABLE" -eq 1 ]; then [ -z "$OUT" ] || rm -f "$OUT" 2>/dev/null || true exit 2 fi + if [ "$HOST_MODE" -eq 1 ] && fm_autoarm_still_owner "$STATE" "$MY_GEN" \ + && fm_recovery_marker_snapshot "$STATE/.watcher-down" \ + && [[ "$FM_RECOVERY_MARKER_TOKEN" == pending:handling:* || "$FM_RECOVERY_MARKER_TOKEN" == announced:handling:* ]] \ + && ! fm_watcher_healthy "$STATE" "$SCRIPT_DIR/fm-watch.sh" "$GRACE" "$FM_HOME"; then + LOST_HANDBACK_COMMITTED=0 + if [ ! -e "$FAILURE_NOTICE" ]; then + printf 'firstmate watcher auto-arm FAILED - the supervision host returned an actionable wake, but its rewake could not be committed.\n' >&2 + autoarm_commit failed "$FAILURE_NOTICE" && LOST_HANDBACK_COMMITTED=1 + else + autoarm_commit failed-suppressed && LOST_HANDBACK_COMMITTED=1 + fi + if [ "$LOST_HANDBACK_COMMITTED" -eq 1 ]; then + [ -z "$OUT" ] || rm -f "$OUT" 2>/dev/null || true + exit 2 + fi + fi [ -z "$OUT" ] || rm -f "$OUT" 2>/dev/null || true exit 0 fi diff --git a/bin/fm-supervision-host.sh b/bin/fm-supervision-host.sh index 4de9d6c705a..26555b69633 100755 --- a/bin/fm-supervision-host.sh +++ b/bin/fm-supervision-host.sh @@ -261,6 +261,8 @@ HANDLE_RC=0 ENGINE_SUBSHELL= SUCCESSOR_PID= SUCCESSOR_OUT= +SUCCESSOR_WATCHER= +SUCCESSOR_GENERATION= ENGINE_RUNNING=0 # The successor arm a predecessor's pass-through left for main, which the # first cycle takes over. @@ -578,10 +580,17 @@ retire_successor() { # Hand the close to main: stop the successor cycle, print the close, why, and # any further "supervision-host:" lines, and exit. exit_to_main() { # [further lines] + local lines=${2:-} rc=0 retire_successor + if [ -n "$SUCCESSOR_GENERATION" ] \ + && ! fm_recovery_marker_publish "$STATE/.watcher-down" downtime >/dev/null 2>&1; then + log_line "to-main downtime-unrestored $1" + lines=${lines:+$lines$'\n'}"supervision-host: watcher downtime could not be restored for the main hand-back" + rc=1 + fi log_line "to-main $1" - emit "supervision-host: $1" "${2:-}" - exit 0 + emit "supervision-host: $1" "$lines" + exit "$rc" } # The outcome store (bin/fm-branch-outcome.sh) owns and validates these rows. diff --git a/docs/supervision-host.md b/docs/supervision-host.md index 3b6e2fb26ce..1f693b69935 100644 --- a/docs/supervision-host.md +++ b/docs/supervision-host.md @@ -227,7 +227,10 @@ The captain row is still durable, and the next drain presents it until it is ack ## Failure direction Every path that cannot finish a wake the engine took hands that wake to main, with one `supervision-host: ` line after the close. -Before handing it back, the host stops its successor cycle. +Before handing it back, the host stops its successor cycle, and whenever a successor generation was recorded (confirmed or not), it explicitly republishes downtime for that generation. +That publication is required even when the successor already exited, because no watcher cleanup remains to make the close deliverable to the arm owner. +If that publication fails, the hand-back adds a `supervision-host: watcher downtime could not be restored` line and the host exits nonzero. +On Claude, a Stop hook whose rewake is refused while the recovery marker is still `pending:handling` and no watcher is live commits the auto-arm failure notice once per failure episode (`failed-suppressed` after that) and still exits 2, so the hand-back reaches main; every other refused rewake stays silent as before. So the owner's next arm starts from the same state as without the host, and the wake stays durable in the queue. ### Paths that hand the wake back diff --git a/docs/watcher-continuity.md b/docs/watcher-continuity.md index 86f4d4e5e2d..9d9e19dadd5 100644 --- a/docs/watcher-continuity.md +++ b/docs/watcher-continuity.md @@ -121,7 +121,7 @@ The Claude turn-end guard owns that notice commit contract, the monotonic failur On a non-Pi primary, a home that runs the supervision host runs `bin/fm-supervision-host.sh` in place of the arm its re-arm owner would start. The host owns successive watcher cycles through the same arm. -The host's successor and pass-through lifecycle is owned by [supervision-host.md](supervision-host.md#postures); the arm's recovery and acknowledgement contracts below still apply. +[supervision-host.md](supervision-host.md#failure-direction) owns the hand-back's downtime restoration, including when the successor already exited; the arm's recovery and acknowledgement contracts below still apply. ## Actionable wake ordering diff --git a/tests/fm-claude-stop-autoarm.test.sh b/tests/fm-claude-stop-autoarm.test.sh index 7b47c25e854..0cae21c30e9 100755 --- a/tests/fm-claude-stop-autoarm.test.sh +++ b/tests/fm-claude-stop-autoarm.test.sh @@ -1411,6 +1411,24 @@ write_host_fixture() { stood-down) printf "printf 'supervision-host stood down: this session no longer owns supervision\\n'\n" ;; + lost-handback|lost-announced-handback) + local marker=pending + [ "$kind" = lost-handback ] || marker=announced + printf "printf '%s:handling:fixture-generation\\\\n' > \"\$FM_HOME/state/.watcher-down\"\\n" "$marker" + cat <<'SH' +printf 'signal: fixture.status\n' +printf 'supervision-host: branch-outcome: fixture\n' +printf 'supervision-host: watcher downtime could not be restored for the main hand-back\n' +exit 1 +SH + ;; + benign-refusal) + cat <<'SH' +printf 'acked:downtime:fixture-generation\n' > "$FM_HOME/state/.watcher-down" +printf 'signal: fixture.status\n' +printf 'supervision-host: branch-outcome: fixture\n' +SH + ;; handed-back-many) cat <<'SH' printf 'pending:downtime:fixture-generation\n' > "$FM_HOME/state/.watcher-down" @@ -1570,6 +1588,56 @@ test_host_stand_down_is_silent() { pass "auto-arm: a host that stood down closes silently without a retry" } +# Main already drained and acknowledged the wake, so the rewake is refused on a +# marker that is no longer downtime: that refusal stays silent and opens no +# failure episode. +test_host_benign_rewake_refusal_opens_no_failure_episode() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/host-benign-refusal") + mkdir -p "$dir/config" + rm -f "$dir/config/supervision-host-off" + : > "$dir/state/task.meta" + write_host_fixture "$dir" benign-refusal + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + expect_code 0 "$status" "a refused rewake on an acknowledged marker must stay silent" + assert_not_contains "$out" "auto-arm FAILED" "a benign refusal must not deliver a failure notice" + assert_absent "$dir/state/.claude-autoarm-failure-notified" "a benign refusal opened a failure episode" + [ "$(epoch_outcome "$dir")" != failed ] || fail "a benign refusal must not record outcome=failed" + pass "auto-arm: a host rewake refused on an acknowledged marker opens no failure episode" +} + +# The host handed a wake back but left the marker in handling (pending or +# announced) with no live successor, so no rewake can commit: the hook delivers +# the failure notice once per episode and keeps exiting 2 without repeating it. +assert_host_lost_handback_notifies_once_per_episode() { + local kind=$1 dir out status + dir=$(make_primary_dir "$TMP_ROOT/host-$kind") + mkdir -p "$dir/config" + rm -f "$dir/config/supervision-host-off" + : > "$dir/state/task.meta" + write_host_fixture "$dir" "$kind" + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + expect_code 2 "$status" "a lost hand-back must reach main" + assert_contains "$out" "auto-arm FAILED - the supervision host returned an actionable wake" "a lost hand-back must deliver the failure notice" + assert_present "$dir/state/.claude-autoarm-failure-notified" "a lost hand-back did not record its failure episode" + [ "$(epoch_outcome "$dir")" = failed ] || fail "a lost hand-back must record outcome=failed, got: $(epoch_outcome "$dir")" + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + expect_code 2 "$status" "a repeated lost hand-back must still reach main" + assert_not_contains "$out" "auto-arm FAILED" "a repeated lost hand-back must not repeat the failure notice" + [ "$(epoch_outcome "$dir")" = failed-suppressed ] \ + || fail "a repeated lost hand-back must record outcome=failed-suppressed, got: $(epoch_outcome "$dir")" +} + +test_host_lost_handback_notifies_once_per_episode() { + assert_host_lost_handback_notifies_once_per_episode lost-handback + pass "auto-arm: a lost host hand-back notifies once per failure episode" +} + +test_host_lost_announced_handback_notifies_once_per_episode() { + assert_host_lost_handback_notifies_once_per_episode lost-announced-handback + pass "auto-arm: a lost host hand-back on an announced marker notifies once per failure episode" +} + test_host_crash_is_retried_then_reported() { local dir out status dir=$(make_primary_dir "$TMP_ROOT/host-crash") @@ -1692,6 +1760,9 @@ test_host_handback_beside_a_quiet_record_carries_no_away_note test_plain_arm_banner_keeps_its_wake_line_cap test_host_handback_carries_every_host_line test_host_stand_down_is_silent +test_host_benign_rewake_refusal_opens_no_failure_episode +test_host_lost_handback_notifies_once_per_episode +test_host_lost_announced_handback_notifies_once_per_episode test_host_crash_is_retried_then_reported test_arguments_never_arm test_fm_lock_status_still_works_with_shared_lib diff --git a/tests/fm-supervision-host.test.sh b/tests/fm-supervision-host.test.sh index 0f64411baae..cab4ae040c6 100755 --- a/tests/fm-supervision-host.test.sh +++ b/tests/fm-supervision-host.test.sh @@ -50,6 +50,7 @@ FAKE_CLAUDE="$FAKEBIN/claude" # held handle, but first block reading the $FM_HOME/stub-release FIFO # until the test writes to it, so the test chooses when the turn # ends +# captain-held the same, but record a captain outcome before the turn ends # emptyresult the same as handle, but print {} as its result # noreport drain and exit cleanly without a report # go-away the captain goes away (the record is written) mid-turn, then @@ -66,10 +67,11 @@ STATE=${FM_STATE_OVERRIDE:-$FM_HOME/state} mode=$(cat "$FM_HOME/stub-mode" 2>/dev/null || echo handle) n=$(( $(ls "$FM_HOME"/engine-call.* 2>/dev/null | wc -l) + 1 )) { - printf 'actor=%s\nholder=%s\nprimary=%s\nturn=%s\n' "${FM_SUPERVISION_ACTOR:-}" \ + printf 'mode=%s\nactor=%s\nholder=%s\nprimary=%s\nturn=%s\n' "$mode" "${FM_SUPERVISION_ACTOR:-}" \ "${FM_LEASE_HOLDER_PID:-}" "${FM_SUPERVISION_PRIMARY_HARNESS:-}" "${FM_BRANCH_REPORT_TURN:-}" for a in "$@"; do printf 'arg=%s\n' "$a"; done } > "$FM_HOME/engine-call.$n" +case "$mode" in held|captain-held) printf 'ready\n' > "$FM_HOME/stub-ready" ;; esac # Like Claude, the reported cost is the conversation's running total. result() { printf '{"type":"result","subtype":"success","is_error":false,"num_turns":3,"total_cost_usd":%s,' "$(awk -v n="$n" 'BEGIN { print n * 0.25 }')" @@ -90,12 +92,13 @@ verdict=routine [ "$mode" != go-away ] || verdict=captain case "$mode" in fail) exit 3 ;; - handle|captain|held|hold-lease|return|return-silent|return-fail|return-fail-silent|return-many|return-lookup-fail|return-first|noack|emptyresult|go-away) - [ "$mode" != held ] || read -r _ < "$FM_HOME/stub-release" + handle|captain|captain-close-before-return|held|captain-held|hold-lease|return|return-silent|return-fail|return-fail-silent|return-many|return-lookup-fail|return-first|noack|emptyresult|go-away) + case "$mode" in held|captain-held) read -r _ < "$FM_HOME/stub-release" ;; esac [ "$mode" != return-first ] || "$FM_REPO/bin/fm-afk-contract.sh" archive >> "$FM_HOME/engine-return.log" 2>&1 [ "$mode" != go-away ] || "$FM_REPO/bin/fm-afk-contract.sh" enter --words 'gone mid-turn' >> "$FM_HOME/engine-return.log" 2>&1 "$FM_REPO/bin/fm-lease.sh" claim "$task" >> "$FM_HOME/engine-lease.log" 2>&1 - if [ "$mode" = captain ]; then + if [ "$mode" = captain ] || [ "$mode" = captain-held ] \ + || [ "$mode" = captain-close-before-return ]; then "$FM_REPO/bin/fm-branch-report.sh" --task "$task" --verdict captain \ --summary "stub escalated: $(printf '%s\n' "$drain" | grep -v '^WAKE_' | tr '\n' ' ' | cut -c1-400)" \ >> "$FM_HOME/engine-report.log" 2>&1 @@ -121,6 +124,16 @@ case "$mode" in fi # shellcheck disable=SC2086 # the printed acknowledgement arguments [ -z "$ack" ] || [ "$mode" = noack ] || "$FM_REPO/bin/fm-wake-drain.sh" $ack >> "$FM_HOME/engine-ack.log" 2>&1 + case "$mode" in captain-close-before-return) + watcher=$(cat "$STATE/.watch.lock/pid" 2>/dev/null || true) + [ -z "$watcher" ] || kill -TERM "$watcher" 2>/dev/null || true + i=0 + while [ -n "$watcher" ] && kill -0 "$watcher" 2>/dev/null && [ "$i" -lt 100 ]; do + sleep 0.05 + i=$((i + 1)) + done + ;; + esac [ "$mode" = hold-lease ] || "$FM_REPO/bin/fm-lease.sh" release "$task" >> "$FM_HOME/engine-lease.log" 2>&1 case "$mode" in return|return-silent|return-fail|return-fail-silent|return-many|return-lookup-fail) "$FM_REPO/bin/fm-afk-contract.sh" archive >> "$FM_HOME/engine-return.log" 2>&1 ;; @@ -1191,6 +1204,109 @@ test_claude_stop_hook_delivers_a_main_only_pass_through() { pass "host+hook: an attended main-only pass-through rewakes main and keeps its successor watcher" } +# Close the confirmed handling watcher after the engine has acknowledged its +# wake but before its captain outcome returns to the host. +test_claude_stop_hook_restores_handoff_when_successor_closed_before_exit_to_main() { + local home + home=$(make_primary_home hook-successor-closed-before-return) + ln -s "$ROOT/.agents" "$home/.agents" + echo captain-close-before-return > "$home/stub-mode" + start_hook_session "$home" + turn_end "$home" + wait_until 150 watcher_live "$home" || fail "closed successor: the Stop hook never started a watcher cycle" + append_status "$home" 'first actionable wake' + wait_until 250 hook_exited "$home" || fail "closed successor: the Stop hook did not finish: $(cat "$home/state/.supervision-host.log")" + assert_re '^supervision-host: branch-outcome: ' "$home/hook.err" "the host must hand its captain outcome to main" + expect_code 2 "$(cat "$home/hook.rc")" "the Stop hook must rewake main after the successor closed" + assert_re '^(pending|announced):downtime:' "$home/state/.watcher-down" \ + "the closed handling successor must leave a deliverable downtime episode" + pass "host+hook: a successor closed before exit_to_main does not suppress the branch-outcome rewake" +} + +assert_claude_stop_hook_notifies_when_closed_successor_downtime_restore_fails() { + local status=$1 home real_mktemp successor + home=$(make_primary_home "hook-successor-restore-fails-$status") + ln -s "$ROOT/.agents" "$home/.agents" + echo captain-held > "$home/stub-mode" + mkfifo "$home/stub-release" + real_mktemp=$(command -v mktemp) + cat > "$home/fakebin/mktemp" </dev/null' _ "$successor" \ + || fail "restore failure: its watcher did not close during the engine turn" + FM_HOME="$home" bash -c '. "$1"; fm_recovery_marker_begin_handling "$2"' _ \ + "$ROOT/bin/fm-wake-lib.sh" "$home/state/.watcher-down" \ + || fail "fixture: could not model the queued successor wake entering handling" + if [ "$status" = announced ]; then + FM_HOME="$home" bash -c '. "$1"; fm_recovery_marker_read "$2" && _fm_recovery_marker_write_locked "$2" handling "${FM_RECOVERY_MARKER_TOKEN##*:}" announced' _ \ + "$ROOT/bin/fm-wake-lib.sh" "$home/state/.watcher-down" \ + || fail "fixture: could not model the handling episode as announced" + fi + assert_re "^$status:handling:" "$home/state/.watcher-down" \ + "fixture: the closed handling successor must leave the marker in handling before the host hands back" + : > "$home/fail-downtime-write" + printf 'continue\n' > "$home/stub-release" + wait_until 250 hook_exited "$home" || fail "restore failure: the Stop hook did not finish" + expect_code 2 "$(cat "$home/hook.rc")" "the Stop hook must notify main when neither hand-back nor downtime restoration commits" + assert_grep 'firstmate watcher auto-arm FAILED' "$home/hook.err" "the refused rewake must turn into a delivered failure notice" + assert_re '^epoch=[0-9]+ owner_pid=[0-9]+ outcome=failed ' "$home/state/.claude-autoarm-epoch" \ + "the failed hand-back must be committed" +} + +test_claude_stop_hook_notifies_when_closed_successor_downtime_restore_fails() { + assert_claude_stop_hook_notifies_when_closed_successor_downtime_restore_fails pending + pass "host+hook: a refused hand-back becomes a delivered failure notice" +} + +test_claude_stop_hook_notifies_when_closed_announced_successor_downtime_restore_fails() { + assert_claude_stop_hook_notifies_when_closed_successor_downtime_restore_fails announced + pass "host+hook: a refused hand-back on an announced handling marker becomes a delivered failure notice" +} + +test_claude_stop_hook_restores_handoff_when_successor_closed_mid_engine_turn() { + local home successor + home=$(make_primary_home hook-successor-closed-before-outcome) + ln -s "$ROOT/.agents" "$home/.agents" + echo captain-held > "$home/stub-mode" + mkfifo "$home/stub-release" + start_hook_session "$home" + turn_end "$home" + wait_until 150 watcher_live "$home" || fail "closed successor: the Stop hook never started a watcher cycle" + append_status "$home" 'first actionable wake' + wait_until 250 test -s "$home/stub-ready" || fail "closed successor: the engine did not reach its hold: hook=$(cat "$home/hook.err" 2>/dev/null) host=$(cat "$home/state/.supervision-host.log" 2>/dev/null) mode=$(cat "$home/stub-mode" 2>/dev/null) engine=$(find "$home" -maxdepth 1 -name 'engine-call.*' -exec sh -c 'cat "$1"' _ {} \; 2>/dev/null) errors=$(cat "$home"/engine-errors.* 2>/dev/null)" + successor=$(cat "$home/state/.watch.lock/pid") + append_status "$home" 'wake while the engine is handling' + wait_until 250 bash -c '! kill -0 "$1" 2>/dev/null' _ "$successor" \ + || fail "closed successor: its watcher did not close during the engine turn" + FM_HOME="$home" bash -c '. "$1"; fm_recovery_marker_begin_handling "$2"' _ \ + "$ROOT/bin/fm-wake-lib.sh" "$home/state/.watcher-down" \ + || fail "fixture: could not model the queued successor wake entering handling" + assert_re '^pending:handling:' "$home/state/.watcher-down" \ + "fixture: the closed handling successor must leave the marker in handling before the host hands back" + printf 'continue\n' > "$home/stub-release" + wait_until 250 hook_exited "$home" || fail "closed successor: the Stop hook did not finish: $(cat "$home/state/.supervision-host.log")" + assert_re '^supervision-host: branch-outcome: ' "$home/hook.err" "the host must hand its captain outcome to main" + expect_code 2 "$(cat "$home/hook.rc")" "the Stop hook must rewake main after the successor closed" + assert_re '^epoch=[0-9]+ owner_pid=[0-9]+ outcome=rewake ' "$home/state/.claude-autoarm-epoch" \ + "the hand-back must commit the rewake" + assert_re '^(pending|announced):downtime:' "$home/state/.watcher-down" \ + "the closed handling successor must leave a deliverable downtime episode" + pass "host+hook: a successor that closes during a held engine turn does not suppress the branch-outcome rewake" +} + # The live repro (2026-09-28): a quiet record live with no daemon flag parked a # present Claude captain, whose worker's captain outcomes waited for a return. # Through the real Stop hook the outcome now rewakes main, with no away note. @@ -2772,6 +2888,10 @@ test_superseded_host_leaves_the_owner_untouched() { pass "host: a host under a superseded auto-arm generation stands down without touching the owner" } +test_claude_stop_hook_restores_handoff_when_successor_closed_before_exit_to_main +test_claude_stop_hook_restores_handoff_when_successor_closed_mid_engine_turn +test_claude_stop_hook_notifies_when_closed_successor_downtime_restore_fails +test_claude_stop_hook_notifies_when_closed_announced_successor_downtime_restore_fails test_park_exit_probe_uses_half_second_child_sleeps test_report_surface_enforces_actor_turn_and_scope test_report_after_the_return_is_queued_for_main From 65e2aa443a42108689eee260a0d792608ec3540b Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Fri, 2 Oct 2026 00:28:14 -0700 Subject: [PATCH 16/33] fix: reduce remote-job polling process churn (#6363) * perf: cut remote-job idle process creation in the three hot loops Post-update host measurement still attributes most idle churn to three per-sample loops: result-consumer state reads and date calls, the delta reader's capture/hash pass on every poll, and the lane preemption scan's per-field pipelines. This drops each to its minimum without touching the contracts around them. * fm_remote_job_read_state gains an optional result-variable form backed by fm_remote_job_read_line, a builtin-only bounded record read (regular non-symlink file, byte bound, one newline-terminated line, tolerated unterminated tail, no carriage returns). fm_remote_job_wait samples state and the SECONDS clock with no per-sample children; one date call converts the epoch deadline once. * fm-remote-delta-read stats the log each poll and re-runs the bounded capture and hashing only when size, mtime, ctime, inode, or device change. The snapshot's own stat writes the comparison key, so a log that moves between the gate and the capture is never read as stable. * worker_preempting_waiter_exists reads state, home, and the staged argv head with builtins only. The now-unused worker_job_command goes away. The bounded reads use -d '' -n, which behaves identically on the macOS stock bash 3.2 and current bash; -N does not exist on 3.2. Tests cover the malformed-record corpus, delta identity gating, fork-free lane scanning through counting PATH shims, and same-home versus cross-home preemption. No signal traps or sleep contracts change. * no-mistakes(review): Restore subsecond delta keys and byte-bounded builtin record reads * no-mistakes(document): Clarify delta snapshot caching and coarse-timestamp fallback * no-mistakes(lint): Scope UTF-8 regression locales to individual function calls * no-mistakes(ci): Fixed both lint failures by applying the documented production-library analysis boundary at the two affected test imports. Runtime behavior is unchanged; the library remains independently linted. Canonical full-analysis lint passed for the library and both suites, as did bash syntax checks and git diff --check --- bin/fm-remote-delta-read.sh | 131 +++++++++------ bin/fm-remote-job-lib.sh | 59 +++++-- bin/fm-remote-job-worker.sh | 48 ++++-- tests/fm-extension-binding.test.sh | 2 + tests/fm-remote-delta-read.test.sh | 209 ++++++++++++++++++++++++ tests/fm-remote-job-orphan-reap.test.sh | 3 +- tests/fm-remote-job.test.sh | 205 +++++++++++++++++++++++ 7 files changed, 583 insertions(+), 74 deletions(-) create mode 100755 tests/fm-remote-delta-read.test.sh diff --git a/bin/fm-remote-delta-read.sh b/bin/fm-remote-delta-read.sh index 8450628d568..84aef13d051 100755 --- a/bin/fm-remote-delta-read.sh +++ b/bin/fm-remote-delta-read.sh @@ -10,9 +10,14 @@ # the source. A shortened or changed prefix returns a structured continuity-break # result instead of silently rebasing the cursor. # -# An unchanged snapshot is retried after FM_REMOTE_DELTA_POLL_SECONDS (default -# 0.5 seconds). A complete line is visible on the next sample, and the window -# deadline can overshoot by that interval plus snapshot and scheduling work. +# The log is sampled every FM_REMOTE_DELTA_POLL_SECONDS (default 0.5 seconds). +# A complete line is visible on the next sample, and the window deadline can +# overshoot by that interval plus snapshot and scheduling work. +# Each sample of an existing log stats it once. The first sample always runs +# the bounded capture and hashing; later samples skip that work only when the +# size, subsecond mtime and ctime, inode, and device key is unchanged. If either +# timestamp lacks a nonzero subsecond fraction, every sample captures the log +# rather than trusting a coarse key that could hide a same-second rewrite. # The wait remains an ordinary child sleep; signal handling is unchanged. # # Exit 75 means the wait window closed with no complete line. SIGTERM exits the @@ -84,6 +89,30 @@ snapshot_log() { # ) } +delta_subsecond() { # : digits, one dot, and a nonzero fraction + case "$1" in *[!0-9.]* | *.*.*) return 1 ;; esac + case "$1" in [0-9]*.*[1-9]*) ;; *) return 1 ;; esac +} + +# The file identity a snapshot was taken against: GNU and BSD stat spell the +# fields differently, so the poll selects the syntax once by capability. The +# mtime and ctime keep their subsecond fraction; a key without one (a stat or +# filesystem with whole-second timestamps) is discarded, because it cannot tell +# a same-second same-size rewrite apart, and that poll takes a full snapshot. +delta_log_key() { # : sets KEY to "size:mtime:ctime:inode:device" or empty + local rest mtime ctime + if [ "$DELTA_KEY_GNU_STAT" = 1 ]; then + KEY=$(stat -c '%s:%.9Y:%.9Z:%i:%d' "$1" 2>/dev/null) || KEY= + else + KEY=$(stat -f '%z:%Fm:%Fc:%i:%d' "$1" 2>/dev/null) || KEY= + fi + rest=${KEY#*:} + mtime=${rest%%:*} + rest=${rest#*:} + ctime=${rest%%:*} + delta_subsecond "$mtime" && delta_subsecond "$ctime" || KEY= +} + resolve_log() { # local rel=$1 home_real parent_real parent base path case "$rel" in ''|/*|*'//'*) die "log must be a nonempty relative path" ;; esac @@ -134,61 +163,69 @@ trap 'rm -rf -- "$TMP"' EXIT trap 'exit 75' TERM : > "$TMP/empty" EMPTY_HASH=$(sha256_file "$TMP/empty") -START=$(date +%s) +if stat -c '%s' / >/dev/null 2>&1; then DELTA_KEY_GNU_STAT=1; else DELTA_KEY_GNU_STAT=0; fi +START=$SECONDS +LAST_KEY= while :; do if [ -e "$LOG" ] || [ -L "$LOG" ]; then [ -f "$LOG" ] && [ ! -L "$LOG" ] || die "log changed into an unsafe file: $REL" - snapshot_log "$LOG" "$TMP/source" "$TMP/size" \ - || die "log could not be captured safely: $REL" - SIZE=$(tr -d ' ' < "$TMP/size") - if [ "$SIZE" -lt "$OFFSET" ]; then - copy_prefix "$TMP/source" "$SIZE" "$TMP/prefix" - ACTUAL=$(sha256_file "$TMP/prefix") - emit_break truncated "$SIZE" "$ACTUAL" - exit 0 - fi - copy_prefix "$TMP/source" "$OFFSET" "$TMP/prefix" - ACTUAL=$(sha256_file "$TMP/prefix") - if [ "$ACTUAL" != "$PREFIX" ]; then - emit_break prefix-changed "$SIZE" "$ACTUAL" - exit 0 - fi - if [ "$SIZE" -gt "$OFFSET" ]; then - tail -c "+$((OFFSET + 1))" "$TMP/source" | head -c "$MAX_BYTES" > "$TMP/chunk" || true - COMPLETE_BYTES=$(LC_ALL=C od -An -v -tu1 "$TMP/chunk" | awk ' - { for (i = 1; i <= NF; i++) { bytes++; if ($i == 10) complete=bytes } } - END { print complete + 0 } - ') - if [ "$COMPLETE_BYTES" -eq 0 ]; then : > "$TMP/payload"; else head -c "$COMPLETE_BYTES" "$TMP/chunk" > "$TMP/payload"; fi - BYTES=$(LC_ALL=C wc -c < "$TMP/payload" | tr -d ' ') - if [ "$BYTES" -gt 0 ]; then - TO=$((OFFSET + BYTES)) - copy_prefix "$TMP/source" "$TO" "$TMP/to-prefix" - TO_HASH=$(sha256_file "$TMP/to-prefix") - PAYLOAD_HASH=$(sha256_file "$TMP/payload") - printf 'schema=fm-remote-delta.v1\n' - printf 'status=delta\n' - printf 'path=%s\n' "$REL" - printf 'from_offset=%s\n' "$OFFSET" - printf 'to_offset=%s\n' "$TO" - printf 'from_prefix_sha256=%s\n' "$PREFIX" - printf 'to_prefix_sha256=%s\n' "$TO_HASH" - printf 'payload_sha256=%s\n' "$PAYLOAD_HASH" - printf 'payload_bytes=%s\n' "$BYTES" - printf 'reason=\n\n' - cat "$TMP/payload" + delta_log_key "$LOG" + if [ -z "$KEY" ] || [ "$KEY" != "$LAST_KEY" ]; then + snapshot_log "$LOG" "$TMP/source" "$TMP/size" \ + || die "log could not be captured safely: $REL" + # The gate stat precedes the capture, so the snapshot is at least as new + # as its key: a log that moved in between changes the key and is + # captured again on the next poll, never mistaken for stable. + LAST_KEY=$KEY + IFS= read -r SIZE < "$TMP/size" + if [ "$SIZE" -lt "$OFFSET" ]; then + copy_prefix "$TMP/source" "$SIZE" "$TMP/prefix" + ACTUAL=$(sha256_file "$TMP/prefix") + emit_break truncated "$SIZE" "$ACTUAL" exit 0 fi - if [ $((SIZE - OFFSET)) -ge "$MAX_BYTES" ]; then - emit_break line-exceeds-bound "$SIZE" "$ACTUAL" + copy_prefix "$TMP/source" "$OFFSET" "$TMP/prefix" + ACTUAL=$(sha256_file "$TMP/prefix") + if [ "$ACTUAL" != "$PREFIX" ]; then + emit_break prefix-changed "$SIZE" "$ACTUAL" exit 0 fi + if [ "$SIZE" -gt "$OFFSET" ]; then + tail -c "+$((OFFSET + 1))" "$TMP/source" | head -c "$MAX_BYTES" > "$TMP/chunk" || true + COMPLETE_BYTES=$(LC_ALL=C od -An -v -tu1 "$TMP/chunk" | awk ' + { for (i = 1; i <= NF; i++) { bytes++; if ($i == 10) complete=bytes } } + END { print complete + 0 } + ') + if [ "$COMPLETE_BYTES" -eq 0 ]; then : > "$TMP/payload"; else head -c "$COMPLETE_BYTES" "$TMP/chunk" > "$TMP/payload"; fi + BYTES=$(LC_ALL=C wc -c < "$TMP/payload" | tr -d ' ') + if [ "$BYTES" -gt 0 ]; then + TO=$((OFFSET + BYTES)) + copy_prefix "$TMP/source" "$TO" "$TMP/to-prefix" + TO_HASH=$(sha256_file "$TMP/to-prefix") + PAYLOAD_HASH=$(sha256_file "$TMP/payload") + printf 'schema=fm-remote-delta.v1\n' + printf 'status=delta\n' + printf 'path=%s\n' "$REL" + printf 'from_offset=%s\n' "$OFFSET" + printf 'to_offset=%s\n' "$TO" + printf 'from_prefix_sha256=%s\n' "$PREFIX" + printf 'to_prefix_sha256=%s\n' "$TO_HASH" + printf 'payload_sha256=%s\n' "$PAYLOAD_HASH" + printf 'payload_bytes=%s\n' "$BYTES" + printf 'reason=\n\n' + cat "$TMP/payload" + exit 0 + fi + if [ $((SIZE - OFFSET)) -ge "$MAX_BYTES" ]; then + emit_break line-exceeds-bound "$SIZE" "$ACTUAL" + exit 0 + fi + fi fi elif [ "$OFFSET" -ne 0 ] || [ "$PREFIX" != "$EMPTY_HASH" ]; then emit_break missing 0 "$EMPTY_HASH" exit 0 fi - NOW=$(date +%s) - [ $((NOW - START)) -lt "$WAIT" ] || exit 75 + [ $((SECONDS - START)) -lt "$WAIT" ] || exit 75 sleep "$POLL_SECONDS" done diff --git a/bin/fm-remote-job-lib.sh b/bin/fm-remote-job-lib.sh index a60daf41d01..ce6e4090737 100755 --- a/bin/fm-remote-job-lib.sh +++ b/bin/fm-remote-job-lib.sh @@ -499,15 +499,43 @@ fm_remote_job_write_state() { # queued|running|done mv -f -- "$tmp" "$job/state" } -fm_remote_job_read_state() { # - local job=$1 value extra - fm_remote_job_regular_bounded "$job/state" 64 || return 1 - IFS= read -r value < "$job/state" || return 1 - if IFS= read -r extra < <(tail -n +2 "$job/state"); then - : "$extra" - return 1 +# Reads a one-line record bounded to bytes with builtins only, matching +# fm_remote_job_regular_bounded plus the former read/tail checks: a regular +# non-symlink file of at most bytes, one newline-terminated line, a +# tolerated unterminated tail, no carriage returns, and a non-empty value. +# The -d '' -n read treats NUL as the delimiter, so an ordinary +# record (no NULs) is pulled whole at once: the read fails at end of file, +# and success means either bytes landed (the file busts the +# bound) or a NUL stopped it early (already malformed). -N cannot do this: +# the stock /bin/bash on macOS is 3.2, which has -n but no -N. The local +# LC_ALL=C makes -n count bytes rather than multibyte characters, so the byte +# bound holds in a UTF-8 locale. +fm_remote_job_read_line() { # + local file=$1 max=$2 result_var=$3 content + local LC_ALL=C + [ -f "$file" ] && [ ! -L "$file" ] || return 1 + ! IFS= read -r -d '' -n "$((max + 1))" content < "$file" 2>/dev/null || return 1 + case "$content" in *$'\r'* | *$'\n'*$'\n'*) return 1 ;; esac + case "$content" in *$'\n'*) ;; *) return 1 ;; esac + content=${content%%$'\n'*} + [ -n "$content" ] || return 1 + printf -v "$result_var" '%s' "$content" +} + +# Reads the one-word state record with builtins only: the result consumers and +# the lane preemption scan call this once per sample, so it cannot afford the +# bounded-size subshell or a tail process substitution. Passing a result +# variable name avoids the command substitution fork; without one the value is +# printed as before. +fm_remote_job_read_state() { # [result-variable] + local job=$1 result_var=${2:-} read_value + fm_remote_job_read_line "$job/state" 64 read_value || return 1 + case "$read_value" in queued|running|'done') ;; *) return 1 ;; esac + if [ -n "$result_var" ]; then + printf -v "$result_var" '%s' "$read_value" + else + printf '%s\n' "$read_value" fi - case "$value" in queued|running|'done') printf '%s\n' "$value" ;; *) return 1 ;; esac } fm_remote_job_read_number() { # queue_deadline|timeout|deadline|seq @@ -690,7 +718,7 @@ fm_remote_job_stage() { # [args...]; stdi fm_remote_job_wait() { # ; honors FM_REMOTE_JOB_DISCONNECT_PROBE local account_home=$1 id=$2 job state queue_deadline execution_timeout wait_deadline exit_value - local now next_probe=0 + local deadline_ticks next_probe=0 fm_remote_job_prepare_state "$account_home" || return 1 job=$(fm_remote_job_job_dir "$id") || { FM_REMOTE_JOB_ERROR="remote job record disappeared or became unsafe" @@ -709,8 +737,12 @@ fm_remote_job_wait() { # ; honors FM_REMOTE_JOB_DISCONNECT_PR return 1 } wait_deadline=$((queue_deadline + execution_timeout + FM_REMOTE_JOB_WAIT_GRACE)) + # SECONDS is the loop's clock so no time child runs per sample: one date + # read here converts the epoch deadline into the shell's own tick counter + # with the same whole-second granularity. + deadline_ticks=$((SECONDS + wait_deadline - $(date +%s))) while :; do - state=$(fm_remote_job_read_state "$job" 2>/dev/null || true) + fm_remote_job_read_state "$job" state 2>/dev/null || state= case "$state" in 'done') if ! fm_remote_job_regular_bounded "$job/stdout" "$FM_REMOTE_JOB_MAX_BYTES" || @@ -733,13 +765,12 @@ fm_remote_job_wait() { # ; honors FM_REMOTE_JOB_DISCONNECT_PR queued|running) ;; *) FM_REMOTE_JOB_ERROR="remote job state is invalid"; return 1 ;; esac - now=$(date +%s) - if [ "$now" -ge "$wait_deadline" ]; then + if [ "$SECONDS" -ge "$deadline_ticks" ]; then FM_REMOTE_JOB_ERROR="remote job did not complete within its bounded wait" return 1 fi - if [ -n "${FM_REMOTE_JOB_DISCONNECT_PROBE:-}" ] && [ "$now" -ge "$next_probe" ]; then - next_probe=$((now + 1)) + if [ -n "${FM_REMOTE_JOB_DISCONNECT_PROBE:-}" ] && [ "$SECONDS" -ge "$next_probe" ]; then + next_probe=$((SECONDS + 1)) if ! "$FM_REMOTE_JOB_DISCONNECT_PROBE"; then fm_remote_job_cancel "$account_home" "$id" 2>/dev/null || true FM_REMOTE_JOB_ERROR="remote job caller disconnected; the job was cancelled" diff --git a/bin/fm-remote-job-worker.sh b/bin/fm-remote-job-worker.sh index 73191029ff0..118d75c3493 100755 --- a/bin/fm-remote-job-worker.sh +++ b/bin/fm-remote-job-worker.sh @@ -751,25 +751,49 @@ worker_run_with_timeout() { # [args...] return "$rc" } -worker_job_command() { # ; the first argv element of a staged record - local job=$1 first= - fm_remote_job_regular_bounded "$job/argv" "$FM_REMOTE_JOB_MAX_BYTES" || return 1 - IFS= read -r -d '' first < "$job/argv" || [ -n "$first" ] || return 1 - printf '%s\n' "$first" -} - worker_preempting_waiter_exists() { # - local lane_home=$1 job state command job_home + local lane_home=$1 job state command job_home field_terminated remaining chunk + # The argv byte bound counts with read -n and ${#...}, which count bytes only + # in the C locale. + local LC_ALL=C for job in "$FM_REMOTE_JOB_JOBS"/job-*; do [ -d "$job" ] && [ ! -L "$job" ] || continue - state=$(fm_remote_job_read_state "$job" 2>/dev/null || true) + fm_remote_job_read_state "$job" state 2>/dev/null || continue [ "$state" = queued ] || continue fm_remote_job_cancelled "$job" && continue # Lanes are per home, so only a waiter for this lane's own home may - # preempt; another home's queue drains through its own lane. - job_home=$(worker_read_text "$job" home 8192 2>/dev/null || true) + # preempt; another home's queue drains through its own lane. The record + # fields are read with builtins only: this scan runs once a second in + # every lane that executes a preemptible long poll, so no field read may + # spawn a child process. + fm_remote_job_read_line "$job/home" 8192 job_home 2>/dev/null || job_home= [ "$job_home" = "$lane_home" ] || continue - command=$(worker_job_command "$job" 2>/dev/null || true) + # The staged argv record must fit within FM_REMOTE_JOB_MAX_BYTES: bound + # the first NUL-delimited field, then walk the remaining NUL-terminated + # fields and any unterminated tail, still with builtins only. -d '' -n + # is the bounded read on the macOS stock bash (3.2 has -n but no -N); + # never pass -n 0, whose behavior diverges across bash versions. + command= + if [ -f "$job/argv" ] && [ ! -L "$job/argv" ]; then + { field_terminated= + IFS= read -r -d '' -n "$((FM_REMOTE_JOB_MAX_BYTES + 1))" command && field_terminated=1 + if [ -n "$field_terminated" ]; then + if [ "${#command}" -gt "$FM_REMOTE_JOB_MAX_BYTES" ]; then + false + else + remaining=$((FM_REMOTE_JOB_MAX_BYTES - ${#command} - 1)) + chunk= + while [ "$remaining" -ge 0 ] && IFS= read -r -d '' -n "$((remaining + 1))" chunk; do + [ "${#chunk}" -le "$remaining" ] || break + remaining=$((remaining - ${#chunk} - 1)) + done + remaining=$((remaining - ${#chunk})) + [ "$remaining" -ge 0 ] + fi + else + [ -n "$command" ] + fi; } < "$job/argv" 2>/dev/null || command= + fi fm_remote_job_command_preemptible "$command" || return 0 done return 1 diff --git a/tests/fm-extension-binding.test.sh b/tests/fm-extension-binding.test.sh index 22e23f1a645..ab7530a49df 100644 --- a/tests/fm-extension-binding.test.sh +++ b/tests/fm-extension-binding.test.sh @@ -108,6 +108,8 @@ extension_test_cleanup() { ( # worker.pid names the serving child; the copied remote helper stops its # known isolated supervisor tree so it cannot respawn during teardown. + # Production libraries are linted independently by fm-lint.sh. + # shellcheck source=/dev/null . "$REMOTE_ROOT/bin/fm-remote-job-lib.sh" fm_remote_job_stop_worker_tree "$(cat "$TMP_ROOT/remote-jobs/worker.pid")" ) 2>/dev/null || true diff --git a/tests/fm-remote-delta-read.test.sh b/tests/fm-remote-delta-read.test.sh new file mode 100755 index 00000000000..47de313e5ce --- /dev/null +++ b/tests/fm-remote-delta-read.test.sh @@ -0,0 +1,209 @@ +#!/usr/bin/env bash +# Behavior tests for bin/fm-remote-delta-read.sh, the append-only reply-log +# reader a remote lane runs as its preemptible long poll. +# +# Pins, through the executable interface: +# * the delta schema: offsets, prefix and payload hashes, and payload bytes +# * every continuity-break reason: truncated, prefix-changed, missing, and +# line-exceeds-bound, plus an unsafe symlink or traversal target +# * an incomplete tail line is withheld until a newline completes it +# * exit 75 when the wait window closes with nothing appended +# * the per-poll executable boundary: an unchanged log costs one stat per +# sample, and the bounded capture/hashing path runs only when the file's +# stat identity changed - a same-size in-place rewrite still breaks the +# continuity hash, so statting cheaper never hides a change. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +ROOT=$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P) +TMP_ROOT=$(fm_test_tmproot fm-remote-delta-read) +mkdir -p "$TMP_ROOT" +TMP_ROOT=$(cd "$TMP_ROOT" && pwd -P) +DELTA_HOME="$TMP_ROOT/home" +DELTA_LOG_REL=state/replies.status +mkdir -p "$DELTA_HOME/state" +READER="$ROOT/bin/fm-remote-delta-read.sh" + +EMPTY_SHA=$(: | shasum -a 256 | awk '{print $1}') +sha() { printf '%b' "$1" | shasum -a 256 | awk '{print $1}'; } + +run_reader() { # [rel] + FM_HOME="$DELTA_HOME" FM_REMOTE_DELTA_POLL_SECONDS=0.05 \ + "$READER" "${4:-$DELTA_LOG_REL}" "$1" "$2" "$3" +} + +# A growing log returns the complete appended lines with exact boundaries. +: > "$DELTA_HOME/$DELTA_LOG_REL" +run_reader 0 "$EMPTY_SHA" 4 > "$TMP_ROOT/growth.out" & +READER_PID=$! +sleep 0.3 +printf 'first line\n' >> "$DELTA_HOME/$DELTA_LOG_REL" +wait "$READER_PID" || fail "a delta on growth did not exit 0" +OUT=$(<"$TMP_ROOT/growth.out") +assert_contains "$OUT" 'status=delta' 'the grown log did not produce a delta' +assert_contains "$OUT" 'from_offset=0' 'the delta did not start at the caller cursor' +assert_contains "$OUT" 'to_offset=11' 'the delta did not stop at the complete line' +assert_contains "$OUT" "from_prefix_sha256=$EMPTY_SHA" 'the delta did not echo the caller prefix hash' +assert_contains "$OUT" 'payload_sha256='"$(sha 'first line\n')" 'the payload hash is not the appended bytes' +assert_contains "$OUT" 'payload_bytes=11' 'the payload byte count is wrong' +[ "$(tail -n 1 "$TMP_ROOT/growth.out")" = 'first line' ] || fail 'the delta did not carry the appended line' +pass 'an appended line produces a delta with exact offsets, hashes, and payload' + +# An unchanged log closes the window with 75 and never runs the snapshot path: +# one stat per sample is the whole per-poll cost. +DELTA_SHIM="$TMP_ROOT/delta-shim" +EXEC_LOG="$TMP_ROOT/delta-execs" +mkdir -p "$DELTA_SHIM" +for TOOL in perl shasum sha256sum od tail head wc tr date stat dirname basename; do + REAL=$(PATH=/usr/bin:/bin command -v "$TOOL" 2>/dev/null || true) + [ -n "$REAL" ] || continue + cat > "$DELTA_SHIM/$TOOL" <> "\$FM_TEST_EXEC_LOG" +exec $REAL "\$@" +SH + chmod +x "$DELTA_SHIM/$TOOL" +done +: > "$EXEC_LOG" +: > "$DELTA_HOME/$DELTA_LOG_REL" +FM_TEST_EXEC_LOG="$EXEC_LOG" PATH="$DELTA_SHIM:/usr/bin:/bin" run_reader 0 "$EMPTY_SHA" 2 > /dev/null && \ + fail "an unchanged log did not exit 75" || RC=$? +[ "${RC:-0}" -eq 75 ] || fail "an unchanged log closed its window with $RC instead of 75" +perl_execs=$(grep -cx perl "$EXEC_LOG" || true) +stat_execs=$(grep -cx stat "$EXEC_LOG" || true) +# The first poll always takes one snapshot: it must validate the caller's +# cursor prefix before waiting. The gate only suppresses the repeats. +[ "$perl_execs" -eq 1 ] || fail "an unchanged log ran the bounded capture $perl_execs times" +for TOOL in od tail head wc date; do + hits=$(grep -cx "$TOOL" "$EXEC_LOG" || true) + [ "$hits" -eq 0 ] || fail "an unchanged log ran $TOOL $hits times in the poll loop" +done +[ "$stat_execs" -ge 5 ] || fail "the unchanged window did not keep polling stat ($stat_execs)" +pass 'an unchanged log costs one stat per poll and exits 75 at the window' + +# Growth still pays the capture and hashing tools exactly when bytes appear. +: > "$EXEC_LOG" +FM_TEST_EXEC_LOG="$EXEC_LOG" PATH="$DELTA_SHIM:/usr/bin:/bin" run_reader 0 "$EMPTY_SHA" 4 > "$TMP_ROOT/growth2.out" & +READER_PID=$! +sleep 0.3 +printf 'counted change\n' >> "$DELTA_HOME/$DELTA_LOG_REL" +wait "$READER_PID" || fail 'the shimmed growth run did not exit 0' +assert_contains "$(<"$TMP_ROOT/growth2.out")" 'status=delta' 'the shimmed run lost the delta' +[ "$(grep -cx perl "$EXEC_LOG" || true)" -ge 1 ] || fail 'growth did not run the bounded capture' +[ "$(grep -cx shasum "$EXEC_LOG" || true)" -ge 2 ] || fail 'growth did not hash prefix and payload' +pass 'the capture and hashing path runs exactly once a real change lands' + +# A shrunk file reports the truncation with the hash of what actually remains. +printf 'alpha\nbeta\n' > "$DELTA_HOME/$DELTA_LOG_REL" +PREFIX_SHA=$(sha 'alpha\nbeta\n') +run_reader 11 "$PREFIX_SHA" 4 > "$TMP_ROOT/truncated.out" & +READER_PID=$! +sleep 0.3 +printf 'a\n' > "$DELTA_HOME/$DELTA_LOG_REL" +wait "$READER_PID" || fail 'the truncated read did not exit 0' +OUT=$(<"$TMP_ROOT/truncated.out") +assert_contains "$OUT" 'status=continuity-broken' 'truncation did not produce a break' +assert_contains "$OUT" 'reason=truncated' 'truncation was not named' +assert_contains "$OUT" 'to_offset=2' 'the break did not report the shrunk size' +assert_contains "$OUT" "to_prefix_sha256=$(sha 'a\n')" 'the break did not hash the remaining prefix' +pass 'a shrunk log breaks continuity as truncated with the remaining hash' + +# A same-size in-place rewrite changes only mtime/ctime: the stat gate must +# still take the snapshot, where the prefix hash catches the changed bytes. +# This rewrite lands in a later epoch second. +printf 'alpha\nbeta\n' > "$DELTA_HOME/$DELTA_LOG_REL" +run_reader 11 "$PREFIX_SHA" 4 > "$TMP_ROOT/rewrite.out" & +READER_PID=$! +sleep 1.1 +printf 'OMEGA\nbeta\n' > "$DELTA_HOME/$DELTA_LOG_REL" +wait "$READER_PID" || fail 'the rewritten read did not exit 0' +OUT=$(<"$TMP_ROOT/rewrite.out") +assert_contains "$OUT" 'status=continuity-broken' 'a same-size rewrite did not produce a break' +assert_contains "$OUT" 'reason=prefix-changed' 'the same-size rewrite was not named prefix-changed' +assert_contains "$OUT" 'to_offset=11' 'the break did not report the current size' +pass 'a same-size in-place rewrite breaks continuity as prefix-changed' + +# A same-size rewrite of the same inode within the snapshot's own second leaves +# size, inode, device, and whole-second mtime and ctime unchanged: only the +# subsecond stat key can tell it moved. Each attempt starts on a second +# boundary, rewrites once the first capture ran, and is retried only if the +# rewrite still crossed into the next ctime second. +ctime_second() { perl -e 'print +(stat shift)[10]' "$1"; } +SAME_SECOND= +for _ in 1 2 3; do + perl -MTime::HiRes=time,sleep -e 'sleep(1 - (time - int(time)))' + printf 'alpha\nbeta\n' > "$DELTA_HOME/$DELTA_LOG_REL" + BEFORE_SECOND=$(ctime_second "$DELTA_HOME/$DELTA_LOG_REL") + : > "$EXEC_LOG" + FM_TEST_EXEC_LOG="$EXEC_LOG" PATH="$DELTA_SHIM:/usr/bin:/bin" \ + run_reader 11 "$PREFIX_SHA" 2 > "$TMP_ROOT/same-second.out" & + READER_PID=$! + for _ in $(seq 1 50); do grep -qx perl "$EXEC_LOG" && break; sleep 0.01; done + sleep 0.15 + printf 'OMEGA\nbeta\n' > "$DELTA_HOME/$DELTA_LOG_REL" + AFTER_SECOND=$(ctime_second "$DELTA_HOME/$DELTA_LOG_REL") + RC=0 + wait "$READER_PID" || RC=$? + [ "$BEFORE_SECOND" = "$AFTER_SECOND" ] || continue + SAME_SECOND=1 + [ "$RC" -eq 0 ] || fail "the same-second rewrite read exited $RC instead of 0" + OUT=$(<"$TMP_ROOT/same-second.out") + assert_contains "$OUT" 'reason=prefix-changed' 'a same-second same-size rewrite was not detected' + break +done +[ -n "$SAME_SECOND" ] || fail 'no attempt landed the rewrite in the same ctime second' +pass 'a same-second same-size rewrite of the same inode breaks continuity' + +# A log that disappears mid-wait breaks as missing only for a nonzero cursor. +printf 'alpha\nbeta\n' > "$DELTA_HOME/$DELTA_LOG_REL" +run_reader 11 "$PREFIX_SHA" 4 > "$TMP_ROOT/missing.out" & +READER_PID=$! +sleep 0.3 +rm -f -- "$DELTA_HOME/$DELTA_LOG_REL" +wait "$READER_PID" || fail 'the missing-file read did not exit 0' +OUT=$(<"$TMP_ROOT/missing.out") +assert_contains "$OUT" 'status=continuity-broken' 'a removed log did not produce a break' +assert_contains "$OUT" 'reason=missing' 'the removed log was not named missing' +pass 'a removed log breaks continuity as missing' + +# A removed log is not a break for a cursor at the origin: it keeps waiting, +# which is what a first poll against a not-yet-created log relies on. +run_reader 0 "$EMPTY_SHA" 1 > /dev/null && fail 'a missing log at offset 0 did not wait' || RC=$? +[ "${RC:-0}" -eq 75 ] || fail "a missing log at offset 0 exited $RC instead of 75" +pass 'a missing log at the origin cursor keeps waiting until the window closes' + +# An incomplete tail line is withheld until its newline lands, then delivered +# whole rather than as a fragment. +printf 'whole\n' > "$DELTA_HOME/$DELTA_LOG_REL" +run_reader 6 "$(sha 'whole\n')" 4 > "$TMP_ROOT/partial.out" & +READER_PID=$! +sleep 0.3 +printf 'frag' >> "$DELTA_HOME/$DELTA_LOG_REL" +sleep 0.4 +printf -- '-ment\n' >> "$DELTA_HOME/$DELTA_LOG_REL" +wait "$READER_PID" || fail 'the completed line did not exit 0' +OUT=$(<"$TMP_ROOT/partial.out") +assert_contains "$OUT" 'status=delta' 'the completed line did not produce a delta' +assert_contains "$OUT" 'to_offset=16' 'the delta did not stop at the completed line' +assert_contains "$OUT" 'payload_bytes=10' 'the payload did not carry the whole line' +[ "$(tail -n 1 "$TMP_ROOT/partial.out")" = 'frag-ment' ] || fail 'the payload did not join the fragment' +pass 'an unterminated tail is withheld until the newline completes it' + +# The same continuity rules apply to the schema's other break and refusal +# surfaces, with the wait window never entered. +printf 'past-bound tail' > "$DELTA_HOME/$DELTA_LOG_REL" +FM_HOME="$DELTA_HOME" FM_REMOTE_DELTA_MAX_BYTES=8 \ + run_reader 0 "$EMPTY_SHA" 1 > "$TMP_ROOT/bound.out" || fail 'the bound break did not exit 0' +assert_contains "$(<"$TMP_ROOT/bound.out")" 'reason=line-exceeds-bound' \ + 'a tail line longer than the payload bound did not break' +run_reader 0 "$EMPTY_SHA" 1 '../outside' > /dev/null 2>&1 && \ + fail 'a traversing path was accepted' || true +printf 'real\n' > "$DELTA_HOME/state/real.status" +ln -sfn real.status "$DELTA_HOME/state/link.status" +run_reader 0 "$EMPTY_SHA" 1 'state/link.status' > /dev/null 2>&1 && \ + fail 'a symlinked log was accepted' || true +pass 'the reader refuses traversal, symlinks, and oversized tail lines' + +printf 'delta-read contract tests complete\n' diff --git a/tests/fm-remote-job-orphan-reap.test.sh b/tests/fm-remote-job-orphan-reap.test.sh index 667be925aa3..9e0a43a91bc 100755 --- a/tests/fm-remote-job-orphan-reap.test.sh +++ b/tests/fm-remote-job-orphan-reap.test.sh @@ -114,7 +114,8 @@ start_worker() { export FM_REMOTE_JOB_STATE_ROOT="$state_root" export FM_REMOTE_JOB_PLATFORM_OVERRIDE=Linux export FM_REMOTE_JOB_ORPHAN_GRACE_SECONDS=1 - # shellcheck source=bin/fm-remote-job-lib.sh + # Production libraries are linted independently by fm-lint.sh. + # shellcheck source=/dev/null . "$ROOT/bin/fm-remote-job-lib.sh" fm_remote_job_start_linux_worker "$root" "$account_home" >&2 || exit 1 deadline=$(( $(date +%s) + 10 )) diff --git a/tests/fm-remote-job.test.sh b/tests/fm-remote-job.test.sh index a7a5b96873a..b546876d126 100755 --- a/tests/fm-remote-job.test.sh +++ b/tests/fm-remote-job.test.sh @@ -26,6 +26,7 @@ STALL_WORKER_PID= STALL_REPLACEMENT_PID= STALL_JOB_GROUP= QUIET_WORKER_PID= +SCAN_LANE_PID= mkdir -p "$REMOTE_ROOT/bin" "$REMOTE_HOME" "$ACCOUNT_HOME" "$RUNTIME_BIN" # worker.pid records the serving child, not its restart supervisor, so stopping # that pid alone leaves the supervisor to respawn - the leak @@ -38,6 +39,7 @@ cleanup_remote_job_fixture() { [ -z "$LOST_TERM_PID" ] || kill -KILL "$LOST_TERM_PID" 2>/dev/null || true [ -z "$REPLACEMENT_OWNER_PID" ] || kill -KILL "$REPLACEMENT_OWNER_PID" 2>/dev/null || true [ -z "$QUIET_WORKER_PID" ] || kill -KILL "$QUIET_WORKER_PID" 2>/dev/null || true + [ -z "$SCAN_LANE_PID" ] || kill -KILL "$SCAN_LANE_PID" 2>/dev/null || true local stall_pid for stall_pid in "$STALL_WORKER_PID" "$STALL_REPLACEMENT_PID"; do [ -n "$stall_pid" ] || continue @@ -1255,6 +1257,209 @@ quiet_stop "$QUIET_WORKER_PID" QUIET_WORKER_PID= pass "an idle worker still repairs queue permissions and stops promptly on TERM" +# fm_remote_job_read_state is the per-sample read of the result consumers and +# the lane preemption scan, so it is built from builtins and must keep the +# published contract: a regular non-symlink file of at most 64 bytes, one +# newline-terminated line, and a value in the published set. An unterminated +# trailing fragment inside the size bound is still tolerated, matching the +# former tail -n +2 check. +STATE_CORPUS="$TMP_ROOT/state-corpus" +mkdir -p "$STATE_CORPUS/job-x" +state_accepts() { #