From 3a6f4ee2e18b3ac60de57acac4f9baf13eabbbaf Mon Sep 17 00:00:00 2001 From: sdivanl Date: Tue, 22 Sep 2026 12:42:35 +0800 Subject: [PATCH 1/3] fix: surface parked launch prompts as not started --- bin/fm-busy-lib.sh | 156 ++++++++++++- bin/fm-crew-state.sh | 13 +- bin/fm-test-run.sh | 1 + docs/verification/runtime-backends.md | 86 ++++++++ tests/fm-busy-state.test.sh | 143 ++++++++++++ tests/fm-crew-state.test.sh | 35 +++ .../fm-launch-prompt-signals-live-e2e.test.sh | 205 ++++++++++++++++++ 7 files changed, 628 insertions(+), 11 deletions(-) create mode 100644 tests/fm-launch-prompt-signals-live-e2e.test.sh diff --git a/bin/fm-busy-lib.sh b/bin/fm-busy-lib.sh index d8f7a0ee111..01d86c5a60c 100755 --- a/bin/fm-busy-lib.sh +++ b/bin/fm-busy-lib.sh @@ -44,19 +44,48 @@ # Classifier-only sources (never written into a record): # endpoint-gone, herdr-native, grok-regex, rovo-regex, agy-regex, muse-session-log, # cursor-transcript, missing, malformed, gen-mismatch, source-mismatch, -# kimi-unverified, codex-unverified, capture-failed, no-target +# kimi-unverified, codex-unverified, capture-failed, no-target, launch-prompt # # Classification (fm_busy_classify): busy | idle | unknown | dead, always # with the producing source as the second token. Precedence: # 1. dead endpoint (fm_busy_classify_live only) -> dead endpoint-gone # 2. standalone Kimi before verification -> unknown kimi-unverified -# 3. a valid, gen-matching, source-trusted record -> its state and source +# 3. a valid, gen-matching, source-trusted record -> its state and source, +# UNLESS the record is still the untouched seed fm-spawn wrote at arm +# time (state=busy source=fm-spawn - no adapter hook has posted since +# launch) AND the caller supplied a captured tail that matches that +# harness's own recognized interactive-prompt signature (a trust +# dialog, sign-in screen, or first-run menu - fm_busy_launch_prompt_parked +# owns the per-harness table). That combination classifies unknown +# launch-prompt instead: the launch never actually started the brief, so +# it must not read as proof of an active turn. A record that has +# advanced past fm-spawn (any real hook event) is NEVER reclassified +# this way, however its rendered tail looks, so a genuinely working turn +# keeps its ordinary busy verdict and the general BUSY_TURN_MAX_SECS +# bound is unchanged. # 4. no record at all: herdr's native busy verdict is trusted as busy # (generation state is sufficient for busy, not for idle), then the # muse session-log and cursor transcript pull sources, then the # Grok/Rovo/AGY temporary regex fallbacks classify a grok, rovo, or agy # task from its rendered tail, then unknown missing # 5. malformed, stale, or untrusted records -> unknown, never a fallback +# +# fm_busy_launch_prompt_parked (the launch-prompt classifier-only source): a +# launch whose busy record never advanced past the fm-spawn seed is +# indistinguishable, from the record alone, between "still reading its +# brief" and "parked on an interactive prompt the harness never gets past +# without a human" - a Claude/Gemini/Pi workspace-trust dialog, a sign-in or +# auth-method picker, or a first-run setup menu. Left alone this reads as +# ordinary busy for the full BUSY_TURN_MAX_SECS (one hour) before the +# separate wedge-suspect bound even looks at it. The signature table matches +# each harness's own verified rendered dialog text (see +# .agents/skills/harness-adapters/references/harness/*.md and +# docs/verification/*.md for the evidence), scoped to the exact harness that +# renders it so one adapter's ordinary output can never match another's +# dialog. This is a best-effort backstop, not prevention: it never suppresses +# a real busy verdict once any hook has posted, and it defers to whatever +# harness-specific trust pre-registration already exists (fm-claude-trust.sh, +# GEMINI_CLI_TRUST_WORKSPACE) to stop the dialog from appearing at all. # Grok, Rovo, and AGY are the ONLY rendered-text classifications that survive the # redesign, because none of their structured lifecycles was credited-live-verified # in the approved audit (Rovo's clean ACP stopReason lives outside the TUI @@ -867,12 +896,122 @@ fm_busy_agy_tail_busy() { | grep -qiE 'esc[[:space:]]+to[[:space:]]+cancel' } +# --- launch-prompt signatures (fm_busy_launch_prompt_parked) ---------------- +# +# Each function consumes a captured pane tail on stdin (the caller's whole +# tail40, NOT reduced to the last 12 non-blank lines the way the Grok/Rovo/AGY +# busy footers above are): a bordered dialog box renders many short lines of +# pure border/padding (`│ ... │`) that are NOT whitespace-only, so a 12-line +# non-blank reduction was verified live to push the box's own heading text +# (e.g. Gemini's "How would you like to authenticate for this project?") +# outside the window entirely, silently defeating the match. Matching the +# full capture avoids that trap; a signature is still best-effort exactly like +# the footer fallbacks - a screen taller than the capture can still scroll a +# signature out, so absence never proves the pane is NOT parked, only that +# this check cannot confirm it. + +# fm_busy_claude_launch_prompt_tail: Claude's workspace-trust dialog +# ("Quick safety check: Is this a project you created or one you trust?", +# re-verified live on Claude Code 2.1.278, docs/verification/runtime-backends.md +# "Launch-prompt backstop signatures") and its separate external-CLAUDE.md- +# imports dialog ("Allow external CLAUDE.md file imports?", verified by +# disassembly, .agents/skills/harness-adapters/references/harness/claude.md +# "Hook trust" sibling section). fm-claude-trust.sh pre-registers both before +# launch; this is the backstop for when that registration did not take effect. +# Each dialog's own question text is paired with one of its own rendered +# option/footer lines, both required together: the question text alone is +# plausible self-referential prose a firstmate-repo worker could easily render +# on its own (fm-claude-trust.sh's header literally quotes both questions), +# but the option/footer pairing only ever renders inside the real dialog. +fm_busy_claude_launch_prompt_tail() { + local buf + buf=$(cat) + if printf '%s' "$buf" | grep -qiE "${FM_BUSY_CLAUDE_TRUST_PROMPT_REGEX:-Quick safety check: Is this a project you created or one you trust\\?}" \ + && printf '%s' "$buf" | grep -qiE 'No, exit|Enter to confirm'; then + return 0 + fi + printf '%s' "$buf" | grep -qiE "${FM_BUSY_CLAUDE_IMPORTS_PROMPT_REGEX:-Allow external CLAUDE\\.md file imports\\?}" \ + && printf '%s' "$buf" | grep -qiE 'No, disable external imports|Yes, allow external imports' +} + +# fm_busy_pi_launch_prompt_tail: Pi's project-trust dialog. Live-verified on +# pi 0.86.1 (2026-09-22) in a fresh untrusted worktree carrying a project-local +# .pi/extensions/ file (the shape a real ship/scout spawn always launches +# into): the rendered heading is "Trust project folder?" and its declining +# option is literally "Do not trust". An initial guess sourced only from the +# installed binary's UI strings ("Project trust", the internal panel-title +# component name, not this dialog's own rendered heading) was proven wrong by +# that live run and never matched the real screen - which is exactly why this +# class of check must be proven end to end rather than read off strings or a +# name. Matching BOTH the heading and "Do not trust" keeps this from firing on +# a worker's own prose that happens to use the common word "trust" alone. +# Covers omp too: it shares Pi's engine and the same project-trust gate. +fm_busy_pi_launch_prompt_tail() { + local buf + buf=$(cat) + printf '%s' "$buf" | grep -qiE "${FM_BUSY_PI_LAUNCH_PROMPT_REGEX:-Trust project folder\\?}" \ + && printf '%s' "$buf" | grep -qiE 'Do not trust' +} + +# fm_busy_gemini_launch_prompt_tail: Gemini's workspace-trust dialog ("Do you +# trust the files in this folder?"), its first-run auth-method picker ("How +# would you like to authenticate for this project?"), and the credential +# entry it falls through to with no resolvable key ("Enter Gemini API Key"). +# GEMINI_CLI_TRUST_WORKSPACE=true (fm-spawn.sh's launch template) already +# suppresses the first; the other two have no pre-registration and are the +# primary target of this backstop. The trust dialog and the auth-method picker +# were live-verified on gemini 0.60.0 in a credential-less scratch environment +# (docs/verification/runtime-backends.md "Launch-prompt backstop signatures"), +# and each question is paired with one of its own rendered option lines, +# required together, for the same reason as Claude's pairing above: the +# question text alone is plausible prose this very file's own comments could +# render. The auth-method picker's live capture is also what proved the +# full-capture match necessary: its heading renders more than 12 non-blank- +# looking lines above the bordered box's bottom border. The API-key entry +# screen is carried over from .agents/skills/harness-adapters/references/ +# harness/gemini.md "Trust, and why the two documented options are not +# equivalent" rather than this guard's own live capture, and stays a single +# marker: it is reached only after actively selecting that auth method, so +# self-referential prose is a materially smaller risk there. +fm_busy_gemini_launch_prompt_tail() { + local buf + buf=$(cat) + if printf '%s' "$buf" | grep -qiE "${FM_BUSY_GEMINI_TRUST_PROMPT_REGEX:-Do you trust the files in this folder\\?}" \ + && printf '%s' "$buf" | grep -qiE "Trust folder|Don't trust"; then + return 0 + fi + if printf '%s' "$buf" | grep -qiE "${FM_BUSY_GEMINI_AUTH_PROMPT_REGEX:-How would you like to authenticate for this project\\?}" \ + && printf '%s' "$buf" | grep -qiE 'Use Gemini API Key|No authentication method selected'; then + return 0 + fi + printf '%s' "$buf" | grep -qiE "${FM_BUSY_GEMINI_APIKEY_PROMPT_REGEX:-Enter Gemini API Key}" +} + +# fm_busy_launch_prompt_parked: dispatch to the signature above for , +# or fail when this harness has none. Consumes the tail on stdin. Scoped to +# exactly the harnesses fm-spawn.sh arms with the fm-spawn busy source +# (claude*, opencode*, pi, pi-signed, omp, gemini) since only those can ever +# read a pinned "busy fm-spawn" record; codex and standalone Kimi already +# classify unknown before a record is ever consulted, and opencode ships no +# trust dialog at all. +fm_busy_launch_prompt_parked() { # + case "${1:-}" in + claude*) fm_busy_claude_launch_prompt_tail ;; + pi | pi-signed | omp) fm_busy_pi_launch_prompt_tail ;; + gemini) fm_busy_gemini_launch_prompt_tail ;; + *) return 1 ;; + esac +} + # fm_busy_classify: semantic classification for a task whose endpoint the # caller has already established as present. Prints " ": # busy|idle|unknown plus the producing source (see header). Never probes -# process state. is optional pre-captured plain output used only by -# the grok, rovo, and agy arms; when absent each captures through -# fm_backend_capture if available, else reports unknown capture-failed. +# process state. is optional pre-captured plain output: the grok, +# rovo, and agy arms capture it themselves through fm_backend_capture when it +# is absent (or report unknown capture-failed if that is unavailable too), +# while the launch-prompt backstop below has no capture fallback of its own - +# without a supplied tail40 it is skipped entirely and a record still pinned +# at the fm-spawn seed keeps reading busy fm-spawn, unchanged. fm_busy_classify() { # [tail40] local backend=$1 target=$2 harness=$3 id=$4 state=$5 tail40=${6-} local out rc r_state r_source native log @@ -914,7 +1053,12 @@ fm_busy_classify() { # [tail40] out=${out#* } r_source=${out%% *} if fm_busy_source_trusted "$harness" "$r_source"; then - printf '%s %s' "$r_state" "$r_source" + if [ "$r_state" = busy ] && [ "$r_source" = fm-spawn ] && [ -n "$tail40" ] \ + && printf '%s' "$tail40" | fm_busy_launch_prompt_parked "$harness"; then + printf 'unknown launch-prompt' + else + printf '%s %s' "$r_state" "$r_source" + fi else printf 'unknown source-mismatch' fi diff --git a/bin/fm-crew-state.sh b/bin/fm-crew-state.sh index 86239e8b95b..58607b3dcd4 100755 --- a/bin/fm-crew-state.sh +++ b/bin/fm-crew-state.sh @@ -299,12 +299,15 @@ pane_readable() { # # isolated rendered-tail fallback; a herdr crew's native `busy` is accepted # when no record exists, but its native `idle` is NOT, because agent.get # reports generation state (idle while a crew blocks on its own long-running -# foreground tool call) rather than turn state. +# foreground tool call) rather than turn state. The tail is captured +# unconditionally (not just for Grok) so this authoritative read also sees +# fm_busy_lib's launch-prompt backstop: without it, a launch parked on a +# recognized interactive prompt would report `working` here while the +# watcher's own poll (which always captures a tail) already classifies it +# unknown - the exact split issue #1792 describes for a different cause. crew_busy_verdict() { # - local tail40='' - case "$HARNESS" in - grok*) tail40=$(fm_backend_capture "$TASK_BACKEND" "$1" 40 "$EXPECTED_LABEL" 2>/dev/null) || tail40='' ;; - esac + local tail40 + tail40=$(fm_backend_capture "$TASK_BACKEND" "$1" 40 "$EXPECTED_LABEL" 2>/dev/null) || tail40='' fm_busy_classify "$TASK_BACKEND" "$1" "$HARNESS" "$ID" "$STATE" "$tail40" } diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 44e8da93bcc..73e47d095a0 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -351,6 +351,7 @@ family_for_basename() { fm-grok-stop-live-e2e.test.sh|fm-harness-adapter-instructions-live-e2e.test.sh|\ fm-harness-liveness-drift-live-e2e.test.sh|\ fm-muse-signals-live-e2e.test.sh|fm-rovo-signals-live-e2e.test.sh|fm-agy-signals-live-e2e.test.sh|\ + fm-launch-prompt-signals-live-e2e.test.sh|\ fm-herdr-version-floor-live-e2e.test.sh|\ fm-herdr-pi-stale-registration-live-e2e.test.sh|\ fm-opencode-primary-live-e2e.test.sh|fm-pi-branch-live-e2e.test.sh|\ diff --git a/docs/verification/runtime-backends.md b/docs/verification/runtime-backends.md index cb152c4483a..4b260b67733 100644 --- a/docs/verification/runtime-backends.md +++ b/docs/verification/runtime-backends.md @@ -508,6 +508,92 @@ The lab home was deleted and the test entry was removed from the store and verif That automated spawn case runs against a fake claude, so it asserts the store entry and the launch command and nothing more; the live arms above are what establish that the entry actually suppresses the dialog. The composer-classification record below observes the same gate from the other side, where an untrusted worktree left Claude, Grok, and Muse unverified because the guard reads a first-launch trust dialog as an unreadable composer. +## Launch-prompt backstop signatures + +`bin/fm-busy-lib.sh`'s launch-prompt backstop (`fm_busy_launch_prompt_parked`) reclassifies a launch whose busy record is still pinned at the fm-spawn seed as `unknown launch-prompt`, rather than `busy fm-spawn`, when the captured pane matches that harness's own recognized trust, sign-in, or first-run dialog. +Each signature below was live-verified against the real installed binary through `tests/fm-launch-prompt-signals-live-e2e.test.sh` (`FM_LAUNCH_PROMPT_SIGNALS_LIVE=1`), which is what refreshes this record after an upgrade. + +An initial Pi signature sourced only from the installed binary's own UI strings ("Project trust", the internal panel-title component, never the dialog's own rendered heading) was wrong and never matched the real screen. +This guard's first live run caught that before it shipped, which is the evidence for why this class of check must be driven end to end rather than read off strings or a component name. + +Verified 2026-09-22 on Claude Code 2.1.278, pi 0.86.1, and gemini 0.60.0. + +```sh +FM_LAUNCH_PROMPT_SIGNALS_LIVE=1 bash tests/fm-launch-prompt-signals-live-e2e.test.sh +``` + +``` +# live claude version: 2.1.278 (Claude Code) +ok - claude: a real launch parked on its own rendered trust dialog surfaces through the watcher gate +# live pi version: 0.86.1 +ok - pi, pi-signed, omp: a real Pi-engine launch parked on its own rendered trust dialog surfaces through the watcher gate +# live gemini version: 0.60.0 +ok - gemini: a real launch parked on its own rendered auth or trust dialog surfaces through the watcher gate +# checked 3 launch-prompt signature(s) against real installed binaries +``` + +Claude, launched `--dangerously-skip-permissions` into a brand-new worktree under the operator's own already-onboarded config (the shape a real crewmate spawn produces): + +``` + Accessing workspace: + + /tmp/fm-launch-prompt-claude.XXXXXX/wt + + Quick safety check: Is this a project you created or one you trust? (Like your own code, a well-known open source project, or work from your team). If not, take a moment to review what's in this + folder first. + + Claude Code'll be able to read, edit, and execute files here. + + Security guide + + ❯ No, exit + Yes, I trust this folder + + Enter to confirm · Esc to cancel +``` + +Pi, launched into a fresh worktree carrying a project-local `.pi/extensions/` file (the trust-requiring resource that actually gates the dialog) under an isolated `HOME`: + +``` + Trust project folder? + /tmp/fm-launch-prompt-pi.XXXXXX/wt + + This allows pi to load .pi settings and resources, install missing project packages, and execute project extensions. + + → Trust + Trust parent folder (/tmp/fm-launch-prompt-pi.XXXXXX) + Trust (this session only) + Do not trust + Do not trust (this session only) + + ↑↓ navigate enter select escape/ctrl+c cancel +``` + +Gemini, launched `GEMINI_CLI_TRUST_WORKSPACE=true gemini -y` with no `GEMINI_API_KEY` and no prior OAuth credential: + +``` + ? Get started + + How would you like to authenticate for this project? + + ● 1. Sign in with Google + 2. Use Gemini API Key + 3. Vertex AI + + No authentication method selected. + + (Use Enter to select) + + Terms of Services and Privacy Notice for Gemini CLI + + https://geminicli.com/docs/resources/tos-privacy/ +``` + +The real pane renders this inside a bordered box, omitted here for readability; that border is exactly what proves the point below. + +That capture demonstrated why each signature function matches the FULL captured tail rather than the Grok/Rovo/AGY busy-footer convention of the last 12 non-blank lines: a bordered dialog box renders many short lines of pure border and padding (`│ ... │`) that are NOT whitespace-only, so the 12-line reduction pushed this exact heading text out of the window and silently defeated the match on the first attempt. +None of these three runs ever answered its dialog (Escape only, never Enter), so no credential store was written to and no model tokens were spent. + ## Codex hook trust Verified 2026-09-16 on codex-cli 0.151.0, macOS arm64, in a fresh linked worktree of this repository. diff --git a/tests/fm-busy-state.test.sh b/tests/fm-busy-state.test.sh index 7dfef208589..77da1bb0b39 100755 --- a/tests/fm-busy-state.test.sh +++ b/tests/fm-busy-state.test.sh @@ -289,6 +289,141 @@ Ctrl+c:cancel' pass "converted adapters never classify busy from rendered footer text" } +# --- launch-prompt backstop (a launch pinned at fm-spawn, parked on a +# recognized interactive prompt, must classify unknown rather than busy) ------ + +test_launch_prompt_claude_trust_dialog() { + local state out + state=$(new_state_dir launch-prompt-claude) + "$EV" arm "$state" t1 >/dev/null + out=$(fm_busy_classify tmux w1 claude t1 "$state" 'Accessing workspace: /tmp/wt-a +Quick safety check: Is this a project you created or one you trust? +Claude Code'"'"'ll be able to read, edit, and execute files here. +> No, exit + Yes, I trust this folder +Enter to confirm . Esc to cancel') + [ "$out" = "unknown launch-prompt" ] \ + || fail "a launch pinned at fm-spawn parked on Claude's trust dialog must classify unknown launch-prompt, got '$out'" + out=$(fm_busy_classify tmux w1 claude t1 "$state" 'Allow external CLAUDE.md file imports? +This project'"'"'s CLAUDE.md imports files outside the current working directory. +> No, disable external imports + Yes, allow external imports') + [ "$out" = "unknown launch-prompt" ] \ + || fail "a launch pinned at fm-spawn parked on Claude's external-imports dialog must classify unknown launch-prompt, got '$out'" + pass "a Claude launch parked on its trust or external-imports dialog classifies unknown launch-prompt" +} + +test_launch_prompt_pi_trust_dialog() { + local state out h + for h in pi pi-signed omp; do + state=$(new_state_dir "launch-prompt-$h") + "$EV" arm "$state" t1 >/dev/null + out=$(fm_busy_classify tmux w1 "$h" t1 "$state" ' Trust project folder? + /tmp/fm-pi-trust-check/wt + + This allows pi to load .pi settings and resources, install missing project packages, and execute project extensions. + + > Trust + Trust parent folder (/tmp/fm-pi-trust-check) + Trust (this session only) + Do not trust + Do not trust (this session only) + + up/down navigate enter select escape/ctrl+c cancel') + [ "$out" = "unknown launch-prompt" ] \ + || fail "a $h launch pinned at fm-spawn parked on the project-trust dialog must classify unknown launch-prompt, got '$out'" + done + pass "a Pi-family launch (pi, pi-signed, omp) parked on the project-trust dialog classifies unknown launch-prompt" +} + +test_launch_prompt_pi_requires_both_markers() { + local state out + state=$(new_state_dir launch-prompt-pi-partial) + "$EV" arm "$state" t1 >/dev/null + # "trust" alone, with neither the dialog heading nor its decline option, must + # not be read as the dialog - it is an ordinary word a worker's own output + # could easily contain. + out=$(fm_busy_classify tmux w1 pi t1 "$state" 'I trust this approach and will proceed.') + [ "$out" = "busy fm-spawn" ] \ + || fail "ordinary prose containing 'trust' must not classify as a parked launch, got '$out'" + pass "the Pi signature requires both the dialog heading and its decline option, not the bare word trust" +} + +test_launch_prompt_gemini_dialogs() { + local state out + state=$(new_state_dir launch-prompt-gemini-trust) + "$EV" arm "$state" t1 >/dev/null + out=$(fm_busy_classify tmux w1 gemini t1 "$state" 'Do you trust the files in this folder? +● 1. Trust folder (worktree) + 2. Trust parent folder (project) + 3. Don'"'"'t trust') + [ "$out" = "unknown launch-prompt" ] \ + || fail "a Gemini launch parked on the workspace-trust dialog must classify unknown launch-prompt, got '$out'" + + state=$(new_state_dir launch-prompt-gemini-auth) + "$EV" arm "$state" t1 >/dev/null + out=$(fm_busy_classify tmux w1 gemini t1 "$state" 'How would you like to authenticate for this project? +● 2. Use Gemini API Key') + [ "$out" = "unknown launch-prompt" ] \ + || fail "a Gemini launch parked on the auth-method picker must classify unknown launch-prompt, got '$out'" + + state=$(new_state_dir launch-prompt-gemini-apikey) + "$EV" arm "$state" t1 >/dev/null + out=$(fm_busy_classify tmux w1 gemini t1 "$state" 'Enter Gemini API Key +> ') + [ "$out" = "unknown launch-prompt" ] \ + || fail "a Gemini launch parked on the API-key entry dialog must classify unknown launch-prompt, got '$out'" + pass "a Gemini launch parked on its trust, auth-picker, or API-key dialog classifies unknown launch-prompt" +} + +test_launch_prompt_never_shortens_a_working_launch() { + local state out + state=$(new_state_dir launch-prompt-working) + "$EV" arm "$state" t1 >/dev/null + # A genuinely working launch (Claude's ordinary busy footer, rendered before + # its own hook has posted a single event yet) must keep the normal busy + # bound rather than being shortened by this backstop. + out=$(fm_busy_classify tmux w1 claude t1 "$state" '• Working (6s • esc to interrupt)') + [ "$out" = "busy fm-spawn" ] \ + || fail "a genuinely busy launch must not be reclassified, got '$out'" + pass "the launch-prompt backstop never reclassifies a genuinely working launch" +} + +test_launch_prompt_scoped_to_armed_harnesses() { + local state out + # opencode ships no trust dialog (fm-busy-lib.sh header), so it has no + # signature at all: even Claude's own dialog text must not reclassify it. + state=$(new_state_dir launch-prompt-opencode) + "$EV" arm "$state" t1 >/dev/null + out=$(fm_busy_classify tmux w1 opencode t1 "$state" \ + 'Quick safety check: Is this a project you created or one you trust?') + [ "$out" = "busy fm-spawn" ] \ + || fail "opencode has no launch-prompt signature and must stay busy fm-spawn, got '$out'" + pass "the launch-prompt backstop is scoped to harnesses with a verified signature" +} + +test_launch_prompt_never_reclassifies_an_advanced_record() { + local state gen out + state=$(new_state_dir launch-prompt-advanced) + gen=$("$EV" arm "$state" t1) + "$EV" apply "$state" t1 busy --gen "$gen" --source claude-hook --event user-prompt-submit + out=$(fm_busy_classify tmux w1 claude t1 "$state" \ + 'Quick safety check: Is this a project you created or one you trust?') + [ "$out" = "busy claude-hook" ] \ + || fail "a record that has advanced past fm-spawn must never be reclassified by pane text, got '$out'" + pass "the launch-prompt backstop only ever touches the untouched fm-spawn seed" +} + +test_launch_prompt_requires_a_captured_tail() { + local state out + state=$(new_state_dir launch-prompt-no-tail) + "$EV" arm "$state" t1 >/dev/null + out=$(fm_busy_classify tmux w1 claude t1 "$state") + [ "$out" = "busy fm-spawn" ] \ + || fail "with no captured tail the record's own state must stand, got '$out'" + pass "the launch-prompt backstop never runs without a captured tail" +} + test_grok_regex_isolated() { local state out state=$(new_state_dir grok-arm) @@ -474,6 +609,14 @@ test_malformed_record_unknown test_record_without_sidecar_unknown test_source_mismatch_cross_adapter test_converted_adapters_ignore_footer_text +test_launch_prompt_claude_trust_dialog +test_launch_prompt_pi_trust_dialog +test_launch_prompt_pi_requires_both_markers +test_launch_prompt_gemini_dialogs +test_launch_prompt_never_shortens_a_working_launch +test_launch_prompt_scoped_to_armed_harnesses +test_launch_prompt_never_reclassifies_an_advanced_record +test_launch_prompt_requires_a_captured_tail test_grok_regex_isolated test_codex_unverified_gate test_kimi_unverified_gate diff --git a/tests/fm-crew-state.test.sh b/tests/fm-crew-state.test.sh index 8ec1ecc19a1..d90cdd22c03 100755 --- a/tests/fm-crew-state.test.sh +++ b/tests/fm-crew-state.test.sh @@ -1897,6 +1897,40 @@ test_no_run_busy_pane() { pass "no run + a busy semantic record reads working, attributed to its source" } +# A launch pinned at the fm-spawn seed (no hook has posted yet) whose pane +# renders a recognized interactive prompt must read unknown, never working - +# this is the load-bearing link the launch-prompt backstop depends on: +# fm-watch.sh's pause_state_class absorbs a stale pane as "provably working" +# whenever THIS script reports `state: working · source: pane`, so if this +# authoritative read still said working, the watcher would silently swallow +# the wake even though bin/fm-busy-lib.sh's own classifier had already flipped +# to unknown launch-prompt. crew_busy_verdict must therefore capture a real +# tail for every harness, not only grok, so the backstop's own tail-based +# check ever runs here at all. +test_no_run_launch_prompt_parked_is_not_working() { + reset_fakes + local d; d=$(new_case launch-prompt) + make_repo_on_branch "$d/wt" fm/feat-lp + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/feat-lp.meta" "window=fm:fm-feat-lp" "worktree=$d/wt" "kind=ship" "harness=claude" + FM_FAKE_AXI_STATUS="" + FM_FAKE_RUNS_LIST="" + FM_FAKE_BUSY=1 + FM_FAKE_BUSY_TEXT='Quick safety check: Is this a project you created or one you trust? ... +> No, exit + Yes, I trust this folder +Enter to confirm . Esc to cancel' + export FM_FAKE_BUSY_TEXT + # arm only, never apply: the launch turn has never advanced past the seed + # fm-spawn.sh writes at spawn time. + "$ROOT/bin/fm-busy-event.sh" arm "$d/state" feat-lp >/dev/null + local out; out=$(run_crew_state "$d" feat-lp) + assert_not_contains "$out" "state: working" "a launch parked on its trust dialog must never read working" + assert_contains "$out" "state: unknown" "a parked launch reads unknown, not busy or idle" + assert_contains "$out" "launch-prompt" "the unknown verdict names the launch-prompt backstop as its source" + pass "a launch parked on a recognized interactive prompt never reads working, closing the absorb path a stale watcher poll depends on" +} + # A converted adapter must NOT read working from rendered footer text: the # redesign removed that dependency, so a pane painting "esc to interrupt" with # no semantic record is unknown, never working and never silently idle. @@ -4875,6 +4909,7 @@ test_terminal_run_without_live_sibling_is_unchanged test_coarse_run_does_not_probe_other_branch_ci_log_for_ready_status test_other_branch_run_ignored test_no_run_busy_pane +test_no_run_launch_prompt_parked_is_not_working test_no_run_footer_text_alone_is_not_working test_no_run_grok_uses_isolated_fallback test_no_run_herdr_unknown_uses_backend_capture diff --git a/tests/fm-launch-prompt-signals-live-e2e.test.sh b/tests/fm-launch-prompt-signals-live-e2e.test.sh new file mode 100644 index 00000000000..65009c14212 --- /dev/null +++ b/tests/fm-launch-prompt-signals-live-e2e.test.sh @@ -0,0 +1,205 @@ +#!/usr/bin/env bash +# Live guard for bin/fm-busy-lib.sh's launch-prompt backstop (live-harness-optin +# family). Per .agents/skills/firstmate-coding-guidelines "Harness-dependent +# checks", a classifier built on vendor-rendered dialog text must be proven +# against the REAL installed harness, because a stub can only confirm the +# assumption already written into the stub - and this guard exists because that +# assumption was wrong once already: an initial Pi signature, sourced only from +# the installed binary's own UI strings ("Project trust", an internal panel +# title never rendered as the dialog's own heading), silently never matched the +# real screen ("Trust project folder?") until this guard's first live run +# caught it. +# +# For each of claude, pi (covering pi-signed and omp, which share Pi's engine +# and trust gate), and gemini that is actually installed, this drives the REAL +# binary in an isolated tmux server into its genuine interactive launch prompt +# (a fresh untrusted worktree carrying a project-local trust-requiring +# resource for claude and pi, a fresh credential-less environment for gemini), +# captures the pane with the exact production shape (bin/fm-backend.sh's +# fm_backend_tmux_capture: `tmux capture-pane -p -S -40`), arms a scratch +# busy-state record exactly as fm-spawn.sh does at launch, and requires +# fm_busy_classify to report `unknown launch-prompt` instead of the record's +# seeded `busy fm-spawn`. No prompt is ever submitted and no dialog is ever +# answered (Escape only, never Enter), so no model tokens are spent and no +# operator credential store is written to. An absent harness binary is +# reported explicitly and skipped rather than silently passing over it; a run +# that checked nothing fails. +# +# Precondition: this machine's default `claude` config must already be past +# first-run onboarding (a subscription or API key already selected, and a +# theme already chosen) - the guard targets a brand-new SCRATCH WORKTREE under +# the operator's own already-onboarded config, exactly the shape a real +# crewmate spawn produces, never a fresh CLAUDE_CONFIG_DIR. An unonboarded +# machine reports that precondition explicitly rather than failing the +# signature. +# +# Run explicitly with FM_LAUNCH_PROMPT_SIGNALS_LIVE=1. Refresh +# docs/verification/runtime-backends.md ("Launch-prompt backstop signatures") +# from this guard's output after any of claude/pi/gemini upgrades. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +REAL_TMUX=$(command -v tmux 2>/dev/null || true) +SOCKET="fm-launch-prompt-$$" +CHECKED=0 +LABS=() + +note() { printf '# %s\n' "$1"; } +pass() { printf 'ok - %s\n' "$1"; } + +cleanup_all() { + [ -z "${REAL_TMUX:-}" ] || "$REAL_TMUX" -L "$SOCKET" kill-server >/dev/null 2>&1 || true + local lab + for lab in "${LABS[@]:-}"; do + [ -z "$lab" ] || rm -rf -- "$lab" + done +} +trap cleanup_all EXIT + +fail() { printf 'not ok - %s\n' "$1" >&2; exit 1; } + +fm_live_gate opt-in FM_LAUNCH_PROMPT_SIGNALS_LIVE tmux + +# shellcheck source=bin/fm-busy-lib.sh +. "$ROOT/bin/fm-busy-lib.sh" +EV="$ROOT/bin/fm-busy-event.sh" + +# watcher_gate_not_busy: exercise the watcher's production absorb predicate on +# the same real pane capture. The custom tmux socket is intentionally not the +# watcher's default socket, so this checks the pure semantic gate with the +# recorded target while the harness itself remains a real live pane. +watcher_gate_not_busy() { # + local lab=$1 state=$2 target=$3 harness=$4 tail=$5 + mkdir -p "$lab/config" + printf 'window=%s\nbackend=tmux\nharness=%s\n' "$target" "$harness" > "$state/t1.meta" + FM_ROOT_OVERRIDE="$ROOT" + FM_HOME="$lab" + FM_STATE_OVERRIDE="$state" + FM_CONFIG_OVERRIDE="$lab/config" + export FM_ROOT_OVERRIDE FM_HOME FM_STATE_OVERRIDE FM_CONFIG_OVERRIDE + # shellcheck source=bin/fm-watch.sh + . "$ROOT/bin/fm-watch.sh" + if window_is_busy "$target" "$tail"; then + fail "$harness: the watcher still treats the real parked prompt as busy" + fi +} + +# check_harness: launch (checked with fm_busy_classify, which may +# differ from the tmux name when several harnesses share one real +# binary) via into a fresh worktree carrying +# (path,content - empty means none), wait up to 15s for to +# render, capture the pane the production way, arm a scratch busy-state +# record, and require the launch-prompt backstop to classify it unknown +# launch-prompt. Never answers the dialog: Escape only, never Enter. +# +# Writes the captured tail to rather than returning it on stdout: +# a caller that needs the tail (the Pi case, which reuses it for pi-signed and +# omp) must NOT wrap this whole function in a command substitution just to +# capture that output, because `fail` calls `exit`, and `exit` inside a +# `$(...)` subshell only ends that subshell - a real failure would be silently +# swallowed there instead of failing the guard. +check_harness() { # + local harness=$1 session=$2 extra_path=$3 extra_content=$4 expect=$5 tail_out=$6 + local target="$session:w" lab state tail out + shift 6 + lab=$(mktemp -d "${TMPDIR:-/tmp}/fm-launch-prompt-$harness.XXXXXX") || fail "$harness: could not create the isolated lab" + LABS+=("$lab") + mkdir -p "$lab/wt" + git -C "$lab/wt" init -q || fail "$harness: could not initialize the isolated worktree" + if [ -n "$extra_path" ]; then + mkdir -p "$lab/wt/$(dirname "$extra_path")" + printf '%s' "$extra_content" > "$lab/wt/$extra_path" + fi + + "$REAL_TMUX" -L "$SOCKET" new-session -d -s "$session" -n w -c "$lab/wt" -- "$@" \ + || fail "$harness: could not launch the real binary" + + tail='' + for _ in $(seq 1 75); do + tail=$("$REAL_TMUX" -L "$SOCKET" capture-pane -p -t "$target" -S -40 2>/dev/null) || true + printf '%s' "$tail" | grep -qiE "$expect" && break + sleep 0.2 + done + if ! printf '%s' "$tail" | grep -qiE "$expect"; then + "$REAL_TMUX" -L "$SOCKET" kill-session -t "$session" >/dev/null 2>&1 || true + fail "$harness: the real launch never rendered its expected prompt ('$expect') within 15s - captured tail: +$tail" + fi + + state="$lab/state" + mkdir -p "$state" + "$EV" arm "$state" t1 >/dev/null || fail "$harness: could not arm the scratch busy-state record" + out=$(fm_busy_classify tmux w1 "$harness" t1 "$state" "$tail") + [ "$out" = "unknown launch-prompt" ] \ + || fail "$harness: real launch parked on its prompt classified '$out', expected 'unknown launch-prompt'" + watcher_gate_not_busy "$lab" "$state" "$target" "$harness" "$tail" + + "$REAL_TMUX" -L "$SOCKET" send-keys -t "$target" Escape >/dev/null 2>&1 || true + "$REAL_TMUX" -L "$SOCKET" kill-session -t "$session" >/dev/null 2>&1 || true + CHECKED=$((CHECKED + 1)) + [ -z "$tail_out" ] || printf '%s' "$tail" > "$tail_out" +} + +CLAUDE_BIN=$(command -v claude 2>/dev/null || true) +if [ -x "${CLAUDE_BIN:-}" ]; then + VERSION_OUT=$("$CLAUDE_BIN" --version 2>&1) || fail "claude --version failed: $VERSION_OUT" + note "live claude version: $VERSION_OUT" + check_harness claude fm-lp-claude-$$ '' '' \ + 'Is this a project you created or one you trust' '' \ + "$CLAUDE_BIN" --dangerously-skip-permissions hello + pass "claude: a real launch parked on its own rendered trust dialog surfaces through the watcher gate" +else + note "claude not installed - launch-prompt signature not checked" +fi + +PI_BIN=$(command -v pi 2>/dev/null || true) +if [ -x "${PI_BIN:-}" ]; then + VERSION_OUT=$("$PI_BIN" --version 2>&1) || fail "pi --version failed: $VERSION_OUT" + note "live pi version: $VERSION_OUT" + # A fresh, isolated HOME is required so pi's own trust store has no prior + # decision for this scratch worktree; a project-local .pi/extensions/ file + # is what actually gates a fresh worktree behind the dialog (pi only asks + # when the directory holds a trust-requiring resource), exactly the shape + # fm-spawn.sh's own pi launch always carries. + PI_HOME_LAB=$(mktemp -d "${TMPDIR:-/tmp}/fm-launch-prompt-pi-home.XXXXXX") || fail "pi: could not create the isolated HOME" + LABS+=("$PI_HOME_LAB") + PI_TAIL_FILE=$(mktemp "${TMPDIR:-/tmp}/fm-launch-prompt-pi-tail.XXXXXX") || fail "pi: could not create the tail capture file" + LABS+=("$PI_TAIL_FILE") + check_harness pi fm-lp-pi-$$ '.pi/extensions/dummy.ts' 'export default {};' \ + 'Trust project folder' "$PI_TAIL_FILE" \ + env HOME="$PI_HOME_LAB" "$PI_BIN" hello + # pi-signed and omp share Pi's engine and the same project-trust gate + # (fm_busy_launch_prompt_parked), so the one real capture also proves them, + # each against its own freshly armed fm-spawn seed record. + for h in pi-signed omp; do + hstate=$(mktemp -d "${TMPDIR:-/tmp}/fm-launch-prompt-$h.XXXXXX") || fail "$h: could not create the isolated state dir" + LABS+=("$hstate") + "$EV" arm "$hstate" t1 >/dev/null || fail "$h: could not arm the scratch busy-state record" + out=$(fm_busy_classify tmux w1 "$h" t1 "$hstate" "$(cat "$PI_TAIL_FILE")") + [ "$out" = "unknown launch-prompt" ] \ + || fail "$h: the same real Pi trust-dialog capture classified '$out', expected 'unknown launch-prompt'" + done + pass "pi, pi-signed, omp: a real Pi-engine launch parked on its own rendered trust dialog surfaces through the watcher gate" +else + note "pi not installed - launch-prompt signature not checked" +fi + +GEMINI_BIN=$(command -v gemini 2>/dev/null || true) +if [ -x "${GEMINI_BIN:-}" ]; then + VERSION_OUT=$("$GEMINI_BIN" --version 2>&1) || fail "gemini --version failed: $VERSION_OUT" + note "live gemini version: $VERSION_OUT" + check_harness gemini fm-lp-gemini-$$ '' '' \ + 'How would you like to authenticate for this project|Do you trust the files in this folder|Enter Gemini API Key' '' \ + env GEMINI_CLI_TRUST_WORKSPACE=true GEMINI_API_KEY= "$GEMINI_BIN" -y hello + pass "gemini: a real launch parked on its own rendered auth or trust dialog surfaces through the watcher gate" +else + note "gemini not installed - launch-prompt signature not checked" +fi + +[ "$CHECKED" -gt 0 ] || fail "no installed harness could be checked; this run verified nothing" +note "checked $CHECKED launch-prompt signature(s) against real installed binaries" +cleanup_all +trap - EXIT From 807bdec8f3af94814cd75f07324d2a7ca54760b7 Mon Sep 17 00:00:00 2001 From: sdivanl Date: Tue, 22 Sep 2026 13:04:50 +0800 Subject: [PATCH 2/3] no-mistakes(document): docs: record launch-prompt busy backstop classification --- docs/architecture.md | 6 +++++- docs/tmux-backend.md | 2 +- 2 files changed, 6 insertions(+), 2 deletions(-) diff --git a/docs/architecture.md b/docs/architecture.md index b01f62dc5d5..4e6c3e6e667 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -219,7 +219,11 @@ Every classification returns a verdict of busy, idle, unknown, or dead together Each converted adapter reports its own turn lifecycle through a machine-readable contract the vendor already exposes, rather than through rendered footer text: Pi and pi-signed through the Firstmate-owned extension's `agent_start` and `agent_settled` confirmed by `ctx.isIdle()`, omp through its extension's `agent_start` and `agent_end` without `willContinue`, OpenCode through its plugin's semantic `session.status`, Claude through owned `UserPromptSubmit`, `Stop`, `StopFailure`, and `SessionEnd` hooks, Muse through its session log, and Cursor through its conversation transcript. Kimi behind Pi inherits Pi's lifecycle. -Codex and standalone Kimi classify unknown behind explicit probes until a semantic source is live-verified for them, and Grok, Rovo, and AGY each keep one clearly isolated rendered-tail fallback that can only ever classify their own task. +Codex and standalone Kimi classify unknown behind explicit probes until a semantic source is live-verified for them, and Grok, Rovo, and AGY each keep one clearly isolated rendered-tail busy fallback that can only ever classify their own task. +The one case where the contract reads rendered text for a converted adapter is the launch-prompt backstop (`fm_busy_launch_prompt_parked` in `bin/fm-busy-lib.sh`): when a record is still the untouched `fm-spawn` seed and the caller supplied a captured pane matching that harness's own recognized interactive launch prompt - a workspace-trust dialog, sign-in screen, or first-run menu - `fm_busy_classify` reports `unknown launch-prompt` instead of `busy fm-spawn`. +That keeps a launch that never began its brief from holding the busy-age exemption for the whole `FM_BUSY_TURN_MAX_SECS` bound and surfaces it through the ordinary not-provably-working path instead. +A record any real hook event has advanced is never reclassified this way however its pane looks, no captured tail means the record's own state stands, and the general busy bound is unchanged. +The per-harness signature table lives in `bin/fm-busy-lib.sh`'s header, and [runtime backend verification](verification/runtime-backends.md#launch-prompt-backstop-signatures) owns the live evidence. Missing, malformed, stale, untrusted, or unverified semantic state is unknown, never idle, and unknown is never promoted to busy either. Ordinary task-state consumers act only on an exact busy verdict, so an unreadable worker surfaces for a closer look instead of being absorbed as still-working or written off as finished. diff --git a/docs/tmux-backend.md b/docs/tmux-backend.md index da4ddb523ee..bd917644e4d 100644 --- a/docs/tmux-backend.md +++ b/docs/tmux-backend.md @@ -80,7 +80,7 @@ A bare shell prompt is `unknown`, so away-mode escalation is never injected into Busy state is not read from rendered text on this backend. A task's busy, idle, unknown, or dead verdict comes from the semantic busy-state contract owned by `bin/fm-busy-lib.sh`; [architecture](architecture.md#busy-state-is-semantic-per-adapter) owns its boundaries. -The one remaining rendered-tail reader is Grok's isolated fallback inside that contract, which can only classify a Grok task. +The isolated rendered-tail busy fallbacks that remain are harness-scoped, so one adapter's output can never classify another's task. The submit acknowledgement and away-mode supervisor-pane busy guard below still consult rendered output, but only to decide whether input can be delivered, never to decide recorded task state. The supervisor guard selects only the detected primary harness's signature rather than a global union of vendor patterns. From 2be0b6c50ad67abfc5964038e20a449d4fcdfd55 Mon Sep 17 00:00:00 2001 From: sdivanl Date: Tue, 22 Sep 2026 18:13:15 +0800 Subject: [PATCH 3/3] no-mistakes(document): docs: align tail40 and rendered-text comments with launch-prompt backstop --- bin/fm-busy-lib.sh | 8 +++++--- bin/fm-watch.sh | 6 ++++-- 2 files changed, 9 insertions(+), 5 deletions(-) diff --git a/bin/fm-busy-lib.sh b/bin/fm-busy-lib.sh index 01d86c5a60c..9644152a7f6 100755 --- a/bin/fm-busy-lib.sh +++ b/bin/fm-busy-lib.sh @@ -86,8 +86,9 @@ # a real busy verdict once any hook has posted, and it defers to whatever # harness-specific trust pre-registration already exists (fm-claude-trust.sh, # GEMINI_CLI_TRUST_WORKSPACE) to stop the dialog from appearing at all. -# Grok, Rovo, and AGY are the ONLY rendered-text classifications that survive the -# redesign, because none of their structured lifecycles was credited-live-verified +# Apart from the launch-prompt backstop above, Grok, Rovo, and AGY are the ONLY +# rendered-text busy fallbacks that survive the redesign, because none of their +# structured lifecycles was credited-live-verified # in the approved audit (Rovo's clean ACP stopReason lives outside the TUI # path firstmate drives, see references/harness/rovo.md; agy 1.2.0 exposes no # hook surface at all, see references/harness/agy.md); each is scoped to @@ -1183,7 +1184,8 @@ fm_busy_classify_live() { # [expe # fm_busy_classify_meta: classify a task from its recorded metadata, so every # consumer resolves backend, target, and harness the same way instead of # re-deriving them. Requires fm-backend.sh to be sourced. is -# optional pre-captured plain output reused by the Grok arm. +# optional pre-captured plain output reused by the contract's rendered-text +# checks: the Grok/Rovo/AGY busy fallbacks and the launch-prompt backstop. fm_busy_classify_meta() { # [tail40] local meta=$1 id=$2 state=$3 tail40=${4-} backend target harness [ -f "$meta" ] || { printf 'unknown missing'; return 0; } diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index 31cf64aa43f..137dc8d9cc8 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -345,8 +345,10 @@ hash_pane() { # verdict returns 0: idle, unknown, and dead all return 1, so a converted # adapter whose semantic state is missing, malformed, stale, or unverified is # treated as not-provably-working and surfaces rather than being absorbed. -# is the same bounded capture already read for hashing and is -# consumed only by the Grok-scoped fallback inside the contract. +# is the same bounded capture already read for hashing and is passed +# into the contract's harness-scoped rendered-text checks: the Grok/Rovo/AGY +# busy fallbacks and the launch-prompt backstop that keeps a launch pinned at +# its fm-spawn seed from reading as provably working. window_is_busy() { # local w=$1 tail40=$2 task meta verdict task=$(window_to_task "$w" "$STATE")