From d3a7b240995143a481d4a363e6d90837e95ee704 Mon Sep 17 00:00:00 2001 From: keenvc Date: Wed, 16 Sep 2026 16:15:10 -0400 Subject: [PATCH 01/35] feat(harness): add the cline crewmate/scout adapter with ClinePass dispatch (#1) * feat(harness): add the cline crewmate/scout adapter with ClinePass dispatch Add Cline CLI 3.0.62 as a verified crewmate/scout harness following the agy pattern: ancestry detection on the native .cline process, a launch-then-send TUI launch, per-task .cline/hooks busy/turn-end wiring under a new cline-hook busy source, Escape interrupt, /exit, and control-plane tables. Wire the ClinePass open-weights pool into the crew-dispatch example and document the known composer-empty gap (placeholder luminance above the shared ghost ceiling) with a tmux live guard as the refresh command. * no-mistakes(review): fix(docs,quota): correct cline resume grouping and cline-pass family id * no-mistakes(document): docs: cover cline in tmux liveness/anchoring list and configuration.md secondmate-refusal note * no-mistakes(lint): {"summary": "lint: no code changes needed, fm-lint.sh passes with shellcheck on PATH"} * chore(gitignore): drop the stray .omc handoff artifact and ignore .omc/ The no-mistakes gate agent's Claude Code oh-my-claudecode plugin writes .omc/handoffs/last-session-end.md into the run worktree at session end, and a later pipeline step committed it into this branch. Remove the committed file and ignore .omc/ so a home-environment handoff artifact can never ride into a PR. --------- Co-authored-by: firstmate-worker --- .agents/skills/harness-adapters/SKILL.md | 7 +- .../references/harness/cline.md | 55 +++++ .gitignore | 2 + AGENTS.md | 2 +- bin/fm-agent-process-lib.sh | 4 + bin/fm-bootstrap.sh | 3 +- bin/fm-busy-lib.sh | 4 + bin/fm-composer-lib.sh | 17 +- bin/fm-control-lib.sh | 41 +++- bin/fm-harness.sh | 19 +- bin/fm-quota-choose.sh | 7 + bin/fm-spawn.sh | 183 +++++++++++++++- bin/fm-test-run.sh | 4 +- docs/agent-control.md | 2 +- docs/architecture.md | 2 +- docs/configuration.md | 4 +- docs/documentation-audiences.json | 8 + docs/examples/crew-dispatch.json | 17 +- docs/tmux-backend.md | 3 +- docs/trace-context.md | 2 +- docs/verification/cline.md | 149 +++++++++++++ tests/fm-cline-harness.test.sh | 207 ++++++++++++++++++ tests/fm-cline-signals-live-e2e.test.sh | 190 ++++++++++++++++ 23 files changed, 898 insertions(+), 34 deletions(-) create mode 100644 .agents/skills/harness-adapters/references/harness/cline.md create mode 100644 docs/verification/cline.md create mode 100755 tests/fm-cline-harness.test.sh create mode 100755 tests/fm-cline-signals-live-e2e.test.sh diff --git a/.agents/skills/harness-adapters/SKILL.md b/.agents/skills/harness-adapters/SKILL.md index ca4f1233245..dbe5092bea5 100644 --- a/.agents/skills/harness-adapters/SKILL.md +++ b/.agents/skills/harness-adapters/SKILL.md @@ -3,7 +3,7 @@ name: harness-adapters description: >- Agent-only reference for firstmate harness operations. Use before spawning or recovering a crewmate or secondmate, handling a trust dialog, sending a harness-specific skill invocation, interrupting or exiting an agent, resuming an exited agent, or verifying a new harness adapter. - Contains verified facts for claude, codex, opencode, pi, pi-signed, grok, kimi, cursor, gemini, muse, rovo, omp, and agy. + Contains verified facts for claude, codex, opencode, pi, pi-signed, grok, kimi, cursor, gemini, muse, rovo, omp, agy, and cline. user-invocable: false metadata: internal: true @@ -35,7 +35,7 @@ For recovery and control, use the exact `harness=` in `state/.meta`; never i Deliver lifecycle actions only through `../../../bin/fm-control.sh interrupt|exit|relaunch`. Never type an interrupt key or exit command through `fm-send`, where routing-marked lifecycle text becomes chat. Trust handling is complete only when inspection proves the target started processing its instructions; delivery success alone is not proof. -Muse, Gemini, and AGY are verified only for crewmate and scout work, never a secondmate or primary. +Muse, Gemini, AGY, and Cline are verified only for crewmate and scout work, never a secondmate or primary. ## Detection @@ -95,7 +95,8 @@ A new tool remains undispatchable until the `verify` plan, its harness entry, ev "muse": "references/harness/muse.md", "rovo": "references/harness/rovo.md", "omp": "references/harness/omp.md", - "agy": "references/harness/agy.md" + "agy": "references/harness/agy.md", + "cline": "references/harness/cline.md" } } ``` diff --git a/.agents/skills/harness-adapters/references/harness/cline.md b/.agents/skills/harness-adapters/references/harness/cline.md new file mode 100644 index 00000000000..67d15350030 --- /dev/null +++ b/.agents/skills/harness-adapters/references/harness/cline.md @@ -0,0 +1,55 @@ +# Cline CLI + +Cline's `cline` TUI, verified end to end on 2026-09-16 with cline 3.0.62 on Linux through the tmux and Herdr backends. +Verified as a CREWMATE and SCOUT adapter only; `../../../../../bin/fm-spawn.sh` refuses a secondmate launch on it because `../../../../../docs/supervision-protocols/` carries no cline wake protocol. +`../../../../../docs/verification/cline.md` owns how every fact below was established and what is still unproven. + +## Operating facts + +| Fact | Value | +|---|---| +| Binary | `cline` from `PATH`, refused if absent. The installed launcher is a Node wrapper that spawns the long-lived agent as a native binary whose live process name is exactly `.cline` (verified: `ps -o comm=` reports `.cline` with `argv[0]` `.cline`). | +| Launch | `cline -i -c --auto-approve true -m / --thinking `, launched BARE and given its brief only after the readiness gate below. `--model` takes the full `/` id (`cline-pass/deepseek-v4-flash`); cline derives the provider from the prefix, so no separate `-P` is passed. | +| Brief delivery | Launch-then-send, the kimi/rovo shape, because cline's one-time "Introducing Cline Desktop" first-run splash consumes the first submitted line; the gate dismisses the splash with Escape, waits for cline's `Auto-approve` status row, then types the absolute brief pointer once and confirms delivery from the recorded `busy cline-hook` state (never by re-driving Enter). | +| Busy state | Workspace hook config files under `.cline/hooks` (source `cline-hook` in `../../../../../bin/fm-busy-lib.sh`): `TaskStart` opens a turn; `TaskComplete`, `TaskCancel`, `TaskError`, and `SessionShutdown` all close it. `../../../../../bin/fm-spawn.sh` arms the busy generation, writes the files before launch, and excludes `.cline/` from git's view. | +| Rendered tail | The in-transcript busy row reads `⠸ Thinking... (esc to cancel)`; when the turn ends the same row is rewritten as `▶ Thinking:` with the token gone. The bottom status row (`⏵⏵ Auto-approve all enabled (Shift+Tab)`) does NOT change between busy and idle, so it is not a signal. | +| Turn end | The `TaskComplete` hook touches `state/.turn-ended` (the watcher NOTIFICATION) in addition to closing the busy record. | +| Exit | `/exit`, one Enter; the process exits. | +| Interrupt | Single `Escape`, which stops the running turn and leaves the composer at its `Ask anything...` placeholder with no prompt repollution, so no clear key follows. | +| Skill | No verified slash-skill form for injected instructions; use natural language. | +| Autonomy | `--auto-approve true` (cline's documented default, passed explicitly) auto-approves every tool call for the run. | +| Marker | None; a live TUI carries no cline-identity variable. Detected by ancestry alone. | +| Resume | `--id ` resumes an existing session and `cline history --json` lists session ids, but no verified pane-resume contract exists; use deterministic relaunch. | +| Model | `-m /`; a value without `/` is refused by cline with `invalid model format. Expected format: modelType/model`. `bin/fm-spawn.sh` passes the id through unchanged. | +| Effort | `--thinking none\|low\|medium\|high\|xhigh`; `max` stays in task metadata under the record-and-omit contract. | +| Composer | A bordered composer with the bare agent glyph `❯` and a muted-truecolor idle placeholder (`Ask anything...`, or the fresh-session `What can I do for you?`). Both placeholders are in `../../../../../bin/fm-composer-lib.sh`'s fleet-wide idle set, but cline renders them at ~135.5 perceived luminance, just above the shared ghost-luma ceiling of 128, so on the styled tmux/herdr captures an idle cline composer classifies `pending`, never `empty`. This is the same known gap `../../verification/rovo.md` documents; cline readiness/delivery therefore lead with the `Auto-approve` status row and the recorded busy hook, and cline steering relies on the shared queued-Enter busy conversion. | + +## Trust, dialogs, and the first-run splash + +Cline was not observed to gate a fresh worktree behind a folder-trust dialog, and no launch flag or trust store was needed; `--auto-approve true` covers tool approval. +The one first-run obstacle is the "Introducing Cline Desktop" splash, which renders only until it is dismissed once per profile and, if present, swallows the first submitted line. +The readiness gate (`cline_wait_for_ready` in `../../../../../bin/fm-spawn.sh`) detects the splash text and sends one Escape before polling for the idle composer, so a fresh-profile worker cannot lose its brief. + +## Credential precondition + +A verified cline worker ran under a signed-in ClinePass subscription with no key export. +Authorize with `cline auth -p cline-pass` (or the positional `cline auth cline-pass`) on a TTY; the credential lands in `~/.cline/data/settings/providers.json`. +The unauthenticated failure mode was not observed, so treat any auth prompt or refusal as a credential blocker under `../../../../../AGENTS.md` section 9, fix the environment, and retire the endpoint rather than typing into it. + +## Detection + +Detected by ancestry alone: `../../../../../bin/fm-harness.sh` matches the anchored process name `.cline` (and, as a backup for the node wrapper, the anchored script-path fragments `/bin/cline` and `@cline/cli`). +No environment marker is promoted: cline publishes none, and it does not clear an inherited `CLAUDECODE`, so `../../../../../bin/fm-spawn.sh` clears the foreign primary markers at the launch boundary for the same reason cursor and muse do. +cline is deliberately absent from the session-lock name vocabulary in `../../../../../bin/fm-session-lock-lib.sh`, where muse, gemini, rovo, and agy are also absent: a crewmate-only adapter must never own a home session lock. + +## Worker busy state and turn end + +`../../../../../bin/fm-spawn.sh` arms the busy generation with a seed record of `idle/fm-spawn` (the bare launch is not a submitted turn), then writes the five `.cline/hooks` files before launch. +`TaskStart` writes `busy cline-hook` when the brief pointer is submitted, so spawn delivery confirmation reads the same recorded state the supervisor later reads; the `(esc to cancel)` rendered token is only a fallback for a pane whose hook has not landed. +Teardown and relaunch retire the hook files through `fm_control_harness_wiring_paths` in `../../../../../bin/fm-control-lib.sh`. + +## Primary integration + +Unsupported and unverified. +`../../../../../docs/supervision-protocols/` carries no cline protocol, no turn-end guard adapter exists for it, and this adapter verified only the crewmate-side launch, busy state, interrupt, and exit. +`references/common/primary-hooks.md`'s unsupported-boundary rule applies: never invent a wake protocol from a similar TUI. diff --git a/.gitignore b/.gitignore index 3eece43c35f..0371e90c640 100644 --- a/.gitignore +++ b/.gitignore @@ -4,6 +4,8 @@ data/ scratchpad* .no-mistakes/ .lavish/ +.omc/ + .fm-secondmate-home .fm-secondmate-parent .DS_Store diff --git a/AGENTS.md b/AGENTS.md index 86809c25398..b8479ac6fda 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -211,7 +211,7 @@ A silent bootstrap section needs no action; for any printed actionable diagnosti ## 4. Harness and runtime dispatch Load `harness-adapters` before every spawn or recovery and before trust handling, skill invocation, interrupt, exit, resume, or adapter verification. -The verified harnesses are `claude`, `codex`, `opencode`, `pi`, `pi-signed`, `grok`, `kimi`, `cursor`, and `omp`, plus `muse`, `gemini`, `rovo`, and `agy` for crewmates and scouts only; never dispatch on an unverified adapter. +The verified harnesses are `claude`, `codex`, `opencode`, `pi`, `pi-signed`, `grok`, `kimi`, `cursor`, and `omp`, plus `muse`, `gemini`, `rovo`, `agy`, and `cline` for crewmates and scouts only; never dispatch on an unverified adapter. If static `config/crew-harness` or `config/secondmate-harness` names an unverified adapter, report it and fall back only to a verified adapter rather than launching it. `docs/configuration.md` owns dispatch-profile and runtime-backend schemas, `bin/fm-harness.sh` owns static resolution, and `bin/fm-spawn.sh` owns launch flags and fail-closed validation. diff --git a/bin/fm-agent-process-lib.sh b/bin/fm-agent-process-lib.sh index dcf4b59ff4e..662308e4732 100644 --- a/bin/fm-agent-process-lib.sh +++ b/bin/fm-agent-process-lib.sh @@ -46,6 +46,10 @@ fm_agent_process_classify_name() { # [argv0] -> agent|shell|other # single binary, comm=agy with argv[0]=agy), and a glob would claim # unrelated commands containing that fragment. agy) printf 'agent' ;; + # cline (Cline CLI) launches its agent as a native binary whose live process + # name is exactly `.cline` (verified, cline 3.0.62). Anchored, never + # *cline*, so an unrelated command cannot be misread as this harness. + .cline|cline) printf 'agent' ;; zsh|bash|sh|dash|ash|ksh|mksh|tcsh|csh|fish) printf 'shell' ;; *) if fm_harness_path_name "$path" >/dev/null || fm_harness_path_name "$argv0" >/dev/null; then diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index 747f2c3a024..7f892a84ba0 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -1114,7 +1114,7 @@ crew_dispatch_validate() { return 0 fi err=$(jq -r ' - def verified($h): ["claude","codex","opencode","pi","pi-signed","grok","kimi","cursor","agy","muse","rovo","omp"] | index($h); + def verified($h): ["claude","codex","opencode","pi","pi-signed","grok","kimi","cursor","agy","muse","rovo","omp","cline"] | index($h); def effort_ok($h; $m; $e): if $e == null then true elif ($e | type) != "string" then false @@ -1123,6 +1123,7 @@ crew_dispatch_validate() { elif $h == "codex" then ((["low","medium","high","xhigh"] | index($e)) != null or ($e == "max" and $m == "gpt-5.6-luna")) elif $h == "grok" then (["low","medium","high"] | index($e)) elif $h == "agy" then (["low","medium","high"] | index($e)) + elif $h == "cline" then (["low","medium","high","xhigh"] | index($e)) elif $h == "pi" or $h == "pi-signed" or $h == "omp" then (["low","medium","high","xhigh","max"] | index($e)) elif $h == "muse" then (["low","medium","high","xhigh","max"] | index($e)) elif $h == "rovo" then (["low","medium","high","max"] | index($e)) diff --git a/bin/fm-busy-lib.sh b/bin/fm-busy-lib.sh index d8f7a0ee111..ce871274fd0 100755 --- a/bin/fm-busy-lib.sh +++ b/bin/fm-busy-lib.sh @@ -34,6 +34,9 @@ # claude-hook Claude lifecycle hooks (UserPromptSubmit/Stop/StopFailure/SessionEnd) # gemini-hook Gemini agent hooks (BeforeAgent opens; AfterAgent and # SessionEnd close) +# cline-hook Cline workspace hook config files (.cline/hooks): +# TaskStart opens a turn; TaskComplete, TaskCancel, +# TaskError, and SessionShutdown all close it # codex-hook, codex-appserver reserved: Codex, gated by # fm_busy_codex_semantic_source # kimi-wire, kimi-hook reserved: standalone Kimi, gated by fm_busy_kimi_verified @@ -198,6 +201,7 @@ fm_busy_sources_for_harness() { # ;; opencode*) adapter=opencode-plugin ;; gemini*) adapter=gemini-hook ;; + cline*) adapter=cline-hook ;; pi|pi-signed) adapter=pi-ext ;; omp) adapter=omp-ext ;; kimi*) diff --git a/bin/fm-composer-lib.sh b/bin/fm-composer-lib.sh index d919b61f53c..dffd9d1ea07 100644 --- a/bin/fm-composer-lib.sh +++ b/bin/fm-composer-lib.sh @@ -370,6 +370,15 @@ FM_DELIVERY_CURSOR_BUSY_REGEX_DEFAULT='ctrl\+c to stop' # acknowledgement. Delivery guard only; recorded worker state comes from the # agy-regex fold in bin/fm-busy-lib.sh. FM_DELIVERY_AGY_BUSY_REGEX_DEFAULT='esc[[:space:]]+to[[:space:]]+cancel' +# cline (Cline CLI) renders an in-transcript busy row while a turn runs: a +# braille spinner, `Thinking...`, and the `(esc to cancel)` token (verified +# live, cline 3.0.62). Its idle composer is the `Ask anything...` placeholder +# and its bottom status row does NOT change between busy and idle, so this +# in-transcript token is the delivery guard's only busy signature. When the +# turn ends the row is rewritten as `Thinking:` with the token gone, so a +# finished turn cannot fake an acknowledgement. Delivery guard only; recorded +# worker state comes from the cline-hook fold in bin/fm-busy-lib.sh. +FM_DELIVERY_CLINE_BUSY_REGEX_DEFAULT='\(esc to cancel\)' FM_DELIVERY_KIMI_BUSY_REGEX_DEFAULT='^[[:space:]]*(🌑|🌒|🌓|🌔|🌕|🌖|🌗|🌘)[[:space:]]+·[[:space:]]+' fm_busy_lines_match() { # [harness] @@ -386,6 +395,7 @@ fm_busy_lines_match() { # [harness] omp) regex=$FM_DELIVERY_OMP_BUSY_REGEX_DEFAULT ;; grok) regex=$FM_DELIVERY_GROK_BUSY_REGEX_DEFAULT ;; agy) regex=$FM_DELIVERY_AGY_BUSY_REGEX_DEFAULT ;; + cline) regex=$FM_DELIVERY_CLINE_BUSY_REGEX_DEFAULT ;; kimi) regex=$FM_DELIVERY_KIMI_BUSY_REGEX_DEFAULT ;; cursor) regex=$FM_DELIVERY_CURSOR_BUSY_REGEX_DEFAULT ;; '') regex=$FM_DELIVERY_BUSY_REGEX_DEFAULT ;; @@ -417,7 +427,12 @@ FM_COMPOSER_SHELL_PROMPT_GLYPHS=$(printf '%s\n' '>' '$' '%' '#') # `Add a follow-up` once a turn has completed (verified live on cursor-agent # 2026.08.11-e8db854). FM_COMPOSER_IDLE_RE overrides for an unverified harness; # matching is case-insensitive. -FM_COMPOSER_IDLE_RE_DEFAULT='^Type a message\.\.\.$|^Ask anything(\.\.\.|…)|^Plan, search, build anything$|^Add a follow-up$' +# cline renders `Ask anything...` in a session that already has turns and the +# welcome placeholder `What can I do for you?` in a fresh session (verified live, +# cline 3.0.62); both are dim/muted placeholders in an otherwise-empty bordered +# composer and both must read `empty`, or a first cline spawn's readiness gate +# would time out on a fresh profile. +FM_COMPOSER_IDLE_RE_DEFAULT='^Type a message\.\.\.$|^Ask anything(\.\.\.|…)|^Plan, search, build anything$|^Add a follow-up$|^What can I do for you\?$' # Opencode draws a mode/model footer line INSIDE its left-bar composer # ("Build · GPT-5.5 Fast OpenAI · high"). It is composer furniture, not typed diff --git a/bin/fm-control-lib.sh b/bin/fm-control-lib.sh index 516a00b4364..6d2fa239c4c 100644 --- a/bin/fm-control-lib.sh +++ b/bin/fm-control-lib.sh @@ -63,7 +63,7 @@ fm_control_verb_allowed() { # # than guessed at, exactly as a spawn on it would be. fm_control_harness_supported() { # case "${1-}" in - claude|codex|opencode|pi|pi-signed|grok|kimi|cursor|gemini|muse|rovo|omp|agy) return 0 ;; + claude|codex|opencode|pi|pi-signed|grok|kimi|cursor|gemini|muse|rovo|omp|agy|cline) return 0 ;; esac return 1 } @@ -83,6 +83,7 @@ fm_control_harness_family() { # pi-signed) printf 'pi-signed' ;; omp) printf 'omp' ;; agy) printf 'agy' ;; + cline*) printf 'cline' ;; claude*) printf 'claude' ;; codex*) printf 'codex' ;; opencode*) printf 'opencode' ;; @@ -96,9 +97,10 @@ fm_control_harness_family() { # esac } -# Which task kinds an adapter is verified to run. muse, gemini, rovo, and agy -# are crewmate/scout adapters only: none has a primary supervision protocol, -# and bin/fm-spawn.sh refuses a --secondmate launch on any of them. The control +# Which task kinds an adapter is verified to run. muse, gemini, rovo, agy, and +# cline are crewmate/scout adapters only: none has a primary supervision +# protocol, and bin/fm-spawn.sh refuses a --secondmate launch on any of them. +# The control # plane asks this BEFORE it stops anything, so an incompatible relaunch target is # refused while the current agent is still running rather than after it has # been stopped. @@ -106,7 +108,7 @@ fm_control_harness_supports_kind() { # local harness=${1-} kind=${2-} fm_control_harness_supported "$harness" || return 1 case "$harness" in - muse|gemini|rovo|agy) [ "$kind" != secondmate ] || return 1 ;; + muse|gemini|rovo|agy|cline) [ "$kind" != secondmate ] || return 1 ;; esac return 0 } @@ -120,10 +122,13 @@ fm_control_harness_supports_kind() { # # with an idle composer and no repollution (verified live, agy 1.2.0 through # Herdr). omp (Oh My Pi) shares Pi's single Escape, empty composer # afterwards, and /quit exit (verified omp 18.1.2 in a PTY, re-verified 18.1.11 -# through Herdr). +# through Herdr). cline cancels a running turn on a single Escape while its +# busy row reads `(esc to cancel)`; the turn stops and the composer returns to +# the `Ask anything...` placeholder with no prompt repollution (verified live, +# cline 3.0.62). fm_control_interrupt_key() { # case "${1-}" in - claude|codex|opencode|pi|pi-signed|omp|kimi|cursor|gemini|muse|rovo|agy) printf 'Escape' ;; + claude|codex|opencode|pi|pi-signed|omp|kimi|cursor|gemini|muse|rovo|agy|cline) printf 'Escape' ;; grok) printf 'C-c' ;; *) return 1 ;; esac @@ -134,7 +139,7 @@ fm_control_interrupt_key() { # fm_control_interrupt_repeat() { # case "${1-}" in opencode) printf '2' ;; - claude|codex|pi|pi-signed|omp|grok|kimi|cursor|gemini|muse|rovo|agy) printf '1' ;; + claude|codex|pi|pi-signed|omp|grok|kimi|cursor|gemini|muse|rovo|agy|cline) printf '1' ;; *) return 1 ;; esac } @@ -155,7 +160,7 @@ fm_control_interrupt_repeat() { # fm_control_interrupt_clear_key() { # case "${1-}" in muse) printf 'C-u' ;; - claude|codex|opencode|pi|pi-signed|omp|grok|kimi|cursor|gemini|rovo|agy) ;; + claude|codex|opencode|pi|pi-signed|omp|grok|kimi|cursor|gemini|rovo|agy|cline) ;; *) return 1 ;; esac } @@ -170,7 +175,10 @@ fm_control_interrupt_ack_source() { # # rovo's TUI prints "Agent cancelled" on Escape, but for parity with # claude/cursor this stays 'none': the ack is a rendered string, not a # recorded state source, and rovo has no busy wiring to confirm against. - claude|codex|opencode|pi|pi-signed|omp|grok|kimi|cursor|gemini|rovo|agy) printf 'none' ;; + # cline records an abort through its TaskCancel/SessionShutdown hooks, + # which clear the busy record, but the control plane still claims no + # rendered acknowledgement string. + claude|codex|opencode|pi|pi-signed|omp|grok|kimi|cursor|gemini|rovo|agy|cline) printf 'none' ;; *) return 1 ;; esac } @@ -178,7 +186,7 @@ fm_control_interrupt_ack_source() { # # The command that exits the agent from its own composer. fm_control_exit_command() { # case "${1-}" in - claude|opencode|grok|kimi|cursor|muse|rovo) printf '/exit' ;; + claude|opencode|grok|kimi|cursor|muse|rovo|cline) printf '/exit' ;; codex|pi|pi-signed|omp|gemini|agy) printf '/quit' ;; *) return 1 ;; esac @@ -244,6 +252,17 @@ fm_control_harness_wiring_paths() { # printf '%s\n' "$state/$id.muse-session-current" ;; cursor) printf '%s\n' "$state/$id.cursor-session" ;; + # cline discovers its hook config files from the workspace's .cline/hooks + # directory at session start, so the per-task wiring is a set of + # worktree-resident executable files. Relaunch (including a harness switch) + # removes each one; the shared clear path also prunes the emptied parents. + cline) + printf '%s\n' "$wt/.cline/hooks/TaskStart" + printf '%s\n' "$wt/.cline/hooks/TaskComplete" + printf '%s\n' "$wt/.cline/hooks/TaskCancel" + printf '%s\n' "$wt/.cline/hooks/TaskError" + printf '%s\n' "$wt/.cline/hooks/SessionShutdown" + ;; # gemini's busy-state and turn-end hooks live in a firstmate-owned # settings file the launch reaches through GEMINI_CLI_SYSTEM_SETTINGS_PATH, # so retiring that one file retires the whole incarnation's wiring. Nothing diff --git a/bin/fm-harness.sh b/bin/fm-harness.sh index 7989643f1b6..f9d63282b83 100755 --- a/bin/fm-harness.sh +++ b/bin/fm-harness.sh @@ -1,6 +1,6 @@ #!/usr/bin/env bash # Detect the agent harness this process tree runs on. -# Usage: fm-harness.sh print own harness: claude|codex|opencode|pi|pi-signed|grok|kimi|cursor|gemini|muse|rovo|omp|agy|unknown +# Usage: fm-harness.sh print own harness: claude|codex|opencode|pi|pi-signed|grok|kimi|cursor|gemini|muse|rovo|omp|agy|cline|unknown # fm-harness.sh crew print the effective CREWMATE harness # (config/crew-harness; "default" resolves to own) # fm-harness.sh secondmate print the harness the PRIMARY uses to launch @@ -133,7 +133,8 @@ harness_marker() { # identified, and any rule that must be RELIABLE under grok has to test the hook # markers too (see .claude/settings.json Stop entries, docs/turnend-guard.md). [ "${GROK_AGENT:-}" = "1" ] && { echo grok; return; } - # codex, opencode, kimi, muse, and agy publish no harness-identity marker at all, so + # codex, opencode, kimi, muse, agy, and cline publish no harness-identity marker at + # all, so # they are never named here and are identified by ancestry alone. That is the # whole reason a foreign marker must not outrank ancestry: with markers winning # unconditionally, any retained CLAUDECODE would silently rename one of them. @@ -228,6 +229,15 @@ harness_process_verdict() { # # inherited launcher value, not an agy identity), so like muse it is # detected by ancestry alone. agy) echo "comm agy"; return ;; + # cline (Cline CLI 3.0.62, an OpenTUI terminal app) launches its agent as a + # native binary whose process name is exactly `.cline` (verified live: the + # long-lived agent process under the `node` wrapper reports `comm=.cline` + # with argv[0] `.cline`). Anchored, never `*cline*`, so unrelated commands + # cannot be misread as this harness. cline publishes no harness-identity + # marker of its own, so like muse and agy it is detected by ancestry alone. + # The cline hub daemon also runs as `.cline`, but it is reparented to init + # and is never an ancestor of a worker, so it cannot claim this identity. + .cline|cline) echo "comm cline"; return ;; node*|python*) # Bare interpreter: match the harness name in its script path. args=$(ps -o args= -p "$pid" 2>/dev/null) @@ -241,6 +251,11 @@ harness_process_verdict() { # *opencode*) echo "args opencode"; return ;; *grok*) echo "args grok"; return ;; *" pi "*|*/pi) echo "args pi"; return ;; + # cline's node wrapper is `node .../bin/cline`, and its CLI bundle path + # carries `@cline/cli`. Both are anchored path fragments, so an + # unrelated node script whose name merely contains "cline" is not + # claimed. + *"/bin/cline"*|*"@cline/cli"*) echo "args cline"; return ;; esac ;; esac } diff --git a/bin/fm-quota-choose.sh b/bin/fm-quota-choose.sh index 3c7fa891c56..a00ffe08aec 100755 --- a/bin/fm-quota-choose.sh +++ b/bin/fm-quota-choose.sh @@ -332,6 +332,13 @@ provider_for_harness() { kimi) printf 'kimi\n' ;; cursor) printf 'cursor\n' ;; muse) printf 'meta\n' ;; + # ClinePass (provider cline-pass, harness cline) is a subscription the + # current quota-axi snapshot does not model, so this maps to its own family + # name and quota-axi reports it as unknown. Unmodeled provider quota is + # disclosed uncertainty, not a refusal: the agent-driven quota-array-dispatch + # procedure keeps an unmodeled candidate eligible, while this helper's + # known-positive-only rule skips it rather than dying on an unknown harness. + cline) printf 'cline-pass\n' ;; *) return 1 ;; esac } diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index 88fa2fcfabe..592cae6d930 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -135,7 +135,7 @@ # profile consultation. A --secondmate spawn is exempt and resolves the SECONDMATE # harness (config/secondmate-harness -> config/crew-harness -> own), so the # secondmate-vs-crewmate split is DURABLE across every respawn (recovery, -# /updatefirstmate, restart). A bare adapter name (claude|codex|opencode|pi|pi-signed|grok|kimi|cursor|gemini|muse|rovo|omp|agy) +# /updatefirstmate, restart). A bare adapter name (claude|codex|opencode|pi|pi-signed|grok|kimi|cursor|gemini|muse|rovo|omp|agy|cline) # overrides it for this spawn (either kind). A non-flag string containing # whitespace is treated as a RAW launch command - the escape hatch for verifying # new adapters. For pi and pi-signed, fm-spawn resolves the selected executable @@ -291,7 +291,8 @@ # plus a gitignored .fm-grok-turnend worktree pointer and a state token. # muse installs no hook at all - its plugin engine is off in the default build - so # it writes state/.muse-session to bind the pane to muse's own session event -# log; muse, gemini, and agy are crewmate/scout only and are refused for --secondmate. +# log; muse, gemini, agy, and cline are crewmate/scout only and are refused for +# --secondmate. # rovo installs no hook either - its eventHooks fire at tool granularity only, # never turn-end - so it carries no busy-source wiring at all and no turn-end # hook. A positional brief is dead-on-arrival (rovo loads, never works, and drops @@ -1860,6 +1861,23 @@ launch_template() { # when a supported effort is requested, since a second --config-override # would silently discard the first (confirmed live). rovo) printf '%s' 'env -u CLAUDECODE -u PI_CODING_AGENT -u GROK_AGENT -u FM_PI_HARNESS __ROVOBIN__ run --yolo __MODELFLAG____ROVOCONFIGOVERRIDE__' ;; + # cline (Cline CLI 3.0.62, an OpenTUI terminal app) launches its interactive + # TUI bare with `-i` and receives its brief only after the readiness gate + # below - the kimi/rovo launch-then-send shape. A positional prompt on `-i` + # is NOT reliable: cline's one-time "Introducing Cline Desktop" first-run + # splash consumes the first submitted line, so a brief carried on the launch + # command can be silently swallowed. --auto-approve true is cline's documented + # default and is passed explicitly so an operator's own --auto-approve false + # default can never park the unattended worker. -c pins the exact worktree so + # cline discovers the per-task .cline/hooks wiring written below and cannot + # drift to another workspace. --model takes the full `/` id + # (`cline-pass/deepseek-v4-flash`); cline derives the provider from that + # prefix, so no separate -P is passed. --thinking accepts + # none|low|medium|high|xhigh. The foreign primary markers are cleared so an + # inherited CLAUDECODE cannot outrank cline's ancestry in a process that only + # reads the environment. Busy state and turn-end ride the workspace hook files + # written below, not the launch command. + cline) printf '%s' 'env -u CLAUDECODE -u PI_CODING_AGENT -u GROK_AGENT -u FM_PI_HARNESS __CLINEBIN__ -i -c __WORKTREE__ --auto-approve true __MODELFLAG____EFFORTFLAG__' ;; *) return 1 ;; esac } @@ -1923,7 +1941,10 @@ esac # secondmate whose supervision cycle could never be armed. # agy has none either: it exposes no hook surface for primary supervision and # docs/supervision-protocols/ carries no agy wake protocol (agy 1.2.0). -if [ "$KIND" = secondmate ] && { [ "$HARNESS" = muse ] || [ "$HARNESS" = gemini ] || [ "$HARNESS" = agy ]; }; then +# cline has none either: docs/supervision-protocols/ carries no cline wake +# protocol, and this task verified only the crewmate-side launch, busy state, +# interrupt, and exit. +if [ "$KIND" = secondmate ] && { [ "$HARNESS" = muse ] || [ "$HARNESS" = gemini ] || [ "$HARNESS" = agy ] || [ "$HARNESS" = cline ]; }; then echo "error: $HARNESS is a verified crewmate/scout adapter only and cannot run a secondmate; it has no primary supervision protocol. Select a harness verified for secondmates." >&2 exit 1 fi @@ -1983,6 +2004,12 @@ agy) exit 1 } ;; +cline) + CLINE_BIN=$(command -v cline 2>/dev/null) || { + echo "error: cline executable not found on PATH; install Cline CLI or select a different verified harness" >&2 + exit 1 + } + ;; esac # config/secondmate-harness may carry optional model/effort tokens alongside the @@ -2142,7 +2169,7 @@ model_flag_for_harness() { local harness=$1 model=$2 [ -n "$model" ] && [ "$model" != default ] || return 0 case "$harness" in - claude | codex | opencode | pi | pi-signed | grok | kimi | cursor | gemini | muse | rovo | omp | agy) + claude | codex | opencode | pi | pi-signed | grok | kimi | cursor | gemini | muse | rovo | omp | agy | cline) printf -- '--model %s ' "$(shell_quote "$model")" ;; esac @@ -2185,6 +2212,14 @@ effort_flag_for_harness() { low | medium | high) printf -- '--effort %s ' "$(shell_quote "$effort")" ;; esac ;; + cline) + # cline --thinking validates exactly none|low|medium|high|xhigh; the shared + # vocabulary maps straight across, and max (above cline's ceiling) is + # omitted under the record-and-omit contract. + case "$effort" in + low | medium | high | xhigh) printf -- '--thinking %s ' "$(shell_quote "$effort")" ;; + esac + ;; pi | pi-signed) # Pi 0.80.6 accepts the full shared effort vocabulary, including max, through # its --thinking flag. @@ -3374,6 +3409,82 @@ rovo_endpoint_cleanup() { fm_backend_kill "$BACKEND" "$T" "$tab_id" "fm-$ID" 2>/dev/null || true } +# cline launches bare and receives its brief pointer only after a readiness +# gate, then a delivery-confirmation gate - the kimi/rovo launch-then-send +# shape, forced by cline's first-run "Introducing Cline Desktop" splash, which +# consumes the first submitted line. Both gates route composer-emptiness through +# the shared classifier (fm_backend_composer_state) so they read the same shape +# every steer guard does. Delivery is confirmed from the recorded busy state +# opened by the workspace TaskStart hook (bin/fm-busy-lib.sh, source cline-hook) +# rather than a rendered spinner, and falls back to the pinned `(esc to cancel)` +# token for a pane whose hook has not landed yet. +cline_capture() { + fm_backend_capture "$BACKEND" "$T" 120 "$W" 2>/dev/null || true +} + +cline_composer_is_empty() { + [ "$(fm_backend_composer_state "$BACKEND" "$T" "$W" 2>/dev/null)" = empty ] +} + +# cline's one-time splash renders on a fresh profile and eats the first +# submitted line. Any key but Enter closes it; Escape is delivered once so the +# readiness loop can then see the real composer. +CLINE_SPLASH_DISMISSED=0 + +cline_wait_for_ready() { + local pane i=0 max=${FM_CLINE_READY_POLLS:-60} interval=${FM_CLINE_POLL_INTERVAL:-0.5} + while [ "$i" -lt "$max" ]; do + pane=$(cline_capture) + if printf '%s\n' "$pane" | grep -Fq 'Introducing Cline Desktop' || + printf '%s\n' "$pane" | grep -Fq 'Press Enter to open'; then + if [ "$CLINE_SPLASH_DISMISSED" -eq 0 ]; then + fm_backend_send_key "$BACKEND" "$T" Escape >/dev/null 2>&1 || true + CLINE_SPLASH_DISMISSED=1 + fi + elif printf '%s\n' "$pane" | grep -Fq 'Auto-approve'; then + # cline's own status row proves the TUI is up. Composer-emptiness is NOT + # used as the lead readiness signal: cline renders its idle placeholder + # (`Ask anything...`, or the fresh-session `What can I do for you?`) as a + # muted truecolor grey (~135.5 luma) just ABOVE the fleet-wide ghost-luma + # ceiling of 128, so the shared classifier reads an idle cline composer + # `pending` on the styled tmux/herdr captures. That is the same known gap + # rovo documents (docs/verification/rovo.md); solving it fleet-wide would + # require moving the shared ceiling into codex's starfield band. The status + # row is cline-specific, stable, and present exactly when the TUI is ready. + return 0 + elif cline_composer_is_empty; then + return 0 + fi + i=$((i + 1)) + [ "$i" -ge "$max" ] || sleep "$interval" + done + return 1 +} + +cline_delivery_is_confirmed() { # + local pane=$1 verdict + verdict=$(fm_busy_classify "$BACKEND" "$T" cline "$ID" "$STATE" "$pane" 2>/dev/null || true) + case "$verdict" in busy\ *) return 0 ;; esac + printf '%s\n' "$pane" | grep -qE "$FM_DELIVERY_CLINE_BUSY_REGEX_DEFAULT" +} + +cline_wait_for_delivery() { + local pane i=0 max=${FM_CLINE_DELIVERY_POLLS:-40} interval=${FM_CLINE_POLL_INTERVAL:-0.5} + while [ "$i" -lt "$max" ]; do + pane=$(cline_capture) + cline_delivery_is_confirmed "$pane" && return 0 + i=$((i + 1)) + [ "$i" -ge "$max" ] || sleep "$interval" + done + return 1 +} + +cline_spawn_fail() { # + printf 'failed: %s\n' "$1" >>"$STATE/$ID.status" + echo "error: $1; inspect window $T" >&2 + rovo_endpoint_cleanup +} + # agy carries its brief on the launch command, so it needs no delivery gate, # but a worktree agy does not trust parks the TUI on the folder-trust dialog # and an unanswered dialog sends the turn into agy's scratch directory instead @@ -3670,6 +3781,18 @@ if [ "$KIND" != secondmate ]; then [ "$RELAUNCH" -ne 1 ] || RELAUNCH_REPLACEMENT_BUSY_GEN=$BUSY_GEN fi ;; + cline*) + # cline launches BARE and receives its brief only after the readiness gate + # below, so the launch itself is NOT a submitted turn. Seed the record idle; + # the TaskStart hook flips it busy when the brief pointer is actually + # submitted. Only the plain `cline` adapter (or a raw command whose basename + # begins with it) is armed here; a cline worker is crewmate/scout only. + BUSY_GEN=$("$FM_ROOT/bin/fm-busy-event.sh" arm "$STATE_REAL" "$ID" --state idle --source fm-spawn --event launch) || { + echo "error: failed to arm the busy-state contract for $ID" >&2 + exit 1 + } + [ "$RELAUNCH" -ne 1 ] || RELAUNCH_REPLACEMENT_BUSY_GEN=$BUSY_GEN + ;; kimi*) # Standalone Kimi stays unknown until fm_busy_kimi_verified opens on a # live-verified installed version (bin/fm-busy-lib.sh owns the gate and @@ -3738,6 +3861,28 @@ EOF EOF fi ;; + cline*) + # Semantic busy-state and turn-end hooks for cline (bin/fm-busy-lib.sh, + # source cline-hook). cline discovers hook config files from the workspace's + # .cline/hooks directory at session start, so the wiring is a set of + # executable files written before launch. TaskStart opens a turn; + # TaskComplete (normal end), TaskCancel (abort), TaskError (agent error), and + # SessionShutdown (process exit) all close it, so no abnormal end can leave + # a stale busy record. TaskComplete also touches the turn-ended NOTIFICATION + # for the watcher. Every hook tolerates a refused event (|| true) so a + # stale-gen writer can never break cline's own lifecycle. exclude_path keeps + # the wiring out of git's view. + mkdir -p "$WT/.cline/hooks" + busy_cmd_prefix="$(shell_quote "$FM_ROOT/bin/fm-busy-event.sh") apply $(shell_quote "$STATE_REAL") $(shell_quote "$ID")" + busy_suffix="--gen $(shell_quote "$BUSY_GEN") --source cline-hook" + printf '#!/bin/sh\n%s\n' "$busy_cmd_prefix busy $busy_suffix --event task-start >/dev/null 2>&1 || true" >"$WT/.cline/hooks/TaskStart" + printf '#!/bin/sh\ntouch %s; %s\n' "$(shell_quote "$TURNEND")" "$busy_cmd_prefix idle $busy_suffix --event task-complete >/dev/null 2>&1 || true" >"$WT/.cline/hooks/TaskComplete" + printf '#!/bin/sh\n%s\n' "$busy_cmd_prefix idle $busy_suffix --event task-cancel >/dev/null 2>&1 || true" >"$WT/.cline/hooks/TaskCancel" + printf '#!/bin/sh\n%s\n' "$busy_cmd_prefix idle $busy_suffix --event task-error >/dev/null 2>&1 || true" >"$WT/.cline/hooks/TaskError" + printf '#!/bin/sh\n%s\n' "$busy_cmd_prefix idle $busy_suffix --event session-shutdown >/dev/null 2>&1 || true" >"$WT/.cline/hooks/SessionShutdown" + chmod +x "$WT/.cline/hooks/TaskStart" "$WT/.cline/hooks/TaskComplete" "$WT/.cline/hooks/TaskCancel" "$WT/.cline/hooks/TaskError" "$WT/.cline/hooks/SessionShutdown" + exclude_path '.cline/' + ;; opencode*) mkdir -p "$WT/.opencode/plugins" cat >"$WT/.opencode/plugins/fm-busy-state.js" </` id cline expects (for example `cline-pass/deepseek-v4-flash` or `cline-pass/glm-5.3`); cline derives the provider from that prefix, so no separate provider field is needed. Every profile array is an implicit quota-aware choice resolved through `quota-array-dispatch`. If no dispatch rule fits, firstmate resolves `default` through the same object-or-array path before falling back to `config/crew-harness`. Except for `ultra`, which refuses unsupported profiles under the native-effort contract above, an effort value the chosen harness does not accept is recorded as `effort=` in task meta for traceability but omitted from the launch flags. diff --git a/docs/documentation-audiences.json b/docs/documentation-audiences.json index 618caa941ec..d513496b1e1 100644 --- a/docs/documentation-audiences.json +++ b/docs/documentation-audiences.json @@ -184,6 +184,10 @@ "path": ".agents/skills/harness-adapters/references/harness/claude.md", "audience": "agent-runtime" }, + { + "path": ".agents/skills/harness-adapters/references/harness/cline.md", + "audience": "agent-runtime" + }, { "path": ".agents/skills/harness-adapters/references/harness/codex.md", "audience": "agent-runtime" @@ -444,6 +448,10 @@ "path": "docs/verification/agy.md", "audience": "maintainer-verification" }, + { + "path": "docs/verification/cline.md", + "audience": "maintainer-verification" + }, { "path": "docs/verification/dispatch-auth.md", "audience": "maintainer-verification" diff --git a/docs/examples/crew-dispatch.json b/docs/examples/crew-dispatch.json index b404e95e777..2e08cff25ea 100644 --- a/docs/examples/crew-dispatch.json +++ b/docs/examples/crew-dispatch.json @@ -14,13 +14,24 @@ "when": "The task is a big or ambiguous multi-file feature, a risky refactor, or work that requires holding many moving parts in mind.", "use": [ { "harness": "claude", "model": "claude-sonnet-5", "effort": "high" }, - { "harness": "codex", "model": "gpt-5.5", "effort": "high" } + { "harness": "codex", "model": "gpt-5.5", "effort": "high" }, + { "harness": "cline", "model": "cline-pass/deepseek-v4-pro", "effort": "high" } ], - "why": "Use a strong coding profile for big, ambiguous work; resolve the alternatives through quota-array-dispatch." + "why": "Use a strong coding profile for big, ambiguous work; resolve the alternatives through quota-array-dispatch. The ClinePass open-weights seat keeps a parallel lane open when a subscription quota dies." + }, + { + "when": "The task is a medium-sized coding ship that can run in parallel with others and a ClinePass open-weights coding model is a cheaper seat than a frontier subscription.", + "use": [ + { "harness": "cline", "model": "cline-pass/glm-5.3", "effort": "high" }, + { "harness": "cline", "model": "cline-pass/kimi-k2.7-code", "effort": "high" }, + { "harness": "cline", "model": "cline-pass/deepseek-v4-flash", "effort": "xhigh" } + ], + "why": "Use the ClinePass pool as a third coding lane for medium ships so a scarce frontier or Gemini seat is not overloaded." } ], "default": [ { "harness": "codex", "model": "gpt-5.5", "effort": "medium" }, - { "harness": "pi", "model": "anthropic/claude-sonnet-5", "effort": "medium" } + { "harness": "pi", "model": "anthropic/claude-sonnet-5", "effort": "medium" }, + { "harness": "cline", "model": "cline-pass/deepseek-v4-flash", "effort": "high" } ] } diff --git a/docs/tmux-backend.md b/docs/tmux-backend.md index 77b836c04fa..9fe70029211 100644 --- a/docs/tmux-backend.md +++ b/docs/tmux-backend.md @@ -48,7 +48,7 @@ Verify setup by spawning a small task and confirming its `fm-` window appear A target-existence check proves only that the pane exists. The deeper tmux agent-liveness probe first verifies exact window membership, then reads process names to distinguish a running harness from a bare idle shell. -It classifies recognized Claude, Codex, OpenCode, Pi, pi-signed, Grok, Kimi, Cursor, Muse, Rovo, and AGY process identities as `alive`, common shells as `dead`, an authoritatively absent window as `missing`, unreadable state as `unreadable`, and every other process as `ambiguous`. +It classifies recognized Claude, Codex, OpenCode, Pi, pi-signed, Grok, Kimi, Cursor, Muse, Rovo, AGY, and cline process identities as `alive`, common shells as `dead`, an authoritatively absent window as `missing`, unreadable state as `unreadable`, and every other process as `ambiguous`. The process-name vocabulary behind those verdicts is owned by `bin/fm-agent-process-lib.sh` and shared with the Herdr adapter, which proves a registered agent against the same names ([herdr-backend.md](herdr-backend.md) "Restart and liveness behavior"). Only `dead` and `missing` authorize recovery because a false dead result could launch a duplicate agent. @@ -63,6 +63,7 @@ Direct executable identities `pi`, `pi-signed`, and `Pi` remain accepted exactly Muse is likewise anchored to the exact `muse` launcher identity or the installed `muse-bin-` prefix, so unrelated names such as `musescore` and `amuse` remain ambiguous. omp is anchored to the exact `omp` identity for the same reason, so `ompd` and `comp` remain ambiguous. AGY is anchored to the exact `agy` identity for the same reason, so unrelated names containing that fragment remain ambiguous. +cline is anchored to the exact `.cline` native-binary identity, with a node-wrapper fallback on the anchored path fragments `/bin/cline` and `@cline/cli`, so a script merely containing "cline" is not accepted. Cursor is identified from its exact `cursor-agent` identity or versioned install tree in the foreground process path or structured argv[0]; a bare `node` or unrelated `agent` remains ambiguous. The CI-enforced portable regression and opt-in real-harness drift guard follow the split owned by `.agents/skills/firstmate-coding-guidelines/SKILL.md`. diff --git a/docs/trace-context.md b/docs/trace-context.md index 6a9cb5e83b9..6cef700cdcd 100644 --- a/docs/trace-context.md +++ b/docs/trace-context.md @@ -23,7 +23,7 @@ When enabled, for each spawn Firstmate resolves one W3C `traceparent` carrier fo This feature parents no SDK span by itself. Because the injected carrier and the recorded carrier are the same string, an observer that reads the metadata reconstructs exactly the identity the child received. -The injection sits at the unconditional pre-launch export site, so it covers ship and scout spawns across `claude`, `codex`, `opencode`, `pi`, `pi-signed`, `grok`, `kimi`, `cursor`, `gemini`, `muse`, `rovo`, and `agy`, plus Secondmate spawns across that same set except the deliberately crewmate-only `gemini`, `muse`, `rovo`, and `agy` adapters. +The injection sits at the unconditional pre-launch export site, so it covers ship and scout spawns across `claude`, `codex`, `opencode`, `pi`, `pi-signed`, `grok`, `kimi`, `cursor`, `gemini`, `muse`, `rovo`, `agy`, and `cline`, plus Secondmate spawns across that same set except the deliberately crewmate-only `gemini`, `muse`, `rovo`, `agy`, and `cline` adapters. This is the same coverage `GOTMPDIR` already has and requires no trace-specific `launch_template()` behavior. Ship and scout spawns reach that site on every spawn backend (`tmux`, `herdr`, `zellij`, `orca`, `cmux`); a Secondmate reaches it on every backend that accepts a Secondmate spawn (`tmux`, `herdr`, `zellij`), because `bin/fm-spawn.sh` rejects a Secondmate on `orca` and `cmux`. diff --git a/docs/verification/cline.md b/docs/verification/cline.md new file mode 100644 index 00000000000..e9e4342a82b --- /dev/null +++ b/docs/verification/cline.md @@ -0,0 +1,149 @@ +# Verification: the cline (Cline CLI) crewmate/scout adapter + +Active empirical facts for firstmate's cline adapter. +The skill tree rooted at [`.agents/skills/harness-adapters/SKILL.md`](../../.agents/skills/harness-adapters/SKILL.md) owns the operating facts through [`references/harness/cline.md`](../../.agents/skills/harness-adapters/references/harness/cline.md); this record owns how they were established and what is still unproven. + +## Subject + +| Field | Value | +|---|---| +| Version | `cline 3.0.62` (core `0.0.83`) | +| Verified | 2026-09-16 | +| Binary | `cline` on `PATH` via `/home/azureuser/.npm-global/bin/cline`; the live agent is `/home/azureuser/.npm-global/lib/node_modules/cline/bin/.cline` | +| Platform | Linux x64 (Ubuntu, kernel 6.14.0) | +| Backend | tmux 3.4 (portable evidence below). The Herdr path is exercised by the live guard named at the end of this record. | + +Every command below ran in a disposable directory with no firstmate fleet state in view. +Model calls ran on the captain's authenticated ClinePass subscription; the token cost was a handful of one-shot replies. + +## Detection: ancestry only, no marker + +``` +$ cline --version +3.0.62 +``` + +A live TUI carries no cline-identity environment variable, so no marker is promoted. +The long-lived agent process is a native binary named `.cline`: + +``` +$ ps -eo pid,ppid,comm,args | grep -i '[c]line' +1563856 1563614 node node /home/azureuser/.npm-global/bin/cline -P cline-pass -m cline-pass/deepseek-v4-flash -i +1563864 1563856 .cline /home/azureuser/.npm-global/lib/node_modules/cline/bin/.cline -P cline-pass -m cline-pass/deepseek-v4-flash -i +``` + +`bin/fm-harness.sh` therefore matches the anchored process name `.cline`, with a node-wrapper backup on the anchored script-path fragments `/bin/cline` and `@cline/cli`. +`tests/fm-cline-harness.test.sh` pins the anchored match and the rejection of unrelated names containing the fragment. + +## Credential precondition + +``` +$ jq -r '.lastUsedProvider' ~/.cline/data/settings/providers.json +clinepass +$ jq -r '.providers | keys[]' ~/.cline/data/settings/providers.json +cline +cline-pass +``` + +ClinePass was signed in through OAuth; `cli-pass` is stored with access and refresh tokens. +Successful runs below prove the credential, so no key export was required. + +## Prompt and model shape + +``` +$ cline --json -P cline-pass -m cline-pass/deepseek-v4-flash "Reply with exactly: READY" +... "text":"READY" ... "model":{"id":"cline-pass/deepseek-v4-flash","provider":"cline-pass", ...} +``` + +The provider id is `cline-pass` (not `clinepass`), and `--model` takes the full `/` id. +`-m cline-pass/deepseek-v4-flash` alone (no `-P`) also resolves, because cline derives the provider from the prefix: + +``` +$ cline --json -m cline-pass/deepseek-v4-flash "Reply exactly: NOFLAG" +... "text":"NOFLAG" ... +``` + +`--thinking low` was accepted on the same binary. +A bare-model value is refused by cline itself with `invalid model format. Expected format: modelType/model`, which is why `bin/fm-spawn.sh` passes the profile model through unchanged. + +## TUI launch and turn lifecycle + +Launched in a tmux pane: + +``` +$ cline -P cline-pass -m cline-pass/deepseek-v4-flash -i "Reply with exactly POSOK" +``` + +The positional prompt auto-submitted and the reply `* POSOK` rendered, so `-i ""` does submit when no first-run splash is showing. + +The workspace `.cline/hooks` directory was then populated with executable `TaskStart`, `TaskComplete`, `TaskCancel`, `TaskError`, `SessionShutdown`, and `UserPromptSubmit` files named for cline's config-file hook events, and the TUI was restarted. +On a clean turn the hook log showed: + +``` +TaskStart +TaskComplete +``` + +`TaskStart` fired as the turn opened and `TaskComplete` fired as it closed. +`UserPromptSubmit` never fired in the TUI, which is why `TaskStart` is used as the open signal. + +While a turn ran the transcript carried the busy row: + +``` +⠸ Thinking... (esc to cancel) +``` + +At turn end the same row was rewritten as `▶ Thinking:` with the token gone; the bottom status row stayed `⏵⏵ Auto-approve all enabled (Shift+Tab)` in both states, so it is not a busy signal. + +## Interrupt and exit + +A single `Escape` while a turn ran stopped it and left the composer at the `Ask anything...` placeholder with no repollution; the hook log gained `SessionShutdown`, and the process stayed alive for further turns. + +`/exit` closed the TUI and printed a session summary before returning to the shell: + +``` +Session Summary + ID 1789518608362_pi4oo + Duration 366s + Model cline-pass:cline-pass/deepseek-v4-flash + CWD /tmp/cline-tui3 + Messages 2 + Continue cline --id 1789518608362_pi4oo +``` + +So exit is `/exit`; resume-by-id is advertised by cline itself, but no firstmate pane-resume contract is claimed (deterministic relaunch is used instead). + +## Composer gap: placeholder luminance above the ghost ceiling + +The idle placeholder renders as a muted truecolor grey, captured live as: + +``` +[1m[38;2;121;184;255m❯[0m[38;2;255;255;255m [38;2;131;137;140mWhat can I do for you?[38;2;255;255;255m +``` + +That foreground is `38;2;131;137;140`, perceived luminance `0.299*131 + 0.587*137 + 0.114*140 = 135.5`, just above `bin/fm-composer-lib.sh`'s fleet-wide `FM_COMPOSER_GHOST_LUMA_MAX` default of 128. +`fm_composer_strip_ghost` therefore leaves it unstripped, and on the styled tmux/herdr captures an idle cline composer classifies `pending`, never `empty` - reproduced live on cline 3.0.62 with both the fresh-session and post-turn placeholders. + +This is the same class of gap `docs/verification/rovo.md` documents and deliberately did not patch by raising the shared ceiling. +Raising the ceiling for cline is also not a free fix: `tests/fm-composer-lib.test.sh`'s codex starfield fixture draws truecolor braille furniture at greys 132, 136, and 138 straddling the 128 ceiling, and the fixture proves those survivors are then resolved by the braille stripper. +A shared ceiling between cline's 135.5 ghost and codex's 136-138 furniture does not exist, so the adapter does not move the shared default. + +Instead, cline's launch-then-send path does not depend on composer-empty: + +- readiness leads with cline's own `Auto-approve` status row (`cline_wait_for_ready`), exactly as rovo's readiness leads with the `Welcome to Rovo!` banner; +- the brief pointer is sent once (one literal send plus one Enter) rather than through the shared retrying submit core, and delivery is confirmed from the recorded `busy cline-hook` state the workspace `TaskStart` hook writes (`cline_wait_for_delivery`); +- steering still rides the shared send path, whose queued-Enter policy converts `pending + busy` to delivered, the same bounded retry rovo and agy already accept on a non-`empty` composer read. + +The blast radius is therefore bounded to composer-emptiness consumers, and the fix, if desired, is a harness-scoped signal the shared composer classifier does not carry today - a follow-up, not this change. + +## Still unproven + +- The Herdr backend path end to end for cline; only tmux was driven interactively in this pass. The live guard below is the repeatable refresh command. +- The first-run "Introducing Cline Desktop" splash dismissal on a genuinely fresh profile. The live guard stages a copied `~/.cline` (a fresh profile) and exercises the splash branch, but a genuinely first-run store on a clean host was not observed. +- A `/abort`-driven cancellation distinct from `Escape`; only `Escape` was exercised. +- Interrupt acknowledgement beyond the `SessionShutdown` hook: cline records the abort through its hooks, but no rendered acknowledgement string is claimed. + +## Refresh command + +`bin/fm-test-run.sh` lists `tests/fm-cline-signals-live-e2e.test.sh` in the `live-harness-optin` family. +Run it after every cline upgrade and before trusting refreshed per-harness evidence; it exercises the installed `cline` for real and fails naming the harness and version when the busy or turn-end signal no longer holds. diff --git a/tests/fm-cline-harness.test.sh b/tests/fm-cline-harness.test.sh new file mode 100755 index 00000000000..f6415b8fecf --- /dev/null +++ b/tests/fm-cline-harness.test.sh @@ -0,0 +1,207 @@ +#!/usr/bin/env bash +# Behavior tests for the verified Cline CLI crewmate/scout adapter. +# +# The facts pinned here are the ones a cline release could silently change and +# the ones a wrong guess would make dangerous: +# 1. cline publishes no harness-identity marker, so detection is ancestry +# alone: the anchored live process name `.cline`, with the node wrapper's +# anchored script-path fragments `/bin/cline` and `@cline/cli` as backup. +# 2. The anchored match must never claim unrelated commands containing the +# fragment, and a structural `.cline` ancestor must outrank a retained +# CLAUDECODE. +# 3. cline is a crewmate/scout adapter only: control tables refuse a +# secondmate target and expose the verified interrupt, exit, and wiring. +# 4. Busy state is a trusted semantic source (cline-hook); the rendered +# `(esc to cancel)` token is a delivery-guard signature only and never +# crosses harnesses. +# 5. The live process name is classified as an agent by backend liveness. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +# bin/fm-harness.sh checks verified ENV markers before ancestry; drop the +# ambient markers so the asserted verdict does not depend on the launching +# harness. +unset CLAUDECODE PI_CODING_AGENT FM_PI_HARNESS GROK_AGENT CURSOR_AGENT CURSOR_INVOKED_AS \ + ATLASSIAN_AGENT_TYPE ROVODEV_CLI GEMINI_CLI AGENT FM_OMP_HARNESS + +# shellcheck source=/dev/null +. "$ROOT/bin/fm-control-lib.sh" +# shellcheck source=/dev/null +. "$ROOT/bin/fm-busy-lib.sh" +# shellcheck source=/dev/null +. "$ROOT/bin/fm-composer-lib.sh" + +HARNESS="$ROOT/bin/fm-harness.sh" +TMP_ROOT=$(fm_test_tmproot fm-cline-harness) + +test_cline_ancestry_detects_the_native_command_name() { + local fakebin out + fakebin=$(fm_fakebin "$TMP_ROOT/anc-native") + cat > "$fakebin/ps" <<'SH' +#!/usr/bin/env bash +case "$*" in + *"comm="*) printf '%s\n' '/home/azureuser/.npm-global/lib/node_modules/cline/bin/.cline'; exit 0 ;; + *"args="*) printf '%s\n' '.cline -i -c /work'; exit 0 ;; +esac +exit 1 +SH + chmod +x "$fakebin/ps" + out=$(PATH="$fakebin:$PATH" "$HARNESS") + [ "$out" = cline ] \ + || fail "the native .cline command must be detected by ancestry, got '$out'" + pass "fm-harness.sh: ancestry detects the native .cline command" +} + +test_cline_ancestry_detects_the_node_wrapper() { + local fakebin out + fakebin=$(fm_fakebin "$TMP_ROOT/anc-node") + cat > "$fakebin/ps" <<'SH' +#!/usr/bin/env bash +case "$*" in + *"comm="*) printf '%s\n' node; exit 0 ;; + *"args="*) printf '%s\n' 'node /home/azureuser/.npm-global/bin/cline -i'; exit 0 ;; +esac +exit 1 +SH + chmod +x "$fakebin/ps" + out=$(PATH="$fakebin:$PATH" "$HARNESS") + [ "$out" = cline ] \ + || fail "the node cline wrapper must be detected by its script path, got '$out'" + pass "fm-harness.sh: ancestry detects the node cline wrapper" +} + +test_cline_ancestry_rejects_unrelated_mentions() { + local fakebin out + fakebin=$(fm_fakebin "$TMP_ROOT/anc-negatives") + cat > "$fakebin/ps" <<'SH' +#!/usr/bin/env bash +case "$*" in + *"comm="*) printf '%s\n' "${FAKE_PS_COMM:?}"; exit 0 ;; + *"args="*) printf '%s\n' "${FAKE_PS_ARGS:?}"; exit 0 ;; +esac +exit 1 +SH + chmod +x "$fakebin/ps" + + out=$(FAKE_PS_COMM=mycline FAKE_PS_ARGS='mycline --serve' \ + PATH="$fakebin:$PATH" "$HARNESS") + [ "$out" != cline ] \ + || fail "an unrelated mycline command must not detect cline, got '$out'" + + out=$(FAKE_PS_COMM=node FAKE_PS_ARGS='node /opt/decline/index.js' \ + PATH="$fakebin:$PATH" "$HARNESS") + [ "$out" != cline ] \ + || fail "a node script merely containing cline must not detect cline, got '$out'" + + out=$(FAKE_PS_COMM=bash FAKE_PS_ARGS='bash -c "echo cline --help"' \ + PATH="$fakebin:$PATH" "$HARNESS") + [ "$out" != cline ] \ + || fail "a later shell argument naming cline must not detect cline, got '$out'" + pass "fm-harness.sh: ancestry rejects unrelated cline mentions" +} + +test_cline_structural_ancestor_outranks_a_retained_marker() { + local fakebin out + # cline does not clear an inherited CLAUDECODE, so a structural .cline + # ancestor must still outrank the retained marker rather than being renamed + # away from it. + fakebin=$(fm_fakebin "$TMP_ROOT/anc-claude") + cat > "$fakebin/ps" <<'SH' +#!/usr/bin/env bash +case "$*" in + *"comm="*) printf '%s\n' .cline; exit 0 ;; + *"args="*) printf '%s\n' '.cline -i'; exit 0 ;; +esac +exit 1 +SH + chmod +x "$fakebin/ps" + out=$(CLAUDECODE=1 PATH="$fakebin:$PATH" "$HARNESS") + [ "$out" = cline ] \ + || fail "a structural .cline ancestor must outrank an inherited CLAUDECODE, got '$out'" + pass "fm-harness.sh: a structural .cline ancestor outranks a retained marker" +} + +test_cline_control_mechanics_are_the_verified_ones() { + fm_control_harness_supported cline || fail "cline must be a supported control harness" + [ "$(fm_control_harness_family cline)" = cline ] || fail "cline must map to its own family" + fm_control_harness_supports_kind cline scout || fail "cline must run scouts" + fm_control_harness_supports_kind cline ship || fail "cline must run ships" + fm_control_harness_supports_kind cline secondmate \ + && fail "cline must refuse secondmates" || true + [ "$(fm_control_interrupt_key cline)" = Escape ] || fail "cline must interrupt on Escape" + [ "$(fm_control_interrupt_repeat cline)" = 1 ] || fail "cline must interrupt on a single press" + [ -z "$(fm_control_interrupt_clear_key cline)" ] || fail "cline must need no clear key" + [ "$(fm_control_interrupt_ack_source cline)" = none ] || fail "cline must have no ack source" + [ "$(fm_control_exit_command cline)" = /exit ] || fail "cline must exit on /exit" + pass "fm-control-lib: cline mechanics are Escape once, no clear key, and /exit" +} + +test_cline_wiring_paths_are_the_workspace_hooks() { + local got + got=$(fm_control_harness_wiring_paths cline /wt /state task1) + local want + want=$(printf '%s\n' \ + '/wt/.cline/hooks/TaskStart' \ + '/wt/.cline/hooks/TaskComplete' \ + '/wt/.cline/hooks/TaskCancel' \ + '/wt/.cline/hooks/TaskError' \ + '/wt/.cline/hooks/SessionShutdown') + [ "$got" = "$want" ] || fail "cline wiring paths were not the workspace hook files, got '$got'" + pass "fm-control-lib: cline wiring paths are the workspace hook files" +} + +test_cline_busy_source_is_trusted() { + local sources + sources=$(fm_busy_sources_for_harness cline) + case " $sources " in + *" cline-hook "*) ;; + *) fail "cline's busy source list must include cline-hook, got '$sources'" ;; + esac + fm_busy_source_trusted cline cline-hook \ + || fail "cline-hook must be trusted for a cline task" + fm_busy_source_trusted cline gemini-hook \ + && fail "cline must never trust another adapter's busy source" || true + pass "fm-busy-lib: cline trusts only its own cline-hook source" +} + +test_cline_delivery_signature_is_harness_scoped() { + printf '(esc to cancel)\n' | fm_busy_lines_match cline \ + || fail "harness=cline must match its esc-to-cancel token" + printf '(esc to cancel)\n' | fm_busy_lines_match agy \ + || fail "harness=agy keeps its own esc-to-cancel token" + printf '(esc to cancel)\n' | fm_busy_lines_match grok \ + && fail "harness=grok must never borrow the cline token" || true + printf 'Ctrl+c:cancel\n' | fm_busy_lines_match cline \ + && fail "harness=cline must never borrow grok's token" || true + printf '(esc to cancel)\n' | fm_busy_lines_match spaceship \ + && fail "an unverified harness must match nothing" || true + pass "fm-composer-lib: cline delivery signature never crosses harnesses" +} + +test_cline_tmux_names_the_native_binary_an_agent() { + local got + # shellcheck source=/dev/null + . "$ROOT/bin/fm-backend.sh" + fm_backend_source tmux || fail "fm_backend_source tmux failed" + got=$(fm_agent_process_classify_name .cline) + [ "$got" = agent ] || fail "tmux liveness must read .cline as an agent, got '$got'" + got=$(fm_agent_process_classify_name cline) + [ "$got" = agent ] || fail "tmux liveness must read cline as an agent, got '$got'" + got=$(fm_agent_process_classify_name mycline) + [ "$got" = other ] || fail "tmux liveness must not read mycline as an agent, got '$got'" + got=$(fm_agent_process_classify_name bash) + [ "$got" = shell ] || fail "tmux liveness must still read bash as a shell, got '$got'" + pass "bin/fm-agent-process-lib.sh: .cline is an agent, fragments are not" +} + +test_cline_ancestry_detects_the_native_command_name +test_cline_ancestry_detects_the_node_wrapper +test_cline_ancestry_rejects_unrelated_mentions +test_cline_structural_ancestor_outranks_a_retained_marker +test_cline_control_mechanics_are_the_verified_ones +test_cline_wiring_paths_are_the_workspace_hooks +test_cline_busy_source_is_trusted +test_cline_delivery_signature_is_harness_scoped +test_cline_tmux_names_the_native_binary_an_agent diff --git a/tests/fm-cline-signals-live-e2e.test.sh b/tests/fm-cline-signals-live-e2e.test.sh new file mode 100755 index 00000000000..b56d5d294f0 --- /dev/null +++ b/tests/fm-cline-signals-live-e2e.test.sh @@ -0,0 +1,190 @@ +#!/usr/bin/env bash +# Live drift guard for the Cline CLI adapter's vendor-controlled surface: +# hook config-file discovery, the rendered busy token, interrupt, and exit. +# Opt-in because it submits real prompts on the captain's ClinePass seat and no +# echo provider exists for cline. +# +# Run after every cline upgrade and before trusting refreshed per-harness +# evidence (docs/verification/cline.md names this as the refresh command). +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +CLINE_BIN=$(command -v cline 2>/dev/null || true) +REAL_TMUX=$(command -v tmux 2>/dev/null || true) +LAB= +SOCKET="fm-cline-signals-$$" +SESSION=cline-signals +TARGET="$SESSION:cline" + +cleanup() { + [ -n "$REAL_TMUX" ] && "$REAL_TMUX" -L "$SOCKET" kill-server >/dev/null 2>&1 || true + [ -z "$LAB" ] || rm -rf -- "$LAB" +} + +fail() { + printf 'not ok - %s\n' "$1" >&2 + cleanup + exit 1 +} + +pass() { + printf 'ok - %s\n' "$1" +} + +fm_live_gate opt-in FM_CLINE_SIGNALS_LIVE cline tmux +[ -n "$CLINE_BIN" ] || fail "cline is not installed" +[ -n "$REAL_TMUX" ] || fail "tmux is not installed" +[ -s "$HOME/.cline/data/settings/providers.json" ] || fail "no cline credential store to stage" + +LAB=$(mktemp -d "${TMPDIR:-/tmp}/fm-cline-signals.XXXXXX") || fail "could not create the isolated cline lab" +trap cleanup EXIT +mkdir -p "$LAB/workspace" "$LAB/home" "$LAB/state" "$LAB/workspace/.cline/hooks" \ + || fail "could not create the isolated cline lab" +git -C "$LAB/workspace" init -q || fail "could not initialize the isolated cline workspace" +git -C "$LAB/workspace" config user.email "guard@local" || fail "could not configure the isolated cline workspace" +git -C "$LAB/workspace" config user.name "guard" || fail "could not configure the isolated cline workspace" +git -C "$LAB/workspace" commit -q --allow-empty -m init || fail "could not seed the isolated cline workspace" +WORKSPACE=$(cd "$LAB/workspace" && pwd -P) || fail "could not resolve the isolated cline workspace" + +# A throwaway HOME holding a copy of ~/.cline keeps every cline write - session +# state, splash dismissal, refreshed tokens - inside the lab and away from the +# operator's real store. +cp -R "$HOME/.cline" "$LAB/home/.cline" || fail "could not stage the throwaway cline credential copy" + +HOOK_LOG="$LAB/hooks.log" +for ev in TaskStart TaskComplete TaskCancel TaskError SessionShutdown; do + { + printf '%s\n' '#!/bin/sh' + printf 'printf "%%s\\n" %s >>"%s"\n' "$ev" "$HOOK_LOG" + } > "$WORKSPACE/.cline/hooks/$ev" || fail "could not write the $ev hook" + chmod +x "$WORKSPACE/.cline/hooks/$ev" || fail "could not arm the $ev hook" +done + +# shellcheck source=/dev/null +. "$ROOT/bin/fm-composer-lib.sh" + +"$REAL_TMUX" -L "$SOCKET" new-session -d -s "$SESSION" -n control -c "$WORKSPACE" \ + || fail "could not start the isolated tmux server" +"$REAL_TMUX" -L "$SOCKET" new-window -d -t "$SESSION:" -n cline -c "$WORKSPACE" \ + || fail "could not open the isolated cline window" + +capture() { + "$REAL_TMUX" -L "$SOCKET" capture-pane -p -t "$TARGET" -S -200 2>/dev/null || true +} + +"$REAL_TMUX" -L "$SOCKET" send-keys -t "$TARGET" -l \ + "HOME=\"$LAB/home\" $CLINE_BIN -i -c \"$WORKSPACE\" --auto-approve true -m cline-pass/deepseek-v4-flash" \ + || fail "could not type the cline launch line" +"$REAL_TMUX" -L "$SOCKET" send-keys -t "$TARGET" Enter \ + || fail "could not submit the cline launch line" + +# A fresh profile may render the one-time "Introducing Cline Desktop" splash, +# which consumes the first submitted line. Dismiss it and wait for the idle +# composer, the same shape bin/fm-spawn.sh's readiness gate uses. +ready=0 +for _ in $(seq 1 180); do + screen=$(capture) + case "$screen" in + *"Introducing Cline Desktop"*|*"Press Enter to open"*) + "$REAL_TMUX" -L "$SOCKET" send-keys -t "$TARGET" Escape >/dev/null 2>&1 || true + sleep 1 + continue + ;; + esac + if printf '%s\n' "$screen" | grep -qE 'Ask anything\.\.\.|What can I do for you\?'; then + ready=1 + break + fi + sleep 0.5 +done +[ "$ready" -eq 1 ] || fail "cline never reached an idle composer" +pass "cline reaches an idle composer after the splash gate" + +# Ask for a computed sum so the awaited token cannot false-positive on the +# echoed launch line. +"$REAL_TMUX" -L "$SOCKET" send-keys -t "$TARGET" -l \ + "Add 12345 and 67890 and reply with exactly the sum and nothing else" \ + || fail "could not type the cline prompt" +"$REAL_TMUX" -L "$SOCKET" send-keys -t "$TARGET" Enter \ + || fail "could not submit the cline prompt" + +busy_seen=0 +for _ in $(seq 1 120); do + screen=$(capture) + if printf '%s\n' "$screen" | grep -qE "$FM_DELIVERY_CLINE_BUSY_REGEX_DEFAULT"; then + busy_seen=1 + break + fi + sleep 0.5 +done +[ "$busy_seen" -eq 1 ] || fail "the pinned (esc to cancel) busy token never rendered" + +done_seen=0 +for _ in $(seq 1 240); do + screen=$(capture) + if printf '%s\n' "$screen" | grep -q '80235\|80,235'; then + done_seen=1 + break + fi + sleep 0.5 +done +[ "$done_seen" -eq 1 ] || fail "the awaited reply never rendered" + +# TaskStart/TaskComplete must have fired from the workspace hook directory. +# TaskComplete lands at turn end, which can trail the rendered reply slightly. +start_count=0 +complete_count=0 +for _ in $(seq 1 40); do + start_count=$(grep -c '^TaskStart$' "$HOOK_LOG" 2>/dev/null || true) + complete_count=$(grep -c '^TaskComplete$' "$HOOK_LOG" 2>/dev/null || true) + if [ "${start_count:-0}" -ge 1 ] && [ "${complete_count:-0}" -ge 1 ]; then break; fi + sleep 0.5 +done +[ "${start_count:-0}" -ge 1 ] || fail "the TaskStart hook never fired" +[ "${complete_count:-0}" -ge 1 ] || fail "the TaskComplete hook never fired" +pass "cline fires the workspace TaskStart/TaskComplete hooks around a turn" + +# Interrupt: a second, slow turn cancelled with a single Escape. +"$REAL_TMUX" -L "$SOCKET" send-keys -t "$TARGET" -l \ + "Run the shell command: sleep 6. Then reply SLOWDONE." \ + || fail "could not type the interrupt probe" +"$REAL_TMUX" -L "$SOCKET" send-keys -t "$TARGET" Enter \ + || fail "could not submit the interrupt probe" +sleep 4 +"$REAL_TMUX" -L "$SOCKET" send-keys -t "$TARGET" Escape \ + || fail "could not send Escape" +# Give the cancelled turn time to settle back to an idle composer before the +# exit command, so /exit is not queued behind a still-running tool call. +for _ in $(seq 1 60); do + screen=$(capture) + printf '%s\n' "$screen" | grep -qE "$FM_DELIVERY_CLINE_BUSY_REGEX_DEFAULT" || break + sleep 0.5 +done + +# /exit must return to a shell. Detect it structurally (the pane's foreground +# process becomes the shell) with cline's own summary as a fallback. If the +# first attempt is swallowed (the cancelled turn can still be settling), retry +# the exact exit command once. +exited=0 +for _ in 1 2 3; do + "$REAL_TMUX" -L "$SOCKET" send-keys -t "$TARGET" -l '/exit' \ + || fail "could not type /exit" + "$REAL_TMUX" -L "$SOCKET" send-keys -t "$TARGET" Enter \ + || fail "could not submit /exit" + for _ in $(seq 1 60); do + screen=$(capture) + fg=$("$REAL_TMUX" -L "$SOCKET" display-message -p -t "$TARGET" '#{pane_current_command}' 2>/dev/null || true) + case "$fg" in bash|zsh|sh|dash|ash|ksh) exited=1 ;; esac + printf '%s\n' "$screen" | grep -Fq 'Session Summary' && exited=1 + [ "$exited" -eq 1 ] && break + sleep 0.5 + done + [ "$exited" -eq 1 ] && break +done +[ "$exited" -eq 1 ] || fail "cline did not exit on /exit" +pass "cline interrupts on Escape and exits on /exit" + +printf 'ok - cline live signals guard passed\n' From a5c4c530c3feb96d53486a59c267f236713afa2c Mon Sep 17 00:00:00 2001 From: keenvc Date: Sat, 19 Sep 2026 06:09:08 +0000 Subject: [PATCH 02/35] feat(bin): add per-spawn Claude config-dir seat to fm-spawn.sh Two Claude subscriptions need to run concurrently across lanes without moving every claude spawn onto one account. --claude-config-dir picks the CLAUDE_CONFIG_DIR one claude spawn's pane resolves into: validated before any worktree or endpoint exists, recorded in the task's own meta, reused unchanged on --relaunch, and threaded through both the pre-launch trust registration and the launch's own environment so the two halves can never land in different stores. A spawn naming no seat is byte-identical to before. --- .../references/harness/claude.md | 3 + bin/fm-spawn.sh | 111 +++++++++++- tests/fm-control-relaunch.test.sh | 25 +++ tests/fm-spawn-dispatch-profile.test.sh | 158 ++++++++++++++++++ 4 files changed, 293 insertions(+), 4 deletions(-) diff --git a/.agents/skills/harness-adapters/references/harness/claude.md b/.agents/skills/harness-adapters/references/harness/claude.md index 1dfc448d077..212b74bba5d 100644 --- a/.agents/skills/harness-adapters/references/harness/claude.md +++ b/.agents/skills/harness-adapters/references/harness/claude.md @@ -13,6 +13,7 @@ Busy hooks verified 2026-07-28 on Claude Code 2.1.220. | Model | `--model `; discover through the interactive `/model` picker, with alias or full-name shape documented by `claude --help`. | | Effort | `--effort `, verified on 2.1.196. | | Permissions | `--dangerously-skip-permissions` by default, or `--permission-mode auto` when `config/claude-permission-mode` is `auto`; the `auto` shape verified on 2.1.269, and `../../../../../docs/configuration.md` "Claude permission mode" owns the file. | +| Config seat | `--claude-config-dir ` on `fm-spawn.sh` picks the `CLAUDE_CONFIG_DIR` this one claude spawn's pane resolves into, validated and recorded in the task's own meta and reused on relaunch, so two claude lanes can sit on different accounts at once; `fm-spawn.sh --help` owns the exact contract. | ## Workspace trust @@ -29,6 +30,8 @@ When the project entry instead already carries an explicit decline (`hasClaudeMd Both flags `false` is Claude Code's default entry for a project never asked, not a decline, and is treated like an absent flag: trust registers and the import dialog still renders. The why-two-entries mechanism and the consent-gating logic live in the script's own header comment, which is the one owner for that contract; the fact worth repeating here is that `../../../bin/fm-spawn.sh` refuses the spawn when the trust flag fails to land, rather than launching a worker that would wedge on that dialog. +A seated spawn (`--claude-config-dir`, see "Config seat" above) passes that same directory as `CLAUDE_CONFIG_DIR` to this registration call, so trust always lands in the exact store the launched process itself reads - never firstmate's own ambient store. + Never try to answer either dialog with a key. Firstmate's key plane carries only Enter, Escape, and C-c with no arrow navigation, so it cannot move a dialog's selection at all, and both dialogs render with the cursor on their declining option, which means a sent Enter ends the session instead of accepting. A visible trust dialog means pre-registration did not take effect (or the project entry already carries an explicit decline) - inspect the store and the spawn's error output rather than sending keys. diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index 9afb13960c2..d1e4d227531 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -47,6 +47,22 @@ # worktree is told once to return, and only a shell that will not go refuses. # --harness is the explicit per-spawn harness/profile adapter. The old # positional harness arg still works for back-compat. +# --claude-config-dir is claude-only: it selects the Claude Code +# config/credential store (CLAUDE_CONFIG_DIR) this one launch's pane resolves +# into, letting two claude lanes run concurrently under different accounts +# without moving every claude lane at once the way setting CLAUDE_CONFIG_DIR +# in firstmate's own environment would. Refused when the resolved harness is +# not claude. Validated before any worktree or endpoint is created: +# must resolve to an existing directory holding a Claude config store +# (.claude.json), or the spawn refuses naming rather than launching a +# worker that would wedge. The resolved directory feeds both +# bin/fm-claude-trust.sh's pre-registration and the launch's own +# CLAUDE_CONFIG_DIR, so the two halves can never land in different stores. +# Recorded in the task's own meta as claude_config_dir= (absent means the +# single-store default, byte-identical to before this flag existed); a +# --relaunch always reuses that recorded value and refuses a fresh +# --claude-config-dir, so a relaunch can never silently move a task to a +# different seat. # --model and --effort are concrete profile # axes chosen by firstmate at intake. They are only threaded into harnesses whose # installed CLIs were verified to support that axis; unsupported axes are omitted @@ -527,6 +543,7 @@ BACKEND_ARG= MODE= YOLO= TRACEPARENT_ARG= +CLAUDE_SEAT_ARG= HARNESS_SET=0 MODEL_SET=0 EFFORT_SET=0 @@ -534,6 +551,7 @@ BACKEND_SET=0 MODE_SET=0 YOLO_SET=0 TRACEPARENT_SET=0 +CLAUDE_SEAT_SET=0 RELAUNCH=0 POS=() want_value= @@ -574,6 +592,10 @@ for a in "$@"; do TRACEPARENT_ARG=$a TRACEPARENT_SET=1 ;; + claude-config-dir) + CLAUDE_SEAT_ARG=$a + CLAUDE_SEAT_SET=1 + ;; *) echo "error: internal parser state for --$want_value" >&2 exit 1 @@ -627,6 +649,11 @@ for a in "$@"; do TRACEPARENT_ARG=${a#--traceparent=} TRACEPARENT_SET=1 ;; + --claude-config-dir) want_value=claude-config-dir ;; + --claude-config-dir=*) + CLAUDE_SEAT_ARG=${a#--claude-config-dir=} + CLAUDE_SEAT_SET=1 + ;; *) POS+=("$a") ;; esac done @@ -662,6 +689,10 @@ done echo "error: --traceparent requires a non-empty value" >&2 exit 1 } +[ "$CLAUDE_SEAT_SET" -eq 0 ] || [ -n "$CLAUDE_SEAT_ARG" ] || { + echo "error: --claude-config-dir requires a non-empty value" >&2 + exit 1 +} # A parent-delivered carrier replaces this home's own resolution, so it is # refused unless it is a secondmate spawn carrying a strictly valid W3C value. # Nothing else may reach the pane's TRACEPARENT export. @@ -704,6 +735,10 @@ if [ "$RELAUNCH" -eq 1 ]; then echo "error: --relaunch reuses the task's recorded yolo posture; --yolo cannot override it" >&2 exit 1 } + [ "$CLAUDE_SEAT_SET" -eq 0 ] || { + echo "error: --relaunch reuses the task's recorded Claude config directory; --claude-config-dir cannot override it" >&2 + exit 1 + } else # Delivery contract (AGENTS.md section 7). A ship task's mode and yolo are # firstmate's per-task decision, so they are required and closed-set validated @@ -1303,6 +1338,7 @@ if [ "${#POS[@]}" -gt 0 ] && [ "${POS[0]}" != "$idpart" ] && case "$idpart" in * [ -z "$MODEL" ] || shared_args+=(--model "$MODEL") [ -z "$EFFORT" ] || shared_args+=(--effort "$EFFORT") [ -z "$BACKEND_ARG" ] || shared_args+=(--backend "$BACKEND_ARG") + [ -z "$CLAUDE_SEAT_ARG" ] || shared_args+=(--claude-config-dir "$CLAUDE_SEAT_ARG") # One delivery contract applies to every pair in a batch, exactly like the shared # harness. Each pair still re-validates it against its own brief, so a batch # spanning several modes is two invocations rather than a silent mixed dispatch. @@ -1462,6 +1498,7 @@ RAW_LAUNCH=0 # validation teardown uses, so a malformed, ambiguous, or foreign record # refuses here exactly as it refuses there. RELAUNCH_PRIOR_HARNESS= +RELAUNCH_PRIOR_CLAUDE_CONFIG_DIR= if [ "$RELAUNCH" -eq 1 ]; then [ "${#POS[@]}" -eq 1 ] || { echo "error: --relaunch takes the task id only; its project or home comes from the task's own record" >&2 @@ -1501,6 +1538,7 @@ if [ "$RELAUNCH" -eq 1 ]; then exit 1 } RELAUNCH_PRIOR_HARNESS=$(fm_meta_get "$RELAUNCH_META" harness) + RELAUNCH_PRIOR_CLAUDE_CONFIG_DIR=$(fm_meta_get "$RELAUNCH_META" claude_config_dir) KIND=$(fm_meta_get "$RELAUNCH_META" kind) [ -n "$KIND" ] || KIND=ship MODE=$(fm_meta_get "$RELAUNCH_META" mode) @@ -2056,6 +2094,48 @@ if [ "$HARNESS" = agy ]; then agy_model_validate "$AGY_BIN" "$MODEL" || exit 1 fi +# --claude-config-dir selects the Claude Code config/credential store +# (CLAUDE_CONFIG_DIR) this one launch's pane resolves into. Validated here, +# before any worktree or endpoint is created, so a bad directory refuses the +# spawn now rather than launching a worker that fails in a way the supervisor +# reads as a stuck agent. Only the directory's existence and shape are +# inspected - its contents are never read or printed - because the directory +# path is the whole interface this flag grants. +fm_claude_seat_validate() { # + local raw=$1 real + real=$(CDPATH='' cd -P -- "$raw" 2>/dev/null && pwd -P) || { + echo "error: --claude-config-dir '$raw' is not an accessible directory" >&2 + return 1 + } + [ -f "$real/.claude.json" ] || { + echo "error: --claude-config-dir '$real' holds no usable Claude configuration (no .claude.json found); log in once with CLAUDE_CONFIG_DIR='$real' claude, then retry" >&2 + return 1 + } + printf '%s\n' "$real" +} +# CLAUDE_SEAT_RECORD is what this task's meta remembers as its seat, carried +# forward on every relaunch regardless of that relaunch's own harness, so a +# temporary switch away from claude and back never loses the assignment. +# CLAUDE_SEAT_DIR is narrower: it is set only when THIS launch will actually +# be a claude launch, and is what feeds the trust pre-registration and launch +# prefix below. +CLAUDE_SEAT_RECORD= +CLAUDE_SEAT_DIR= +if [ "$RELAUNCH" -eq 1 ]; then + CLAUDE_SEAT_RECORD=$RELAUNCH_PRIOR_CLAUDE_CONFIG_DIR + if [ -n "$CLAUDE_SEAT_RECORD" ] && [ "$HARNESS" = claude ]; then + CLAUDE_SEAT_DIR=$(fm_claude_seat_validate "$CLAUDE_SEAT_RECORD") || exit 1 + CLAUDE_SEAT_RECORD=$CLAUDE_SEAT_DIR + fi +elif [ -n "$CLAUDE_SEAT_ARG" ]; then + [ "$HARNESS" = claude ] || { + echo "error: --claude-config-dir applies only to claude spawns; this spawn resolved harness '$HARNESS'" >&2 + exit 1 + } + CLAUDE_SEAT_DIR=$(fm_claude_seat_validate "$CLAUDE_SEAT_ARG") || exit 1 + CLAUDE_SEAT_RECORD=$CLAUDE_SEAT_DIR +fi + secondmate_registry_value() { secondmate_registry_field "$DATA/secondmates.md" "$1" "$2" } @@ -3714,7 +3794,19 @@ claude*) else spawn_trust_args=("$WT" "$PROJ_ABS") fi - if ! "$FM_ROOT/bin/fm-claude-trust.sh" "${spawn_trust_args[@]}" >/dev/null; then + # CLAUDE_SEAT_DIR (from --claude-config-dir) takes the trust registration to + # the exact store this launch's own CLAUDE_CONFIG_DIR prefix below also + # points at, never to firstmate's own ambient CLAUDE_CONFIG_DIR, so trust + # registration and the launched process can never land in different stores. + # Overridden only for this one command; firstmate's own ambient value (if + # any) is untouched for the rest of the spawn. + claude_trust_rc=0 + if [ -n "$CLAUDE_SEAT_DIR" ]; then + CLAUDE_CONFIG_DIR="$CLAUDE_SEAT_DIR" "$FM_ROOT/bin/fm-claude-trust.sh" "${spawn_trust_args[@]}" >/dev/null || claude_trust_rc=$? + else + "$FM_ROOT/bin/fm-claude-trust.sh" "${spawn_trust_args[@]}" >/dev/null || claude_trust_rc=$? + fi + if [ "$claude_trust_rc" -ne 0 ]; then echo "error: could not pre-register Claude workspace trust for $WT; refusing to launch a claude worker that would wedge on the trust dialog; inspect window $T" >&2 exit 1 fi @@ -4196,7 +4288,7 @@ SPAWN_META_PATH=$SPAWN_META_TMP preserve_relaunch_meta() { awk -F= ' BEGIN { - split("window endpoint_task_id worktree project harness kind mode yolo tasktmp model effort busy_gen spawn_gen traceparent backend herdr_session herdr_workspace_id herdr_tab_id herdr_pane_id zellij_session zellij_tab_id zellij_pane_id orca_worktree_id terminal cmux_workspace_id cmux_surface_id home projects control_relaunch_tx", keys, " ") + split("window endpoint_task_id worktree project harness kind mode yolo tasktmp model effort busy_gen spawn_gen traceparent backend herdr_session herdr_workspace_id herdr_tab_id herdr_pane_id zellij_session zellij_tab_id zellij_pane_id orca_worktree_id terminal cmux_workspace_id cmux_surface_id home projects control_relaunch_tx claude_config_dir", keys, " ") for (i in keys) owned[keys[i]] = 1 } !($1 in owned) @@ -4221,6 +4313,11 @@ preserve_relaunch_meta() { # default path's meta stays byte-identical (absent backend= means tmux; # data/fm-backend-design-d7's P1 compatibility contract). [ "$BACKEND" = tmux ] || echo "backend=$BACKEND" + # Absent claude_config_dir= means the single-store default, byte-identical + # to meta written before --claude-config-dir existed. Recorded whenever this + # task has a seat, regardless of this launch's own harness, so a relaunch + # that temporarily switches away from claude and back never loses it. + [ -z "$CLAUDE_SEAT_RECORD" ] || echo "claude_config_dir=$CLAUDE_SEAT_RECORD" if [ "$BACKEND" = herdr ]; then echo "herdr_session=$HERDR_SES" echo "herdr_workspace_id=$HERDR_WORKSPACE_ID" @@ -4389,8 +4486,14 @@ esac # Forward firstmate's own resolved store onto the claude launch so the crewmate # uses the same credential/config firstmate is authenticated with. Only when set; # an unset value is the single-store default and needs no prefix. -if [ "$HARNESS" = claude ] && [ -n "${CLAUDE_CONFIG_DIR:-}" ]; then - LAUNCH="CLAUDE_CONFIG_DIR=$(shell_quote "$CLAUDE_CONFIG_DIR") $LAUNCH" +# CLAUDE_SEAT_DIR (this task's own --claude-config-dir, validated above) takes +# priority over that ambient forwarding, so an explicitly seated task always +# lands in its own store even when firstmate's own ambient CLAUDE_CONFIG_DIR +# names a different one - this is the one place a per-lane seat differs from +# every other Claude lane without moving firstmate's own environment. +CLAUDE_LAUNCH_CONFIG_DIR=${CLAUDE_SEAT_DIR:-${CLAUDE_CONFIG_DIR:-}} +if [ "$HARNESS" = claude ] && [ -n "$CLAUDE_LAUNCH_CONFIG_DIR" ]; then + LAUNCH="CLAUDE_CONFIG_DIR=$(shell_quote "$CLAUDE_LAUNCH_CONFIG_DIR") $LAUNCH" fi if [ "$KIND" = secondmate ]; then sq_home=$(shell_quote "$PROJ_ABS") diff --git a/tests/fm-control-relaunch.test.sh b/tests/fm-control-relaunch.test.sh index f4cde27b896..905313f61bc 100755 --- a/tests/fm-control-relaunch.test.sh +++ b/tests/fm-control-relaunch.test.sh @@ -1620,9 +1620,33 @@ test_spawn_relaunch_refuses_contradicting_flags() { out=$(run_spawn "$dir" rl16 "$dir/proj" --relaunch); rc=$? expect_code 1 "$rc" "a project positional should be refused alongside --relaunch" assert_contains "$out" "takes the task id only" "the refusal should name the positional rule" + out=$(run_spawn "$dir" rl16 --relaunch --claude-config-dir "$dir/seat"); rc=$? + expect_code 1 "$rc" "--claude-config-dir should be refused alongside --relaunch" + assert_contains "$out" "recorded Claude config directory" "the refusal should name the recorded-seat rule" pass "fm-spawn --relaunch: every identity axis comes from the record, and a contradicting flag refuses" } +test_relaunch_preserves_the_recorded_claude_config_dir() { + local dir out rc seat recorded + dir=$(new_case seat-persist rl91) + add_ship_task "$dir" rl91 claude + seat="$dir/claude-seat-b" + mkdir -p "$seat" + printf '{}' > "$seat/.claude.json" + seat=$(cd "$seat" && pwd -P) + printf 'claude_config_dir=%s\n' "$seat" >> "$dir/home/state/rl91.meta" + printf 'zsh' > "$dir/fake/command" + + out=$(run_spawn "$dir" rl91 --relaunch); rc=$? + expect_code 0 "$rc" "a same-harness relaunch with a recorded seat should succeed"$'\n'"$out" + recorded=$(meta_field "$dir" rl91 claude_config_dir) + [ "$recorded" = "$seat" ] \ + || fail "the relaunch must keep the task's recorded Claude config directory, got '$recorded'" + assert_grep "CLAUDE_CONFIG_DIR='$seat'" "$dir/fake/literal" \ + "the replacement launch did not use the task's recorded seat" + pass "fm-spawn --relaunch: reuses the task's recorded Claude config directory for the replacement launch, never a fresh flag" +} + test_spawn_relaunch_refuses_an_unrecorded_task() { local dir out rc dir=$(new_case norecord rl17) @@ -1733,6 +1757,7 @@ test_spawn_relaunch_refuses_a_symlinked_task_record_before_inspection test_spawn_relaunch_keeps_its_early_meta_lock_continuous test_spawn_relaunch_refuses_a_pending_authoritative_close test_spawn_relaunch_refuses_contradicting_flags +test_relaunch_preserves_the_recorded_claude_config_dir test_spawn_relaunch_refuses_an_unrecorded_task test_spawn_relaunch_refuses_a_pane_outside_the_worktree test_relaunch_reverifies_an_already_in_flight_item_instead_of_rewriting_it diff --git a/tests/fm-spawn-dispatch-profile.test.sh b/tests/fm-spawn-dispatch-profile.test.sh index 8ab102e625d..9a0b3ddaca3 100755 --- a/tests/fm-spawn-dispatch-profile.test.sh +++ b/tests/fm-spawn-dispatch-profile.test.sh @@ -921,6 +921,157 @@ test_non_claude_harness_ignores_config_dir() { pass "non-claude harnesses do not receive the claude CLAUDE_CONFIG_DIR prefix" } +# --- --claude-config-dir (per-spawn seat) ----------------------------------- + +# A usable Claude config store, for --claude-config-dir validation to accept. +# Only the directory's existence and the presence of .claude.json are ever +# inspected by the code under test - never its content - so an empty object is +# enough to exercise every path. +make_claude_seat() { # + mkdir -p "$1" + printf '{}' > "$1/.claude.json" + (cd "$1" && pwd -P) +} + +test_claude_config_dir_flag_records_meta_and_launch() { + local rec id out status launch seat + id=profile-claude-seat-z24 + rec=$(make_spawn_case profile-claude-seat claude "$id") + read_case_record "$rec" + seat=$(make_claude_seat "$CASE_DIR/seat") + + out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR" --claude-config-dir "$seat") + status=$? + expect_code 0 "$status" "a seated claude spawn should succeed"$'\n'"$out" + assert_grep "claude_config_dir=$seat" "$HOME_DIR/state/$id.meta" \ + "meta did not record the task's own --claude-config-dir" + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "CLAUDE_CONFIG_DIR='$seat' env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI" \ + "the launch did not use the named seat's config directory" + pass "--claude-config-dir is recorded in the task's own meta and reaches the launched process" +} + +test_claude_config_dir_flag_overrides_firstmates_ambient_store() { + local rec id out status launch seat + id=profile-claude-seat-override-z25 + rec=$(make_spawn_case profile-claude-seat-override claude "$id") + read_case_record "$rec" + seat=$(make_claude_seat "$CASE_DIR/seat") + + # Firstmate's own ambient CLAUDE_CONFIG_DIR names a DIFFERENT store than the + # task's seat, exactly the "two accounts at once" scenario this flag exists + # for: the seat must win, never firstmate's own environment. + out=$(FM_TEST_CLAUDE_CONFIG_DIR="$CASE_DIR/firstmates-own-store" \ + run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR" --claude-config-dir "$seat") + status=$? + expect_code 0 "$status" "a seated claude spawn under a different ambient store should still succeed"$'\n'"$out" + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "CLAUDE_CONFIG_DIR='$seat'" \ + "the task's own seat did not win over firstmate's ambient CLAUDE_CONFIG_DIR" + assert_not_contains "$launch" "firstmates-own-store" \ + "the launch leaked firstmate's own ambient CLAUDE_CONFIG_DIR instead of the task's seat" + pass "--claude-config-dir takes priority over firstmate's own ambient CLAUDE_CONFIG_DIR" +} + +test_two_claude_spawns_resolve_to_different_config_dirs() { + local rec id1 out1 status1 launch1 seat_a + local proj2 wt2 id2 out2 status2 launch2 seat_b + id1=profile-claude-seat-a-z26 + rec=$(make_spawn_case profile-claude-seat-a claude "$id1") + read_case_record "$rec" + seat_a=$(make_claude_seat "$CASE_DIR/seat-a") + + out1=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id1" "$PROJ_DIR" --claude-config-dir "$seat_a") + status1=$? + expect_code 0 "$status1" "first seated claude spawn should succeed"$'\n'"$out1" + launch1=$(cat "$LAUNCH_LOG") + + # A second task, same firstmate home, its OWN worktree (fm_git_worktree + # cannot reuse PROJ_DIR - it registers an origin remote that would collide), + # and a different seat: this is the concurrent-lanes scenario the flag + # exists for, not two sequential reads of one shared value. + id2=profile-claude-seat-b-z27 + proj2="$CASE_DIR/project-b" + wt2="$CASE_DIR/wt-b" + fm_git_worktree "$proj2" "$wt2" "wt-profile-claude-seat-b" + fm_test_spawn_brief "$HOME_DIR" "$id2" + seat_b=$(make_claude_seat "$CASE_DIR/seat-b") + + out2=$(run_ship_spawn "$HOME_DIR" "$wt2" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id2" "$proj2" --claude-config-dir "$seat_b") + status2=$? + expect_code 0 "$status2" "second seated claude spawn should succeed"$'\n'"$out2" + launch2=$(cat "$LAUNCH_LOG") + + assert_contains "$launch1" "CLAUDE_CONFIG_DIR='$seat_a'" "the first task's launch did not use its own seat" + assert_contains "$launch2" "CLAUDE_CONFIG_DIR='$seat_b'" "the second task's launch did not use its own seat" + assert_not_contains "$launch1" "$seat_b" "the first task's launch leaked the second task's seat" + assert_not_contains "$launch2" "$seat_a" "the second task's launch leaked the first task's seat" + assert_grep "claude_config_dir=$seat_a" "$HOME_DIR/state/$id1.meta" "the first task's meta did not record its own seat" + assert_grep "claude_config_dir=$seat_b" "$HOME_DIR/state/$id2.meta" "the second task's meta did not record its own seat" + pass "two claude spawns in the same home with distinct --claude-config-dir values resolve to different config stores" +} + +test_claude_config_dir_missing_directory_refuses_before_endpoint_or_metadata() { + local rec id out status + id=profile-claude-seat-missing-z28 + rec=$(make_spawn_case profile-claude-seat-missing claude "$id") + read_case_record "$rec" + + out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR" --claude-config-dir "$CASE_DIR/no-such-seat") + status=$? + expect_code 1 "$status" "a nonexistent --claude-config-dir must refuse the spawn" + assert_contains "$out" "--claude-config-dir '$CASE_DIR/no-such-seat' is not an accessible directory" \ + "refusal must name the missing directory" + [ ! -s "$LAUNCH_LOG" ] || fail "an invalid seat must launch nothing (got: $(cat "$LAUNCH_LOG"))" + assert_absent "$HOME_DIR/state/$id.meta" "refusal must happen before meta is written" + pass "a nonexistent --claude-config-dir refuses before any endpoint or metadata" +} + +test_claude_config_dir_not_a_directory_refuses() { + local rec id out status + id=profile-claude-seat-notdir-z29 + rec=$(make_spawn_case profile-claude-seat-notdir claude "$id") + read_case_record "$rec" + : > "$CASE_DIR/seat-file" + + out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR" --claude-config-dir "$CASE_DIR/seat-file") + status=$? + expect_code 1 "$status" "a --claude-config-dir that is a file must refuse the spawn" + assert_contains "$out" "is not an accessible directory" "refusal must name the file as not an accessible directory" + assert_absent "$HOME_DIR/state/$id.meta" "refusal must happen before meta is written" + pass "a --claude-config-dir naming a plain file refuses before any endpoint or metadata" +} + +test_claude_config_dir_without_config_refuses() { + local rec id out status + id=profile-claude-seat-empty-z30 + rec=$(make_spawn_case profile-claude-seat-empty claude "$id") + read_case_record "$rec" + mkdir -p "$CASE_DIR/seat-empty" + + out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR" --claude-config-dir "$CASE_DIR/seat-empty") + status=$? + expect_code 1 "$status" "a --claude-config-dir with no .claude.json must refuse the spawn" + assert_contains "$out" "holds no usable Claude configuration" "refusal must name the missing configuration" + assert_absent "$HOME_DIR/state/$id.meta" "refusal must happen before meta is written" + pass "a --claude-config-dir with no Claude configuration refuses before any endpoint or metadata" +} + +test_claude_config_dir_refused_for_non_claude_harness() { + local rec id out status seat + id=profile-codex-seat-refused-z31 + rec=$(make_spawn_case profile-codex-seat-refused codex "$id") + read_case_record "$rec" + seat=$(make_claude_seat "$CASE_DIR/seat") + + out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR" --harness codex --claude-config-dir "$seat") + status=$? + expect_code 1 "$status" "--claude-config-dir on a non-claude spawn must refuse" + assert_contains "$out" "--claude-config-dir applies only to claude spawns" "refusal must name the claude-only rule" + assert_absent "$HOME_DIR/state/$id.meta" "refusal must happen before meta is written" + pass "--claude-config-dir is refused for a spawn that does not resolve to the claude harness" +} + # The captain's attribution policy lives in the `user` settings scope, which a # spawned worker's settings sources are not guaranteed to load. Every claude # launch must therefore carry the policy itself, or a spawned worker writes @@ -1459,6 +1610,13 @@ test_claude_permission_mode_auto_reaches_scout_launch test_claude_permission_mode_invalid_refuses_before_endpoint_or_metadata test_non_claude_harness_ignores_claude_permission_mode test_non_claude_harness_ignores_config_dir +test_claude_config_dir_flag_records_meta_and_launch +test_claude_config_dir_flag_overrides_firstmates_ambient_store +test_two_claude_spawns_resolve_to_different_config_dirs +test_claude_config_dir_missing_directory_refuses_before_endpoint_or_metadata +test_claude_config_dir_not_a_directory_refuses +test_claude_config_dir_without_config_refuses +test_claude_config_dir_refused_for_non_claude_harness test_claude_task_launch_carries_control_channel_authority test_claude_secondmate_launch_omits_task_control_channel_authority test_claude_crewmate_launch_carries_the_attribution_policy From 6c8f420761e71ee6addf0848d75f0bc2d5181470 Mon Sep 17 00:00:00 2001 From: keenvc Date: Sat, 19 Sep 2026 06:47:55 +0000 Subject: [PATCH 03/35] no-mistakes(review): clarify Claude seat refusals for bypass dialog and relaunch --- .omc/handoffs/last-session-end.md | 4 ++++ bin/fm-spawn.sh | 28 +++++++++++++++++++----- tests/fm-control-relaunch.test.sh | 29 +++++++++++++++++++++++++ tests/fm-spawn-dispatch-profile.test.sh | 16 ++++++++++++-- 4 files changed, 69 insertions(+), 8 deletions(-) create mode 100644 .omc/handoffs/last-session-end.md diff --git a/.omc/handoffs/last-session-end.md b/.omc/handoffs/last-session-end.md new file mode 100644 index 00000000000..42268e49e94 --- /dev/null +++ b/.omc/handoffs/last-session-end.md @@ -0,0 +1,4 @@ +# Session ended 2026-09-19T06:47:55Z +- session_id: 60651a2d-4c7d-43b3-9499-54e84b390225 +- reason: other +- transcript: /home/azureuser/.claude/projects/-home-azureuser--no-mistakes-worktrees-6f716179fc56-01M2W4HFN101VRC5WX0AKYB8G4/60651a2d-4c7d-43b3-9499-54e84b390225.jsonl diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index d1e4d227531..170066f5fe3 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -2101,14 +2101,27 @@ fi # reads as a stuck agent. Only the directory's existence and shape are # inspected - its contents are never read or printed - because the directory # path is the whole interface this flag grants. -fm_claude_seat_validate() { # - local raw=$1 real +# A store also has to have accepted the machine-scoped Bypass Permissions +# confirmation, which is separate from the trust dialog, is raised only by +# --dangerously-skip-permissions, and which a spawned pane cannot answer +# (docs/verification/runtime-backends.md). That acceptance leaves no record in +# .claude.json - unlike hasTrustDialogAccepted and the import flags that +# bin/fm-claude-trust.sh reads and writes, it is absent from the store both at +# the top level and under projects. - so it cannot be checked here, and +# the refusal below naming both interactive steps is the only guard against it. +# is how the refusals name the directory, because on a relaunch the +# seat comes from the task's record rather than from a flag the caller passed. +fm_claude_seat_validate() { # + local raw=$1 origin=$2 real real=$(CDPATH='' cd -P -- "$raw" 2>/dev/null && pwd -P) || { - echo "error: --claude-config-dir '$raw' is not an accessible directory" >&2 + echo "error: $origin '$raw' is not an accessible directory" >&2 return 1 } [ -f "$real/.claude.json" ] || { - echo "error: --claude-config-dir '$real' holds no usable Claude configuration (no .claude.json found); log in once with CLAUDE_CONFIG_DIR='$real' claude, then retry" >&2 + echo "error: $origin '$real' holds no usable Claude configuration (no .claude.json found)" >&2 + echo "hint: prepare that store in one interactive sitting - run CLAUDE_CONFIG_DIR='$real' claude --dangerously-skip-permissions once and accept everything it shows: first the login, then that store's own Bypass Permissions confirmation" >&2 + echo "hint: logging in alone is not enough - the Bypass Permissions confirmation is raised only by --dangerously-skip-permissions, and a spawned pane cannot answer it, so a seat that has never accepted it wedges the worker" >&2 + echo "hint: to leave that confirmation unaccepted for this seat, set config/claude-permission-mode to auto, which launches with --permission-mode auto and never asks for bypass mode" >&2 return 1 } printf '%s\n' "$real" @@ -2124,7 +2137,10 @@ CLAUDE_SEAT_DIR= if [ "$RELAUNCH" -eq 1 ]; then CLAUDE_SEAT_RECORD=$RELAUNCH_PRIOR_CLAUDE_CONFIG_DIR if [ -n "$CLAUDE_SEAT_RECORD" ] && [ "$HARNESS" = claude ]; then - CLAUDE_SEAT_DIR=$(fm_claude_seat_validate "$CLAUDE_SEAT_RECORD") || exit 1 + CLAUDE_SEAT_DIR=$(fm_claude_seat_validate "$CLAUDE_SEAT_RECORD" "this task's recorded Claude config directory") || { + echo "hint: the seat is the one recorded in this task's meta and --claude-config-dir cannot override it on a relaunch; restore that directory, or relaunch the task under a non-claude --harness" >&2 + exit 1 + } CLAUDE_SEAT_RECORD=$CLAUDE_SEAT_DIR fi elif [ -n "$CLAUDE_SEAT_ARG" ]; then @@ -2132,7 +2148,7 @@ elif [ -n "$CLAUDE_SEAT_ARG" ]; then echo "error: --claude-config-dir applies only to claude spawns; this spawn resolved harness '$HARNESS'" >&2 exit 1 } - CLAUDE_SEAT_DIR=$(fm_claude_seat_validate "$CLAUDE_SEAT_ARG") || exit 1 + CLAUDE_SEAT_DIR=$(fm_claude_seat_validate "$CLAUDE_SEAT_ARG" "--claude-config-dir") || exit 1 CLAUDE_SEAT_RECORD=$CLAUDE_SEAT_DIR fi diff --git a/tests/fm-control-relaunch.test.sh b/tests/fm-control-relaunch.test.sh index 905313f61bc..862ac32dfee 100755 --- a/tests/fm-control-relaunch.test.sh +++ b/tests/fm-control-relaunch.test.sh @@ -1647,6 +1647,34 @@ test_relaunch_preserves_the_recorded_claude_config_dir() { pass "fm-spawn --relaunch: reuses the task's recorded Claude config directory for the replacement launch, never a fresh flag" } +test_relaunch_refusal_names_the_recorded_seat_not_the_flag() { + local dir out rc seat + dir=$(new_case seat-gone rl92) + add_ship_task "$dir" rl92 claude + seat="$dir/claude-seat-gone" + mkdir -p "$seat" + seat=$(cd "$seat" && pwd -P) + printf 'claude_config_dir=%s\n' "$seat" >> "$dir/home/state/rl92.meta" + # The operator removed the seat after the task was spawned, which is what a + # stuck-crewmate relaunch runs into. A relaunch caller never passed + # --claude-config-dir and is forbidden from passing it, so the refusal must + # point at the task's record instead of at that flag. + rmdir "$seat" + printf 'zsh' > "$dir/fake/command" + + out=$(run_spawn "$dir" rl92 --relaunch); rc=$? + expect_code 1 "$rc" "a relaunch whose recorded seat is gone should refuse" + assert_contains "$out" "this task's recorded Claude config directory '$seat' is not an accessible directory" \ + "the relaunch refusal should name the task's recorded seat as the thing that is gone" + assert_contains "$out" "--claude-config-dir cannot override it on a relaunch" \ + "the relaunch refusal should say the flag is not the caller's fix" + assert_not_contains "$out" "error: --claude-config-dir" \ + "the relaunch refusal must not blame a flag the caller never passed" + [ ! -s "$dir/fake/literal" ] \ + || fail "an unusable recorded seat must launch nothing (got: $(cat "$dir/fake/literal"))" + pass "fm-spawn --relaunch: an unusable recorded seat refuses in the record's own terms, not the flag's" +} + test_spawn_relaunch_refuses_an_unrecorded_task() { local dir out rc dir=$(new_case norecord rl17) @@ -1758,6 +1786,7 @@ test_spawn_relaunch_keeps_its_early_meta_lock_continuous test_spawn_relaunch_refuses_a_pending_authoritative_close test_spawn_relaunch_refuses_contradicting_flags test_relaunch_preserves_the_recorded_claude_config_dir +test_relaunch_refusal_names_the_recorded_seat_not_the_flag test_spawn_relaunch_refuses_an_unrecorded_task test_spawn_relaunch_refuses_a_pane_outside_the_worktree test_relaunch_reverifies_an_already_in_flight_item_instead_of_rewriting_it diff --git a/tests/fm-spawn-dispatch-profile.test.sh b/tests/fm-spawn-dispatch-profile.test.sh index 9a0b3ddaca3..8e69c0f2c92 100755 --- a/tests/fm-spawn-dispatch-profile.test.sh +++ b/tests/fm-spawn-dispatch-profile.test.sh @@ -1052,9 +1052,21 @@ test_claude_config_dir_without_config_refuses() { out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR" --claude-config-dir "$CASE_DIR/seat-empty") status=$? expect_code 1 "$status" "a --claude-config-dir with no .claude.json must refuse the spawn" - assert_contains "$out" "holds no usable Claude configuration" "refusal must name the missing configuration" + assert_contains "$out" "--claude-config-dir '$CASE_DIR/seat-empty' holds no usable Claude configuration" \ + "refusal must name the flag the caller passed and the missing configuration" + # Preparing the store means two interactive steps, not just a login: the + # separate Bypass Permissions confirmation a spawned pane cannot answer is + # raised only under --dangerously-skip-permissions, so the recommended + # command must carry that flag and the refusal must say a login alone is + # not enough. + assert_contains "$out" "CLAUDE_CONFIG_DIR='$CASE_DIR/seat-empty' claude --dangerously-skip-permissions" \ + "refusal must recommend the one command that reaches both interactive steps" + assert_contains "$out" "Bypass Permissions confirmation" \ + "refusal must name the second interactive step, not just the login" + assert_contains "$out" "config/claude-permission-mode to auto" \ + "refusal must offer the launch mode that never meets that confirmation" assert_absent "$HOME_DIR/state/$id.meta" "refusal must happen before meta is written" - pass "a --claude-config-dir with no Claude configuration refuses before any endpoint or metadata" + pass "a --claude-config-dir with no Claude configuration refuses before any endpoint or metadata, naming both interactive steps" } test_claude_config_dir_refused_for_non_claude_harness() { From e0a21195ed46570da8682c9ad172cf87e770ee1a Mon Sep 17 00:00:00 2001 From: keenvc Date: Sat, 19 Sep 2026 07:05:24 +0000 Subject: [PATCH 04/35] no-mistakes(review): refuse Claude seat flag on remote secondmates; drop stray artifact --- .omc/handoffs/last-session-end.md | 2 +- bin/fm-spawn.sh | 8 ++++- tests/fm-spawn-dispatch-profile.test.sh | 42 +++++++++++++++++++++++++ 3 files changed, 50 insertions(+), 2 deletions(-) diff --git a/.omc/handoffs/last-session-end.md b/.omc/handoffs/last-session-end.md index 42268e49e94..9ce768a0a9f 100644 --- a/.omc/handoffs/last-session-end.md +++ b/.omc/handoffs/last-session-end.md @@ -1,4 +1,4 @@ -# Session ended 2026-09-19T06:47:55Z +# Session ended 2026-09-19T07:05:24Z - session_id: 60651a2d-4c7d-43b3-9499-54e84b390225 - reason: other - transcript: /home/azureuser/.claude/projects/-home-azureuser--no-mistakes-worktrees-6f716179fc56-01M2W4HFN101VRC5WX0AKYB8G4/60651a2d-4c7d-43b3-9499-54e84b390225.jsonl diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index 170066f5fe3..bb968c803a3 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -814,6 +814,12 @@ spawn_remote_secondmate() { fm_lock_release "$SPAWN_TASK_LOCK" || true return 3 fi + if [ -n "$CLAUDE_SEAT_ARG" ]; then + fm_lock_release "$registry_lock" || true + fm_lock_release "$SPAWN_TASK_LOCK" || true + echo "error: --claude-config-dir names a Claude config directory on this machine and is not supported for remote secondmates, whose launch happens on another host" >&2 + return 2 + fi host=$(secondmate_registry_field "$DATA/secondmates.md" "$id" host) root=$(secondmate_registry_field "$DATA/secondmates.md" "$id" root) home=$(secondmate_registry_field "$DATA/secondmates.md" "$id" home) @@ -2121,7 +2127,7 @@ fm_claude_seat_validate() { # echo "error: $origin '$real' holds no usable Claude configuration (no .claude.json found)" >&2 echo "hint: prepare that store in one interactive sitting - run CLAUDE_CONFIG_DIR='$real' claude --dangerously-skip-permissions once and accept everything it shows: first the login, then that store's own Bypass Permissions confirmation" >&2 echo "hint: logging in alone is not enough - the Bypass Permissions confirmation is raised only by --dangerously-skip-permissions, and a spawned pane cannot answer it, so a seat that has never accepted it wedges the worker" >&2 - echo "hint: to leave that confirmation unaccepted for this seat, set config/claude-permission-mode to auto, which launches with --permission-mode auto and never asks for bypass mode" >&2 + echo "hint: if that confirmation cannot be accepted, setting config/claude-permission-mode to auto launches with --permission-mode auto and never asks for bypass mode - but that setting is home-wide and moves every claude launch from this home, not this seat alone" >&2 return 1 } printf '%s\n' "$real" diff --git a/tests/fm-spawn-dispatch-profile.test.sh b/tests/fm-spawn-dispatch-profile.test.sh index 8e69c0f2c92..e3e5904a414 100755 --- a/tests/fm-spawn-dispatch-profile.test.sh +++ b/tests/fm-spawn-dispatch-profile.test.sh @@ -1065,6 +1065,10 @@ test_claude_config_dir_without_config_refuses() { "refusal must name the second interactive step, not just the login" assert_contains "$out" "config/claude-permission-mode to auto" \ "refusal must offer the launch mode that never meets that confirmation" + # That file is resolved once per home and applies to every claude launch + # from it, so the hint must not read as a per-seat escape hatch. + assert_contains "$out" "every claude launch from this home, not this seat alone" \ + "refusal must state that the permission-mode escape hatch is home-wide, not seat-scoped" assert_absent "$HOME_DIR/state/$id.meta" "refusal must happen before meta is written" pass "a --claude-config-dir with no Claude configuration refuses before any endpoint or metadata, naming both interactive steps" } @@ -1084,6 +1088,43 @@ test_claude_config_dir_refused_for_non_claude_harness() { pass "--claude-config-dir is refused for a spawn that does not resolve to the claude harness" } +# A remote secondmate launches on another host, where a config directory named +# on this machine means nothing. That route leaves fm-spawn before the seat is +# resolved, so without an early refusal the flag is accepted and dropped and +# the lane silently runs on firstmate's own account. +test_claude_config_dir_refused_for_a_remote_secondmate() { + local rec id out status seat ssh_log + id=profile-claude-seat-remote-z32 + rec=$(make_spawn_case profile-claude-seat-remote claude "$id") + read_case_record "$rec" + seat=$(make_claude_seat "$CASE_DIR/seat") + mkdir -p "$CASE_DIR/remote-home" "$CASE_DIR/remote-root" + printf -- '- %s - remote lane (host: remote-host; root: %s; home: %s; scope: remote work; projects: none; added 2026-09-19)\n' \ + "$id" "$CASE_DIR/remote-root" "$CASE_DIR/remote-home" > "$HOME_DIR/data/secondmates.md" + # The transport itself, so "no dispatch happened" is observable rather than + # inferred: any contact with the remote host would leave a line here. + ssh_log="$CASE_DIR/ssh.log" + cat > "$CASE_DIR/recording-ssh" <> '$ssh_log' +exit 0 +SH + chmod +x "$CASE_DIR/recording-ssh" + + export FM_SSH_BIN="$CASE_DIR/recording-ssh" + out=$(run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" --secondmate --claude-config-dir "$seat") + status=$? + unset FM_SSH_BIN + + [ "$status" -ne 0 ] || fail "--claude-config-dir on a remote secondmate must refuse the spawn"$'\n'"$out" + assert_contains "$out" "--claude-config-dir" "refusal must name the flag that cannot be honored" + assert_contains "$out" "remote secondmates" "refusal must name the route that cannot honor it" + assert_absent "$ssh_log" "the refusal must fire before any remote dispatch" + assert_absent "$HOME_DIR/state/$id.meta" "refusal must happen before meta is written" + [ ! -s "$LAUNCH_LOG" ] || fail "a refused remote seat must launch nothing (got: $(cat "$LAUNCH_LOG"))" + pass "--claude-config-dir is refused for a remote secondmate before any remote dispatch" +} + # The captain's attribution policy lives in the `user` settings scope, which a # spawned worker's settings sources are not guaranteed to load. Every claude # launch must therefore carry the policy itself, or a spawned worker writes @@ -1629,6 +1670,7 @@ test_claude_config_dir_missing_directory_refuses_before_endpoint_or_metadata test_claude_config_dir_not_a_directory_refuses test_claude_config_dir_without_config_refuses test_claude_config_dir_refused_for_non_claude_harness +test_claude_config_dir_refused_for_a_remote_secondmate test_claude_task_launch_carries_control_channel_authority test_claude_secondmate_launch_omits_task_control_channel_authority test_claude_crewmate_launch_carries_the_attribution_policy From ea88453485385e60a18b136e5e73b19774bab5b2 Mon Sep 17 00:00:00 2001 From: keenvc Date: Sat, 19 Sep 2026 07:18:56 +0000 Subject: [PATCH 05/35] no-mistakes(review): cite verified bypass-store probe; untrack stray handoff artifact --- .omc/handoffs/last-session-end.md | 2 +- bin/fm-spawn.sh | 17 +++++++++++------ 2 files changed, 12 insertions(+), 7 deletions(-) diff --git a/.omc/handoffs/last-session-end.md b/.omc/handoffs/last-session-end.md index 9ce768a0a9f..00425e450fe 100644 --- a/.omc/handoffs/last-session-end.md +++ b/.omc/handoffs/last-session-end.md @@ -1,4 +1,4 @@ -# Session ended 2026-09-19T07:05:24Z +# Session ended 2026-09-19T07:18:56Z - session_id: 60651a2d-4c7d-43b3-9499-54e84b390225 - reason: other - transcript: /home/azureuser/.claude/projects/-home-azureuser--no-mistakes-worktrees-6f716179fc56-01M2W4HFN101VRC5WX0AKYB8G4/60651a2d-4c7d-43b3-9499-54e84b390225.jsonl diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index bb968c803a3..5613fef4200 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -2109,12 +2109,17 @@ fi # path is the whole interface this flag grants. # A store also has to have accepted the machine-scoped Bypass Permissions # confirmation, which is separate from the trust dialog, is raised only by -# --dangerously-skip-permissions, and which a spawned pane cannot answer -# (docs/verification/runtime-backends.md). That acceptance leaves no record in -# .claude.json - unlike hasTrustDialogAccepted and the import flags that -# bin/fm-claude-trust.sh reads and writes, it is absent from the store both at -# the top level and under projects. - so it cannot be checked here, and -# the refusal below naming both interactive steps is the only guard against it. +# --dangerously-skip-permissions, and which a spawned pane cannot answer; +# docs/verification/runtime-backends.md:461 records that warning appearing +# against an isolated CLAUDE_CONFIG_DIR, and the section around it verifies the +# workspace-trust dialog only, not this one. No record of that acceptance was +# found in .claude.json: on Claude Code 2.1.278, a read of the live store's key +# NAMES (never values) found nothing bypass- or permission-shaped among its 80 +# top-level keys or across the keys of its 56 projects. entries, where +# hasTrustDialogAccepted and the import-consent flags bin/fm-claude-trust.sh +# reads and writes do live. Whether Claude Code records it somewhere else - a +# separate file, an OS keychain - was not checked. So nothing here can screen +# for it, and the refusal below names both interactive steps instead. # is how the refusals name the directory, because on a relaunch the # seat comes from the task's record rather than from a flag the caller passed. fm_claude_seat_validate() { # From 97606f2f26709e241c957b0745d94867b2b7b773 Mon Sep 17 00:00:00 2001 From: keenvc Date: Sat, 19 Sep 2026 07:26:10 +0000 Subject: [PATCH 06/35] no-mistakes(review): ignore .omc/ and untrack stray handoff artifact --- .gitignore | 1 + .omc/handoffs/last-session-end.md | 4 ---- 2 files changed, 1 insertion(+), 4 deletions(-) delete mode 100644 .omc/handoffs/last-session-end.md diff --git a/.gitignore b/.gitignore index 3eece43c35f..6d1af7f4e80 100644 --- a/.gitignore +++ b/.gitignore @@ -4,6 +4,7 @@ data/ scratchpad* .no-mistakes/ .lavish/ +.omc/ .fm-secondmate-home .fm-secondmate-parent .DS_Store diff --git a/.omc/handoffs/last-session-end.md b/.omc/handoffs/last-session-end.md deleted file mode 100644 index 00425e450fe..00000000000 --- a/.omc/handoffs/last-session-end.md +++ /dev/null @@ -1,4 +0,0 @@ -# Session ended 2026-09-19T07:18:56Z -- session_id: 60651a2d-4c7d-43b3-9499-54e84b390225 -- reason: other -- transcript: /home/azureuser/.claude/projects/-home-azureuser--no-mistakes-worktrees-6f716179fc56-01M2W4HFN101VRC5WX0AKYB8G4/60651a2d-4c7d-43b3-9499-54e84b390225.jsonl From 2f9894e4d359029df9e5b4301a52db538e4a465e Mon Sep 17 00:00:00 2001 From: keenvc Date: Sat, 19 Sep 2026 07:43:46 +0000 Subject: [PATCH 07/35] no-mistakes(review): warn on every accepted Claude seat, simplify refusal --- bin/fm-spawn.sh | 12 ++++----- tests/fm-control-relaunch.test.sh | 6 +++++ tests/fm-spawn-dispatch-profile.test.sh | 36 ++++++++++++------------- 3 files changed, 30 insertions(+), 24 deletions(-) diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index 5613fef4200..d742659507e 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -2119,8 +2119,10 @@ fi # hasTrustDialogAccepted and the import-consent flags bin/fm-claude-trust.sh # reads and writes do live. Whether Claude Code records it somewhere else - a # separate file, an OS keychain - was not checked. So nothing here can screen -# for it, and the refusal below names both interactive steps instead. -# is how the refusals name the directory, because on a relaunch the +# for it: the .claude.json check below is a cheap sanity gate that refuses an +# obviously empty directory, and every ACCEPTED seat carries a notice saying so, +# because a store that was only logged into passes this check and then wedges. +# is how the messages name the directory, because on a relaunch the # seat comes from the task's record rather than from a flag the caller passed. fm_claude_seat_validate() { # local raw=$1 origin=$2 real @@ -2129,12 +2131,10 @@ fm_claude_seat_validate() { # return 1 } [ -f "$real/.claude.json" ] || { - echo "error: $origin '$real' holds no usable Claude configuration (no .claude.json found)" >&2 - echo "hint: prepare that store in one interactive sitting - run CLAUDE_CONFIG_DIR='$real' claude --dangerously-skip-permissions once and accept everything it shows: first the login, then that store's own Bypass Permissions confirmation" >&2 - echo "hint: logging in alone is not enough - the Bypass Permissions confirmation is raised only by --dangerously-skip-permissions, and a spawned pane cannot answer it, so a seat that has never accepted it wedges the worker" >&2 - echo "hint: if that confirmation cannot be accepted, setting config/claude-permission-mode to auto launches with --permission-mode auto and never asks for bypass mode - but that setting is home-wide and moves every claude launch from this home, not this seat alone" >&2 + echo "error: $origin '$real' holds no Claude configuration at all (no .claude.json found); log that store in once with CLAUDE_CONFIG_DIR='$real' claude, then retry" >&2 return 1 } + echo "notice: $origin '$real' is taken as given - a .claude.json is present, which is all this check proves; the seat must also already have accepted that store's own machine-scoped Bypass Permissions confirmation, which only --dangerously-skip-permissions raises, which defaults to declining, and which firstmate cannot answer (its steering plane carries Enter, Escape and Ctrl-C alone), so prepare it once interactively with CLAUDE_CONFIG_DIR='$real' claude --dangerously-skip-permissions, or launch only under config/claude-permission-mode=auto, which never requests bypass mode and so never meets that dialog" >&2 printf '%s\n' "$real" } # CLAUDE_SEAT_RECORD is what this task's meta remembers as its seat, carried diff --git a/tests/fm-control-relaunch.test.sh b/tests/fm-control-relaunch.test.sh index 862ac32dfee..ab485141a64 100755 --- a/tests/fm-control-relaunch.test.sh +++ b/tests/fm-control-relaunch.test.sh @@ -1644,6 +1644,12 @@ test_relaunch_preserves_the_recorded_claude_config_dir() { || fail "the relaunch must keep the task's recorded Claude config directory, got '$recorded'" assert_grep "CLAUDE_CONFIG_DIR='$seat'" "$dir/fake/literal" \ "the replacement launch did not use the task's recorded seat" + # A relaunch reuses a seat nobody re-inspects, so it reaches the same wedge + # as a fresh spawn and has to carry the same warning. + assert_contains "$out" "notice: this task's recorded Claude config directory '$seat'" \ + "a reused recorded seat must carry the acceptance notice naming it" + assert_contains "$out" "Bypass Permissions confirmation" \ + "the acceptance notice must name the confirmation the reuse check cannot verify" pass "fm-spawn --relaunch: reuses the task's recorded Claude config directory for the replacement launch, never a fresh flag" } diff --git a/tests/fm-spawn-dispatch-profile.test.sh b/tests/fm-spawn-dispatch-profile.test.sh index e3e5904a414..88cc3d0ff1d 100755 --- a/tests/fm-spawn-dispatch-profile.test.sh +++ b/tests/fm-spawn-dispatch-profile.test.sh @@ -948,7 +948,16 @@ test_claude_config_dir_flag_records_meta_and_launch() { launch=$(cat "$LAUNCH_LOG") assert_contains "$launch" "CLAUDE_CONFIG_DIR='$seat' env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI" \ "the launch did not use the named seat's config directory" - pass "--claude-config-dir is recorded in the task's own meta and reaches the launched process" + # A seat that was only logged into passes the existence check and then wedges + # on the Bypass Permissions confirmation, so the warning has to reach the + # operator on the path that accepts a seat, not only on a refusal. + assert_contains "$out" "notice: --claude-config-dir '$seat'" \ + "an accepted seat must carry a notice naming it" + assert_contains "$out" "Bypass Permissions confirmation" \ + "the acceptance notice must name the confirmation this check cannot verify" + assert_contains "$out" "config/claude-permission-mode=auto" \ + "the acceptance notice must offer the launch mode that never meets that confirmation" + pass "--claude-config-dir is recorded in the task's own meta, reaches the launched process, and warns what acceptance does not prove" } test_claude_config_dir_flag_overrides_firstmates_ambient_store() { @@ -1052,25 +1061,16 @@ test_claude_config_dir_without_config_refuses() { out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR" --claude-config-dir "$CASE_DIR/seat-empty") status=$? expect_code 1 "$status" "a --claude-config-dir with no .claude.json must refuse the spawn" - assert_contains "$out" "--claude-config-dir '$CASE_DIR/seat-empty' holds no usable Claude configuration" \ + # This check proves one thing - that no configuration exists there at all - + # so the refusal says that and points at the login. What a present + # .claude.json still cannot prove is carried by the notice on the acceptance + # path instead, which every seated spawn reaches. + assert_contains "$out" "--claude-config-dir '$CASE_DIR/seat-empty' holds no Claude configuration at all" \ "refusal must name the flag the caller passed and the missing configuration" - # Preparing the store means two interactive steps, not just a login: the - # separate Bypass Permissions confirmation a spawned pane cannot answer is - # raised only under --dangerously-skip-permissions, so the recommended - # command must carry that flag and the refusal must say a login alone is - # not enough. - assert_contains "$out" "CLAUDE_CONFIG_DIR='$CASE_DIR/seat-empty' claude --dangerously-skip-permissions" \ - "refusal must recommend the one command that reaches both interactive steps" - assert_contains "$out" "Bypass Permissions confirmation" \ - "refusal must name the second interactive step, not just the login" - assert_contains "$out" "config/claude-permission-mode to auto" \ - "refusal must offer the launch mode that never meets that confirmation" - # That file is resolved once per home and applies to every claude launch - # from it, so the hint must not read as a per-seat escape hatch. - assert_contains "$out" "every claude launch from this home, not this seat alone" \ - "refusal must state that the permission-mode escape hatch is home-wide, not seat-scoped" + assert_contains "$out" "CLAUDE_CONFIG_DIR='$CASE_DIR/seat-empty' claude" \ + "refusal must point at logging that store in" assert_absent "$HOME_DIR/state/$id.meta" "refusal must happen before meta is written" - pass "a --claude-config-dir with no Claude configuration refuses before any endpoint or metadata, naming both interactive steps" + pass "a --claude-config-dir with no Claude configuration refuses before any endpoint or metadata" } test_claude_config_dir_refused_for_non_claude_harness() { From 2273cba61bf5171afbbdc37be547a986f0c02d3f Mon Sep 17 00:00:00 2001 From: keenvc Date: Sat, 19 Sep 2026 08:05:12 +0000 Subject: [PATCH 08/35] no-mistakes(review): move seat guidance to docs; mark seated quota candidates --- .../references/harness/claude.md | 11 ++++ .agents/skills/quota-array-dispatch/SKILL.md | 1 + bin/fm-quota-choose.sh | 60 +++++++++++++++---- bin/fm-spawn.sh | 28 ++++----- tests/fm-control-relaunch.test.sh | 6 -- tests/fm-quota-choose.test.sh | 41 +++++++++++++ tests/fm-spawn-dispatch-profile.test.sh | 20 ++----- 7 files changed, 119 insertions(+), 48 deletions(-) diff --git a/.agents/skills/harness-adapters/references/harness/claude.md b/.agents/skills/harness-adapters/references/harness/claude.md index 212b74bba5d..5427109b6ad 100644 --- a/.agents/skills/harness-adapters/references/harness/claude.md +++ b/.agents/skills/harness-adapters/references/harness/claude.md @@ -42,8 +42,19 @@ Never send Enter to that one either: it was observed rendering in the same shape Firstmate cannot move a selection with Enter, Escape, and C-c alone, so it cannot accept this dialog at all, and an operator accepts it once per machine instead. Inspect the pane to identify which dialog is on screen, and report it rather than answering it. A launch under `config/claude-permission-mode=auto` never meets the bypass confirmation, because it does not request bypass mode: on 2.1.269 `claude --permission-mode auto` reached the composer directly with the footer `⏵⏵ auto mode on (shift+tab to cycle)`, so a captain who refuses the bypass dialog selects `auto` there instead of accepting it. +That setting is fleet-wide rather than per seat - `config/claude-permission-mode` is read once per home and governs every claude launch from it, crewmates, scouts, secondmates, and relaunches alike - so choosing `auto` to avoid the dialog for one seat moves every claude lane in that home with it. The workspace-trust dialog is unaffected by the permission mode and still needs the pre-registration above. +### Preparing a config seat + +Preparing a seat for `--claude-config-dir` (see "Config seat" above) is two interactive steps, not one, and the operator does both by hand before the first seated spawn. +First, log the account in for that store: `CLAUDE_CONFIG_DIR= claude`. +Second, accept that store's own bypass-permissions confirmation, which the login alone never raises - only `--dangerously-skip-permissions` asks for it - so the one command that reaches both in a single sitting is `CLAUDE_CONFIG_DIR= claude --dangerously-skip-permissions`, accepting whatever it shows. +A seat that was only logged into is the trap: it holds a `.claude.json`, so `fm-spawn.sh`'s seat check accepts it, and the pane then wedges on the bypass dialog that firstmate cannot answer, which the supervisor reads as a stuck agent. +`../../../../../docs/verification/runtime-backends.md` records that exact counter-case: an isolated `CLAUDE_CONFIG_DIR` holding only a copied `.claude.json` cleared the trust dialog and then surfaced the bypass warning. +A captain who will not accept that dialog for a seat runs the whole home under `config/claude-permission-mode=auto` instead, with the fleet-wide consequence stated above; there is no per-seat way to decline it. +`fm-spawn.sh` cannot verify either step: no record of the bypass acceptance was found in `.claude.json` on 2.1.278 (key names only, top level and `projects.`, where `hasTrustDialogAccepted` and the import-consent flags do live), and whether Claude Code records it elsewhere was not checked, so the seat check confirms only that a store exists at that path. + ## Composer ghost Completed turns can render dim predicted text inside an empty composer, indistinguishable in plain `tmux capture-pane`. diff --git a/.agents/skills/quota-array-dispatch/SKILL.md b/.agents/skills/quota-array-dispatch/SKILL.md index 4b988f1baab..75c015a7115 100644 --- a/.agents/skills/quota-array-dispatch/SKILL.md +++ b/.agents/skills/quota-array-dispatch/SKILL.md @@ -29,6 +29,7 @@ An `exhausted_now` runway vetoes the candidate. The helper selects a candidate only when its applicable quota has a known `effectivePercentRemaining` greater than zero. This is an optional narrow helper with a known limitation: it maps each harness to one primary provider family only, so a candidate whose established provider differs from that primary family is checked against the wrong quota row. omp has no primary family, so the helper keys an `omp:` candidate on its model prefix, mapping only `openai-codex/` and `claude-bridge/` and refusing every other prefix; the helper's header owns that mapping. +A candidate that will run on a non-default config seat carries a seat marker so it is not judged against the default account's row; the helper's header and `--help` own its exact syntax and what it does and does not claim. Authoritative multi-provider routing - including provider discovery from the harness catalog and quota matching by that explicit provider - stays owned by this skill's intake procedure above and AGENTS.md section 4, not by the helper. Use it only when the brief already fixed the candidate order and every candidate's provider is the harness's primary family. It does not replace the reasoning-class, runway-feasibility, or authentication gates above. diff --git a/bin/fm-quota-choose.sh b/bin/fm-quota-choose.sh index 4bfe89247bf..a64f65d85bd 100755 --- a/bin/fm-quota-choose.sh +++ b/bin/fm-quota-choose.sh @@ -2,7 +2,7 @@ # Choose the first quota-eligible candidate from a ranked list. # # Usage: -# fm-quota-choose.sh [--snapshot ] [--candidate ]... +# fm-quota-choose.sh [--snapshot ] [--candidate [@seated]]... # # Reads one already-captured quota-axi default TOON or JSON snapshot from the # provided file, or from stdin when --snapshot is omitted. For each --candidate @@ -36,6 +36,26 @@ # not by this helper. Use this helper only when the brief already fixed the # candidate order and every candidate's provider is the harness's primary family. # +# Seat exception: quota-axi measures one account per provider family, so a lane +# running on a non-default config seat (bin/fm-spawn.sh's --claude-config-dir, +# today the only seat firstmate has) spends an account this snapshot never +# describes. Judging such a candidate by the family's single row is wrong in +# both directions - it can veto a seat with full headroom, or clear a seat that +# is exhausted. Mark that candidate with a trailing `@seated` +# (`claude:default@seated`): the harness and model parse exactly as they do +# without the marker, the suffix must be exactly `@seated` or the candidate is +# refused, and an unmarked candidate keeps today's behavior byte for byte. +# A seated candidate is never measured against the family row at all. Its quota +# is reported as honestly unknown rather than as fabricated per-seat data, and +# AGENTS.md section 4's rule for exactly this case applies - disclosed +# uncertainty keeps a candidate eligible, and only concrete contradictory +# evidence blocks it - so a seated candidate is selected when candidate order +# reaches it, and a notice on stderr says its headroom was never verified. +# There is no per-seat quota evidence anywhere in quota-axi's schema today, so +# nothing can contradict a seated candidate; that is a disclosure, not a claim +# of headroom. The marker is not harness-restricted in code, but claude is the +# only harness with a config-dir seat, so it is the only practical caller. +# # omp (Oh My Pi) has no single primary family, so its candidate model prefix # selects the family: openai-codex/ checks the codex row and # claude-bridge/ checks the claude row, each against the bare for @@ -93,11 +113,24 @@ done [ "${#CANDIDATES[@]}" -gt 0 ] || die "no candidates supplied" -# A candidate is :. A bare harness with no colon means the -# default model. Reject empty harnesses and characters that cannot form a safe -# token. A colon-separated model is legal (e.g. model:codex_bengalfox). +# A candidate is :, optionally marked `@seated`. A bare harness +# with no colon means the default model. Reject empty harnesses and characters +# that cannot form a safe token. A colon-separated model is legal (e.g. +# model:codex_bengalfox). The marker is stripped before the token is validated, +# so `@` remains illegal everywhere else and a suffix that is not exactly +# `@seated` refuses rather than being read as part of the model. +candidate_is_seated() { case "$1" in *@seated) return 0 ;; *) return 1 ;; esac; } +candidate_token() { printf '%s\n' "${1%@seated}"; } + for c in "${CANDIDATES[@]}"; do + token=$c case "$c" in + *@*) + candidate_is_seated "$c" || die "invalid candidate: $c" + token=$(candidate_token "$c") + ;; + esac + case "$token" in ''|:*|*[!A-Za-z0-9._/:-]*) die "invalid candidate: $c" ;; esac done @@ -347,9 +380,10 @@ effective_for_provider_model() { } for c in "${CANDIDATES[@]}"; do - harness=${c%%:*} - model=${c#*:} - [ "$model" = "$c" ] && model="default" + token=$(candidate_token "$c") + harness=${token%%:*} + model=${token#*:} + [ "$model" = "$token" ] && model="default" [ -n "$model" ] || die "invalid candidate: $c" fm_control_harness_supported "$harness" || die "unknown harness: $harness" provider_for_harness "$harness" "$model" >/dev/null || case "$harness" in @@ -360,9 +394,15 @@ done chosen="none" for c in "${CANDIDATES[@]}"; do - harness=${c%%:*} - model=${c#*:} - [ "$model" = "$c" ] && model="default" + token=$(candidate_token "$c") + harness=${token%%:*} + model=${token#*:} + [ "$model" = "$token" ] && model="default" + if candidate_is_seated "$c"; then + echo "notice: candidate '$c' runs on a config seat, whose own account has no row in this quota snapshot; its headroom was not verified and it is treated as eligible with unknown quota" >&2 + chosen="$harness $model" + break + fi provider=$(provider_for_harness "$harness" "$model") scope_model=$model [ "$harness" != omp ] || scope_model=${model#*/} diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index d742659507e..a53db16083d 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -2107,21 +2107,16 @@ fi # reads as a stuck agent. Only the directory's existence and shape are # inspected - its contents are never read or printed - because the directory # path is the whole interface this flag grants. -# A store also has to have accepted the machine-scoped Bypass Permissions -# confirmation, which is separate from the trust dialog, is raised only by -# --dangerously-skip-permissions, and which a spawned pane cannot answer; -# docs/verification/runtime-backends.md:461 records that warning appearing -# against an isolated CLAUDE_CONFIG_DIR, and the section around it verifies the -# workspace-trust dialog only, not this one. No record of that acceptance was -# found in .claude.json: on Claude Code 2.1.278, a read of the live store's key -# NAMES (never values) found nothing bypass- or permission-shaped among its 80 -# top-level keys or across the keys of its 56 projects. entries, where -# hasTrustDialogAccepted and the import-consent flags bin/fm-claude-trust.sh -# reads and writes do live. Whether Claude Code records it somewhere else - a -# separate file, an OS keychain - was not checked. So nothing here can screen -# for it: the .claude.json check below is a cheap sanity gate that refuses an -# obviously empty directory, and every ACCEPTED seat carries a notice saying so, -# because a store that was only logged into passes this check and then wedges. +# The check is deliberately shallow, and says only what it can: a present +# .claude.json proves a store exists there, never that it is logged in or that +# it has accepted claude's once-per-machine bypass-permissions confirmation. +# No record of that acceptance was found in .claude.json (Claude Code 2.1.278, +# key names only, top level and projects., where hasTrustDialogAccepted +# and the import-consent flags bin/fm-claude-trust.sh handles do live), and +# whether Claude Code records it elsewhere was not checked, so nothing here can +# screen for it. What preparing a seat actually requires is owned once by +# .agents/skills/harness-adapters/references/harness/claude.md, which this +# validator points at rather than restating at spawn time. # is how the messages name the directory, because on a relaunch the # seat comes from the task's record rather than from a flag the caller passed. fm_claude_seat_validate() { # @@ -2131,10 +2126,9 @@ fm_claude_seat_validate() { # return 1 } [ -f "$real/.claude.json" ] || { - echo "error: $origin '$real' holds no Claude configuration at all (no .claude.json found); log that store in once with CLAUDE_CONFIG_DIR='$real' claude, then retry" >&2 + echo "error: $origin '$real' holds no Claude configuration at all (no .claude.json found); .agents/skills/harness-adapters/references/harness/claude.md under 'Workspace trust' owns what preparing a seat requires" >&2 return 1 } - echo "notice: $origin '$real' is taken as given - a .claude.json is present, which is all this check proves; the seat must also already have accepted that store's own machine-scoped Bypass Permissions confirmation, which only --dangerously-skip-permissions raises, which defaults to declining, and which firstmate cannot answer (its steering plane carries Enter, Escape and Ctrl-C alone), so prepare it once interactively with CLAUDE_CONFIG_DIR='$real' claude --dangerously-skip-permissions, or launch only under config/claude-permission-mode=auto, which never requests bypass mode and so never meets that dialog" >&2 printf '%s\n' "$real" } # CLAUDE_SEAT_RECORD is what this task's meta remembers as its seat, carried diff --git a/tests/fm-control-relaunch.test.sh b/tests/fm-control-relaunch.test.sh index ab485141a64..862ac32dfee 100755 --- a/tests/fm-control-relaunch.test.sh +++ b/tests/fm-control-relaunch.test.sh @@ -1644,12 +1644,6 @@ test_relaunch_preserves_the_recorded_claude_config_dir() { || fail "the relaunch must keep the task's recorded Claude config directory, got '$recorded'" assert_grep "CLAUDE_CONFIG_DIR='$seat'" "$dir/fake/literal" \ "the replacement launch did not use the task's recorded seat" - # A relaunch reuses a seat nobody re-inspects, so it reaches the same wedge - # as a fresh spawn and has to carry the same warning. - assert_contains "$out" "notice: this task's recorded Claude config directory '$seat'" \ - "a reused recorded seat must carry the acceptance notice naming it" - assert_contains "$out" "Bypass Permissions confirmation" \ - "the acceptance notice must name the confirmation the reuse check cannot verify" pass "fm-spawn --relaunch: reuses the task's recorded Claude config directory for the replacement launch, never a fresh flag" } diff --git a/tests/fm-quota-choose.test.sh b/tests/fm-quota-choose.test.sh index 095e292365f..157821b90b5 100755 --- a/tests/fm-quota-choose.test.sh +++ b/tests/fm-quota-choose.test.sh @@ -259,6 +259,47 @@ fi [ "$err" = "error: invalid candidate: claude:" ] || fail "trailing empty model returned: $err" ok "all candidates are validated before selection" +# A config-seated candidate spends an account this snapshot never measures, so +# the family's single row must not judge it. The fixture's claude rows are a +# barely-positive all_models scope and an exhausted model:fable scope, so +# claude:model:fable is the candidate the default account's own row rejects. +if out=$(call_choose --snapshot "$LAB/captured.json" --candidate claude:model:fable 2>/dev/null); then + fail "seat baseline: an unseated exhausted claude candidate was selected, got '$out'" +fi +[ "$out" = "none" ] || fail "seat baseline: expected 'none' for the unseated candidate, got '$out'" +out=$(call_choose --snapshot "$LAB/captured.json" --candidate claude:model:fable@seated) +[ "$out" = "claude model:fable" ] || fail "seated candidate: expected the default account's exhausted row to be ignored, got '$out'" +ok "a seated candidate is not judged by the default account's quota row" + +# The seat exception removes the wrong-account penalty; it does not promote a +# seated candidate over an earlier eligible one. +out=$(call_choose --snapshot "$LAB/captured.json" --candidate pi:default --candidate claude:model:fable@seated) +[ "$out" = "pi default" ] || fail "seat ordering: an earlier eligible candidate should still win, got '$out'" +out=$(call_choose --snapshot "$LAB/captured.json" --candidate kimi:default --candidate claude:model:fable@seated) +[ "$out" = "claude model:fable" ] || fail "seat ordering: a seated candidate should be reached past an exhausted one, got '$out'" +ok "candidate order still governs a seated candidate" + +# Unknown headroom must be disclosed, never dressed up as measured quota. +seat_err=$(QUOTA_AXI_CALLS="$LAB/seat-calls" QUOTA_AXI_FIXTURE="$FIXTURE" PATH="$FAKEBIN:$PATH" \ + "$BIN/fm-quota-choose.sh" --snapshot "$LAB/captured.json" --candidate claude:model:fable@seated 2>&1 >/dev/null) +printf '%s\n' "$seat_err" | grep -Fq "claude:model:fable@seated" \ + || fail "seat notice did not name the chosen candidate: $seat_err" +printf '%s\n' "$seat_err" | grep -Fq "headroom was not verified" \ + || fail "seat notice did not disclose the unverified headroom: $seat_err" +unseated_err=$(QUOTA_AXI_CALLS="$LAB/seat-calls" QUOTA_AXI_FIXTURE="$FIXTURE" PATH="$FAKEBIN:$PATH" \ + "$BIN/fm-quota-choose.sh" --snapshot "$LAB/captured.json" --candidate pi:default 2>&1 >/dev/null) +[ -z "$unseated_err" ] || fail "an unseated selection should stay silent on stderr: $unseated_err" +ok "choosing a seated candidate discloses its unverified headroom on stderr" + +# The marker is an exact token, not a free-form suffix. +for bad in 'claude:default@seat' 'claude:default@' 'claude@seated:default' '@seated'; do + if err=$(call_choose --snapshot "$LAB/captured.json" --candidate "$bad" 2>&1); then + fail "malformed seat marker '$bad' unexpectedly selected a candidate" + fi + [ "$err" = "error: invalid candidate: $bad" ] || fail "malformed seat marker '$bad' returned: $err" +done +ok "a malformed seat marker fails closed with the invalid-candidate error" + printf '{"schemaVersion":5,"providers":{"provider":"claude","quotaSemantics":{"effectiveAvailability":[{"scope":"all_models","status":"known","effectivePercentRemaining":50,"runway":{"status":"through_reset"}}]}}}\n' > "$MALFORMED" if err=$(call_choose --snapshot "$MALFORMED" --candidate claude:default 2>&1); then fail "malformed provider collection unexpectedly dispatched" diff --git a/tests/fm-spawn-dispatch-profile.test.sh b/tests/fm-spawn-dispatch-profile.test.sh index 88cc3d0ff1d..e982a54c7af 100755 --- a/tests/fm-spawn-dispatch-profile.test.sh +++ b/tests/fm-spawn-dispatch-profile.test.sh @@ -948,16 +948,7 @@ test_claude_config_dir_flag_records_meta_and_launch() { launch=$(cat "$LAUNCH_LOG") assert_contains "$launch" "CLAUDE_CONFIG_DIR='$seat' env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI" \ "the launch did not use the named seat's config directory" - # A seat that was only logged into passes the existence check and then wedges - # on the Bypass Permissions confirmation, so the warning has to reach the - # operator on the path that accepts a seat, not only on a refusal. - assert_contains "$out" "notice: --claude-config-dir '$seat'" \ - "an accepted seat must carry a notice naming it" - assert_contains "$out" "Bypass Permissions confirmation" \ - "the acceptance notice must name the confirmation this check cannot verify" - assert_contains "$out" "config/claude-permission-mode=auto" \ - "the acceptance notice must offer the launch mode that never meets that confirmation" - pass "--claude-config-dir is recorded in the task's own meta, reaches the launched process, and warns what acceptance does not prove" + pass "--claude-config-dir is recorded in the task's own meta and reaches the launched process" } test_claude_config_dir_flag_overrides_firstmates_ambient_store() { @@ -1062,13 +1053,12 @@ test_claude_config_dir_without_config_refuses() { status=$? expect_code 1 "$status" "a --claude-config-dir with no .claude.json must refuse the spawn" # This check proves one thing - that no configuration exists there at all - - # so the refusal says that and points at the login. What a present - # .claude.json still cannot prove is carried by the notice on the acceptance - # path instead, which every seated spawn reaches. + # so the refusal says that and names the document that owns what preparing a + # seat requires, rather than restating it at spawn time. assert_contains "$out" "--claude-config-dir '$CASE_DIR/seat-empty' holds no Claude configuration at all" \ "refusal must name the flag the caller passed and the missing configuration" - assert_contains "$out" "CLAUDE_CONFIG_DIR='$CASE_DIR/seat-empty' claude" \ - "refusal must point at logging that store in" + assert_contains "$out" "harness-adapters/references/harness/claude.md" \ + "refusal must point at the document that owns seat preparation" assert_absent "$HOME_DIR/state/$id.meta" "refusal must happen before meta is written" pass "a --claude-config-dir with no Claude configuration refuses before any endpoint or metadata" } From fb042772aa0e41109296aeb753254a8025351d4b Mon Sep 17 00:00:00 2001 From: keenvc Date: Sat, 19 Sep 2026 08:20:43 +0000 Subject: [PATCH 09/35] no-mistakes(review): judge seated candidates through shared gate; drop gitignore line --- .gitignore | 1 - .omc/handoffs/last-session-end.md | 4 +++ bin/fm-quota-choose.sh | 44 ++++++++++++++++++------------- tests/fm-quota-choose.test.sh | 22 +++++++++++++++- 4 files changed, 50 insertions(+), 21 deletions(-) create mode 100644 .omc/handoffs/last-session-end.md diff --git a/.gitignore b/.gitignore index 6d1af7f4e80..3eece43c35f 100644 --- a/.gitignore +++ b/.gitignore @@ -4,7 +4,6 @@ data/ scratchpad* .no-mistakes/ .lavish/ -.omc/ .fm-secondmate-home .fm-secondmate-parent .DS_Store diff --git a/.omc/handoffs/last-session-end.md b/.omc/handoffs/last-session-end.md new file mode 100644 index 00000000000..b2376877fe7 --- /dev/null +++ b/.omc/handoffs/last-session-end.md @@ -0,0 +1,4 @@ +# Session ended 2026-09-19T08:20:42Z +- session_id: 60651a2d-4c7d-43b3-9499-54e84b390225 +- reason: other +- transcript: /home/azureuser/.claude/projects/-home-azureuser--no-mistakes-worktrees-6f716179fc56-01M2W4HFN101VRC5WX0AKYB8G4/60651a2d-4c7d-43b3-9499-54e84b390225.jsonl diff --git a/bin/fm-quota-choose.sh b/bin/fm-quota-choose.sh index a64f65d85bd..f3a1bbb9177 100755 --- a/bin/fm-quota-choose.sh +++ b/bin/fm-quota-choose.sh @@ -45,16 +45,20 @@ # (`claude:default@seated`): the harness and model parse exactly as they do # without the marker, the suffix must be exactly `@seated` or the candidate is # refused, and an unmarked candidate keeps today's behavior byte for byte. -# A seated candidate is never measured against the family row at all. Its quota -# is reported as honestly unknown rather than as fabricated per-seat data, and -# AGENTS.md section 4's rule for exactly this case applies - disclosed -# uncertainty keeps a candidate eligible, and only concrete contradictory -# evidence blocks it - so a seated candidate is selected when candidate order -# reaches it, and a notice on stderr says its headroom was never verified. -# There is no per-seat quota evidence anywhere in quota-axi's schema today, so -# nothing can contradict a seated candidate; that is a disclosure, not a claim -# of headroom. The marker is not harness-restricted in code, but claude is the -# only harness with a config-dir seat, so it is the only practical caller. +# A seated candidate is never measured against the family row: that lookup is +# skipped and an honestly-unknown `{"status":"unknown"}` value stands in for the +# per-seat data quota-axi does not carry, rather than a fabricated figure. It +# then goes through the SAME eligibility decision as every other candidate, with +# only that decision's unknown branch reading the seat marker - an unknown +# status is eligible when seated and, exactly as before, ineligible when not. +# The exhausted_now veto and the known-positive-percentage path are identical +# for both, so per-seat evidence would block a seated candidate the day one +# exists. Accepting unknown for a seat is AGENTS.md section 4's rule for this +# case: disclosed uncertainty keeps a candidate eligible, only concrete +# contradictory evidence blocks it. A notice on stderr says its headroom was +# never verified, because that is a disclosure and not a claim of headroom. +# The marker is not harness-restricted in code, but claude is the only harness +# with a config-dir seat, so it is the only practical caller. # # omp (Oh My Pi) has no single primary family, so its candidate model prefix # selects the family: openai-codex/ checks the codex row and @@ -398,21 +402,22 @@ for c in "${CANDIDATES[@]}"; do harness=${token%%:*} model=${token#*:} [ "$model" = "$token" ] && model="default" + seated=false if candidate_is_seated "$c"; then - echo "notice: candidate '$c' runs on a config seat, whose own account has no row in this quota snapshot; its headroom was not verified and it is treated as eligible with unknown quota" >&2 - chosen="$harness $model" - break + seated=true + effective='{"status":"unknown"}' + else + provider=$(provider_for_harness "$harness" "$model") + scope_model=$model + [ "$harness" != omp ] || scope_model=${model#*/} + effective=$(effective_for_provider_model "$provider" "$scope_model") fi - provider=$(provider_for_harness "$harness" "$model") - scope_model=$model - [ "$harness" != omp ] || scope_model=${model#*/} - effective=$(effective_for_provider_model "$provider" "$scope_model") if [ -z "$effective" ] || [ "$effective" = "null" ]; then continue fi - if printf '%s\n' "$effective" | jq -e ' + if printf '%s\n' "$effective" | jq -e --argjson seated "$seated" ' if (.runway.status // "") == "exhausted_now" then false - elif .status == "unknown" then false + elif .status == "unknown" then $seated else .effectivePercentRemaining as $remaining | (($remaining | type) == "number") and @@ -420,6 +425,7 @@ for c in "${CANDIDATES[@]}"; do ((.runway.status // "") != "exhausted_now") end ' >/dev/null 2>&1; then + [ "$seated" = false ] || echo "notice: candidate '$c' runs on a config seat, whose own account has no row in this quota snapshot; its headroom was not verified and it is treated as eligible with unknown quota" >&2 chosen="$harness $model" break fi diff --git a/tests/fm-quota-choose.test.sh b/tests/fm-quota-choose.test.sh index 157821b90b5..368d07ae0b7 100755 --- a/tests/fm-quota-choose.test.sh +++ b/tests/fm-quota-choose.test.sh @@ -260,7 +260,9 @@ fi ok "all candidates are validated before selection" # A config-seated candidate spends an account this snapshot never measures, so -# the family's single row must not judge it. The fixture's claude rows are a +# the family's single row must not judge it; it carries an unknown quota value +# through the same eligibility decision every candidate goes through, where +# unknown is eligible only for a seat. The fixture's claude rows are a # barely-positive all_models scope and an exhausted model:fable scope, so # claude:model:fable is the candidate the default account's own row rejects. if out=$(call_choose --snapshot "$LAB/captured.json" --candidate claude:model:fable 2>/dev/null); then @@ -300,6 +302,24 @@ for bad in 'claude:default@seat' 'claude:default@' 'claude@seated:default' '@sea done ok "a malformed seat marker fails closed with the invalid-candidate error" +# The marker exempts a candidate from the default account's quota row, nothing +# else: every gate that does not depend on quota still refuses a seated +# candidate exactly as it refuses an unseated one. +if err=$(call_choose --snapshot "$LAB/captured.json" --candidate bogus:default@seated 2>&1); then + fail "a seated candidate on an unknown harness was selected" +fi +[ "$err" = "error: unknown harness: bogus" ] || fail "seated unknown harness returned: $err" +if err=$(call_choose --snapshot "$LAB/captured.json" --candidate pi:default --candidate 'claude:@seated' 2>&1); then + fail "a seated candidate with an empty model was hidden by an earlier selection" +fi +[ "$err" = "error: invalid candidate: claude:@seated" ] || fail "seated empty model returned: $err" +if err=$(call_choose --snapshot "$LAB/captured.json" --candidate omp:ollama/qwen3:8b@seated 2>&1); then + fail "a seated omp candidate with an unmapped prefix was selected" +fi +[ "$err" = "error: omp quota mapping covers only the openai-codex and claude-bridge prefixes: ollama/qwen3:8b" ] \ + || fail "seated unmapped omp prefix returned: $err" +ok "the seat marker exempts a candidate from the quota row only, never from the other gates" + printf '{"schemaVersion":5,"providers":{"provider":"claude","quotaSemantics":{"effectiveAvailability":[{"scope":"all_models","status":"known","effectivePercentRemaining":50,"runway":{"status":"through_reset"}}]}}}\n' > "$MALFORMED" if err=$(call_choose --snapshot "$MALFORMED" --candidate claude:default 2>&1); then fail "malformed provider collection unexpectedly dispatched" From bf0c3fdbd24ddb0ad0da5cd7176b508ffbc1dd08 Mon Sep 17 00:00:00 2001 From: keenvc Date: Sat, 19 Sep 2026 09:00:11 +0000 Subject: [PATCH 10/35] no-mistakes(review): restore .omc/ ignore rule and untrack handoff artifact --- .gitignore | 1 + .omc/handoffs/last-session-end.md | 4 ---- 2 files changed, 1 insertion(+), 4 deletions(-) delete mode 100644 .omc/handoffs/last-session-end.md diff --git a/.gitignore b/.gitignore index 3eece43c35f..6d1af7f4e80 100644 --- a/.gitignore +++ b/.gitignore @@ -4,6 +4,7 @@ data/ scratchpad* .no-mistakes/ .lavish/ +.omc/ .fm-secondmate-home .fm-secondmate-parent .DS_Store diff --git a/.omc/handoffs/last-session-end.md b/.omc/handoffs/last-session-end.md deleted file mode 100644 index b2376877fe7..00000000000 --- a/.omc/handoffs/last-session-end.md +++ /dev/null @@ -1,4 +0,0 @@ -# Session ended 2026-09-19T08:20:42Z -- session_id: 60651a2d-4c7d-43b3-9499-54e84b390225 -- reason: other -- transcript: /home/azureuser/.claude/projects/-home-azureuser--no-mistakes-worktrees-6f716179fc56-01M2W4HFN101VRC5WX0AKYB8G4/60651a2d-4c7d-43b3-9499-54e84b390225.jsonl From 8d2a85c4b00edc36decdd6c3b13e8fe1c9fc5b7c Mon Sep 17 00:00:00 2001 From: keenvc Date: Sat, 19 Sep 2026 09:08:43 +0000 Subject: [PATCH 11/35] no-mistakes(review): restrict @seated quota marker to the claude harness --- bin/fm-quota-choose.sh | 11 +++++++++-- tests/fm-quota-choose.test.sh | 18 ++++++++++++++++++ 2 files changed, 27 insertions(+), 2 deletions(-) diff --git a/bin/fm-quota-choose.sh b/bin/fm-quota-choose.sh index f3a1bbb9177..b91be14219c 100755 --- a/bin/fm-quota-choose.sh +++ b/bin/fm-quota-choose.sh @@ -57,8 +57,12 @@ # case: disclosed uncertainty keeps a candidate eligible, only concrete # contradictory evidence blocks it. A notice on stderr says its headroom was # never verified, because that is a disclosure and not a claim of headroom. -# The marker is not harness-restricted in code, but claude is the only harness -# with a config-dir seat, so it is the only practical caller. +# The marker is refused on every harness but claude, in the same validation +# loop that refuses a malformed token, because claude is the only harness with +# a config-dir seat (bin/fm-spawn.sh refuses --claude-config-dir for any other +# harness for the same reason). Without that restriction the marker would be an +# unrestricted quota-gate off-switch anywhere else, with no seat behind it to +# justify the unknown. # # omp (Oh My Pi) has no single primary family, so its candidate model prefix # selects the family: openai-codex/ checks the codex row and @@ -394,6 +398,9 @@ for c in "${CANDIDATES[@]}"; do omp) die "omp quota mapping covers only the openai-codex and claude-bridge prefixes: $model" ;; *) die "unknown harness: $harness" ;; esac + if candidate_is_seated "$c" && [ "$harness" != claude ]; then + die "invalid candidate: $c (@seated is only meaningful for the claude harness, the only one with a config seat)" + fi done chosen="none" diff --git a/tests/fm-quota-choose.test.sh b/tests/fm-quota-choose.test.sh index 368d07ae0b7..7c25b8c1a4f 100755 --- a/tests/fm-quota-choose.test.sh +++ b/tests/fm-quota-choose.test.sh @@ -320,6 +320,24 @@ fi || fail "seated unmapped omp prefix returned: $err" ok "the seat marker exempts a candidate from the quota row only, never from the other gates" +# Only claude has a config seat, so the marker anywhere else would be a quota +# off-switch with nothing behind it. The fixture's codex rows are exhausted at +# model:codex_bengalfox, so an accepted marker there would select a candidate +# the snapshot says is spent. +if err=$(call_choose --snapshot "$LAB/captured.json" --candidate codex:default@seated 2>&1); then + fail "a seated non-claude candidate was selected instead of refused" +fi +[ "$err" = "error: invalid candidate: codex:default@seated (@seated is only meaningful for the claude harness, the only one with a config seat)" ] \ + || fail "seated non-claude candidate returned: $err" +if err=$(call_choose --snapshot "$LAB/captured.json" --candidate pi:default --candidate codex:model:codex_bengalfox@seated 2>&1); then + fail "a trailing seated non-claude candidate was hidden by an earlier selection" +fi +[ "$err" = "error: invalid candidate: codex:model:codex_bengalfox@seated (@seated is only meaningful for the claude harness, the only one with a config seat)" ] \ + || fail "trailing seated non-claude candidate returned: $err" +out=$(call_choose --snapshot "$LAB/captured.json" --candidate claude:model:fable@seated) +[ "$out" = "claude model:fable" ] || fail "the claude harness must still accept the seat marker, got '$out'" +ok "the seat marker is refused on every harness but claude" + printf '{"schemaVersion":5,"providers":{"provider":"claude","quotaSemantics":{"effectiveAvailability":[{"scope":"all_models","status":"known","effectivePercentRemaining":50,"runway":{"status":"through_reset"}}]}}}\n' > "$MALFORMED" if err=$(call_choose --snapshot "$MALFORMED" --candidate claude:default 2>&1); then fail "malformed provider collection unexpectedly dispatched" From ce2a294e15b4984a7146b8c75a7726f04e2f7562 Mon Sep 17 00:00:00 2001 From: keenvc Date: Sat, 19 Sep 2026 09:20:46 +0000 Subject: [PATCH 12/35] no-mistakes(review): revert @seated quota marker; keep only the spawn seat --- .agents/skills/quota-array-dispatch/SKILL.md | 1 - bin/fm-quota-choose.sh | 85 ++++---------------- tests/fm-quota-choose.test.sh | 79 ------------------ 3 files changed, 16 insertions(+), 149 deletions(-) diff --git a/.agents/skills/quota-array-dispatch/SKILL.md b/.agents/skills/quota-array-dispatch/SKILL.md index 75c015a7115..4b988f1baab 100644 --- a/.agents/skills/quota-array-dispatch/SKILL.md +++ b/.agents/skills/quota-array-dispatch/SKILL.md @@ -29,7 +29,6 @@ An `exhausted_now` runway vetoes the candidate. The helper selects a candidate only when its applicable quota has a known `effectivePercentRemaining` greater than zero. This is an optional narrow helper with a known limitation: it maps each harness to one primary provider family only, so a candidate whose established provider differs from that primary family is checked against the wrong quota row. omp has no primary family, so the helper keys an `omp:` candidate on its model prefix, mapping only `openai-codex/` and `claude-bridge/` and refusing every other prefix; the helper's header owns that mapping. -A candidate that will run on a non-default config seat carries a seat marker so it is not judged against the default account's row; the helper's header and `--help` own its exact syntax and what it does and does not claim. Authoritative multi-provider routing - including provider discovery from the harness catalog and quota matching by that explicit provider - stays owned by this skill's intake procedure above and AGENTS.md section 4, not by the helper. Use it only when the brief already fixed the candidate order and every candidate's provider is the harness's primary family. It does not replace the reasoning-class, runway-feasibility, or authentication gates above. diff --git a/bin/fm-quota-choose.sh b/bin/fm-quota-choose.sh index b91be14219c..4bfe89247bf 100755 --- a/bin/fm-quota-choose.sh +++ b/bin/fm-quota-choose.sh @@ -2,7 +2,7 @@ # Choose the first quota-eligible candidate from a ranked list. # # Usage: -# fm-quota-choose.sh [--snapshot ] [--candidate [@seated]]... +# fm-quota-choose.sh [--snapshot ] [--candidate ]... # # Reads one already-captured quota-axi default TOON or JSON snapshot from the # provided file, or from stdin when --snapshot is omitted. For each --candidate @@ -36,34 +36,6 @@ # not by this helper. Use this helper only when the brief already fixed the # candidate order and every candidate's provider is the harness's primary family. # -# Seat exception: quota-axi measures one account per provider family, so a lane -# running on a non-default config seat (bin/fm-spawn.sh's --claude-config-dir, -# today the only seat firstmate has) spends an account this snapshot never -# describes. Judging such a candidate by the family's single row is wrong in -# both directions - it can veto a seat with full headroom, or clear a seat that -# is exhausted. Mark that candidate with a trailing `@seated` -# (`claude:default@seated`): the harness and model parse exactly as they do -# without the marker, the suffix must be exactly `@seated` or the candidate is -# refused, and an unmarked candidate keeps today's behavior byte for byte. -# A seated candidate is never measured against the family row: that lookup is -# skipped and an honestly-unknown `{"status":"unknown"}` value stands in for the -# per-seat data quota-axi does not carry, rather than a fabricated figure. It -# then goes through the SAME eligibility decision as every other candidate, with -# only that decision's unknown branch reading the seat marker - an unknown -# status is eligible when seated and, exactly as before, ineligible when not. -# The exhausted_now veto and the known-positive-percentage path are identical -# for both, so per-seat evidence would block a seated candidate the day one -# exists. Accepting unknown for a seat is AGENTS.md section 4's rule for this -# case: disclosed uncertainty keeps a candidate eligible, only concrete -# contradictory evidence blocks it. A notice on stderr says its headroom was -# never verified, because that is a disclosure and not a claim of headroom. -# The marker is refused on every harness but claude, in the same validation -# loop that refuses a malformed token, because claude is the only harness with -# a config-dir seat (bin/fm-spawn.sh refuses --claude-config-dir for any other -# harness for the same reason). Without that restriction the marker would be an -# unrestricted quota-gate off-switch anywhere else, with no seat behind it to -# justify the unknown. -# # omp (Oh My Pi) has no single primary family, so its candidate model prefix # selects the family: openai-codex/ checks the codex row and # claude-bridge/ checks the claude row, each against the bare for @@ -121,24 +93,11 @@ done [ "${#CANDIDATES[@]}" -gt 0 ] || die "no candidates supplied" -# A candidate is :, optionally marked `@seated`. A bare harness -# with no colon means the default model. Reject empty harnesses and characters -# that cannot form a safe token. A colon-separated model is legal (e.g. -# model:codex_bengalfox). The marker is stripped before the token is validated, -# so `@` remains illegal everywhere else and a suffix that is not exactly -# `@seated` refuses rather than being read as part of the model. -candidate_is_seated() { case "$1" in *@seated) return 0 ;; *) return 1 ;; esac; } -candidate_token() { printf '%s\n' "${1%@seated}"; } - +# A candidate is :. A bare harness with no colon means the +# default model. Reject empty harnesses and characters that cannot form a safe +# token. A colon-separated model is legal (e.g. model:codex_bengalfox). for c in "${CANDIDATES[@]}"; do - token=$c case "$c" in - *@*) - candidate_is_seated "$c" || die "invalid candidate: $c" - token=$(candidate_token "$c") - ;; - esac - case "$token" in ''|:*|*[!A-Za-z0-9._/:-]*) die "invalid candidate: $c" ;; esac done @@ -388,43 +347,32 @@ effective_for_provider_model() { } for c in "${CANDIDATES[@]}"; do - token=$(candidate_token "$c") - harness=${token%%:*} - model=${token#*:} - [ "$model" = "$token" ] && model="default" + harness=${c%%:*} + model=${c#*:} + [ "$model" = "$c" ] && model="default" [ -n "$model" ] || die "invalid candidate: $c" fm_control_harness_supported "$harness" || die "unknown harness: $harness" provider_for_harness "$harness" "$model" >/dev/null || case "$harness" in omp) die "omp quota mapping covers only the openai-codex and claude-bridge prefixes: $model" ;; *) die "unknown harness: $harness" ;; esac - if candidate_is_seated "$c" && [ "$harness" != claude ]; then - die "invalid candidate: $c (@seated is only meaningful for the claude harness, the only one with a config seat)" - fi done chosen="none" for c in "${CANDIDATES[@]}"; do - token=$(candidate_token "$c") - harness=${token%%:*} - model=${token#*:} - [ "$model" = "$token" ] && model="default" - seated=false - if candidate_is_seated "$c"; then - seated=true - effective='{"status":"unknown"}' - else - provider=$(provider_for_harness "$harness" "$model") - scope_model=$model - [ "$harness" != omp ] || scope_model=${model#*/} - effective=$(effective_for_provider_model "$provider" "$scope_model") - fi + harness=${c%%:*} + model=${c#*:} + [ "$model" = "$c" ] && model="default" + provider=$(provider_for_harness "$harness" "$model") + scope_model=$model + [ "$harness" != omp ] || scope_model=${model#*/} + effective=$(effective_for_provider_model "$provider" "$scope_model") if [ -z "$effective" ] || [ "$effective" = "null" ]; then continue fi - if printf '%s\n' "$effective" | jq -e --argjson seated "$seated" ' + if printf '%s\n' "$effective" | jq -e ' if (.runway.status // "") == "exhausted_now" then false - elif .status == "unknown" then $seated + elif .status == "unknown" then false else .effectivePercentRemaining as $remaining | (($remaining | type) == "number") and @@ -432,7 +380,6 @@ for c in "${CANDIDATES[@]}"; do ((.runway.status // "") != "exhausted_now") end ' >/dev/null 2>&1; then - [ "$seated" = false ] || echo "notice: candidate '$c' runs on a config seat, whose own account has no row in this quota snapshot; its headroom was not verified and it is treated as eligible with unknown quota" >&2 chosen="$harness $model" break fi diff --git a/tests/fm-quota-choose.test.sh b/tests/fm-quota-choose.test.sh index 7c25b8c1a4f..095e292365f 100755 --- a/tests/fm-quota-choose.test.sh +++ b/tests/fm-quota-choose.test.sh @@ -259,85 +259,6 @@ fi [ "$err" = "error: invalid candidate: claude:" ] || fail "trailing empty model returned: $err" ok "all candidates are validated before selection" -# A config-seated candidate spends an account this snapshot never measures, so -# the family's single row must not judge it; it carries an unknown quota value -# through the same eligibility decision every candidate goes through, where -# unknown is eligible only for a seat. The fixture's claude rows are a -# barely-positive all_models scope and an exhausted model:fable scope, so -# claude:model:fable is the candidate the default account's own row rejects. -if out=$(call_choose --snapshot "$LAB/captured.json" --candidate claude:model:fable 2>/dev/null); then - fail "seat baseline: an unseated exhausted claude candidate was selected, got '$out'" -fi -[ "$out" = "none" ] || fail "seat baseline: expected 'none' for the unseated candidate, got '$out'" -out=$(call_choose --snapshot "$LAB/captured.json" --candidate claude:model:fable@seated) -[ "$out" = "claude model:fable" ] || fail "seated candidate: expected the default account's exhausted row to be ignored, got '$out'" -ok "a seated candidate is not judged by the default account's quota row" - -# The seat exception removes the wrong-account penalty; it does not promote a -# seated candidate over an earlier eligible one. -out=$(call_choose --snapshot "$LAB/captured.json" --candidate pi:default --candidate claude:model:fable@seated) -[ "$out" = "pi default" ] || fail "seat ordering: an earlier eligible candidate should still win, got '$out'" -out=$(call_choose --snapshot "$LAB/captured.json" --candidate kimi:default --candidate claude:model:fable@seated) -[ "$out" = "claude model:fable" ] || fail "seat ordering: a seated candidate should be reached past an exhausted one, got '$out'" -ok "candidate order still governs a seated candidate" - -# Unknown headroom must be disclosed, never dressed up as measured quota. -seat_err=$(QUOTA_AXI_CALLS="$LAB/seat-calls" QUOTA_AXI_FIXTURE="$FIXTURE" PATH="$FAKEBIN:$PATH" \ - "$BIN/fm-quota-choose.sh" --snapshot "$LAB/captured.json" --candidate claude:model:fable@seated 2>&1 >/dev/null) -printf '%s\n' "$seat_err" | grep -Fq "claude:model:fable@seated" \ - || fail "seat notice did not name the chosen candidate: $seat_err" -printf '%s\n' "$seat_err" | grep -Fq "headroom was not verified" \ - || fail "seat notice did not disclose the unverified headroom: $seat_err" -unseated_err=$(QUOTA_AXI_CALLS="$LAB/seat-calls" QUOTA_AXI_FIXTURE="$FIXTURE" PATH="$FAKEBIN:$PATH" \ - "$BIN/fm-quota-choose.sh" --snapshot "$LAB/captured.json" --candidate pi:default 2>&1 >/dev/null) -[ -z "$unseated_err" ] || fail "an unseated selection should stay silent on stderr: $unseated_err" -ok "choosing a seated candidate discloses its unverified headroom on stderr" - -# The marker is an exact token, not a free-form suffix. -for bad in 'claude:default@seat' 'claude:default@' 'claude@seated:default' '@seated'; do - if err=$(call_choose --snapshot "$LAB/captured.json" --candidate "$bad" 2>&1); then - fail "malformed seat marker '$bad' unexpectedly selected a candidate" - fi - [ "$err" = "error: invalid candidate: $bad" ] || fail "malformed seat marker '$bad' returned: $err" -done -ok "a malformed seat marker fails closed with the invalid-candidate error" - -# The marker exempts a candidate from the default account's quota row, nothing -# else: every gate that does not depend on quota still refuses a seated -# candidate exactly as it refuses an unseated one. -if err=$(call_choose --snapshot "$LAB/captured.json" --candidate bogus:default@seated 2>&1); then - fail "a seated candidate on an unknown harness was selected" -fi -[ "$err" = "error: unknown harness: bogus" ] || fail "seated unknown harness returned: $err" -if err=$(call_choose --snapshot "$LAB/captured.json" --candidate pi:default --candidate 'claude:@seated' 2>&1); then - fail "a seated candidate with an empty model was hidden by an earlier selection" -fi -[ "$err" = "error: invalid candidate: claude:@seated" ] || fail "seated empty model returned: $err" -if err=$(call_choose --snapshot "$LAB/captured.json" --candidate omp:ollama/qwen3:8b@seated 2>&1); then - fail "a seated omp candidate with an unmapped prefix was selected" -fi -[ "$err" = "error: omp quota mapping covers only the openai-codex and claude-bridge prefixes: ollama/qwen3:8b" ] \ - || fail "seated unmapped omp prefix returned: $err" -ok "the seat marker exempts a candidate from the quota row only, never from the other gates" - -# Only claude has a config seat, so the marker anywhere else would be a quota -# off-switch with nothing behind it. The fixture's codex rows are exhausted at -# model:codex_bengalfox, so an accepted marker there would select a candidate -# the snapshot says is spent. -if err=$(call_choose --snapshot "$LAB/captured.json" --candidate codex:default@seated 2>&1); then - fail "a seated non-claude candidate was selected instead of refused" -fi -[ "$err" = "error: invalid candidate: codex:default@seated (@seated is only meaningful for the claude harness, the only one with a config seat)" ] \ - || fail "seated non-claude candidate returned: $err" -if err=$(call_choose --snapshot "$LAB/captured.json" --candidate pi:default --candidate codex:model:codex_bengalfox@seated 2>&1); then - fail "a trailing seated non-claude candidate was hidden by an earlier selection" -fi -[ "$err" = "error: invalid candidate: codex:model:codex_bengalfox@seated (@seated is only meaningful for the claude harness, the only one with a config seat)" ] \ - || fail "trailing seated non-claude candidate returned: $err" -out=$(call_choose --snapshot "$LAB/captured.json" --candidate claude:model:fable@seated) -[ "$out" = "claude model:fable" ] || fail "the claude harness must still accept the seat marker, got '$out'" -ok "the seat marker is refused on every harness but claude" - printf '{"schemaVersion":5,"providers":{"provider":"claude","quotaSemantics":{"effectiveAvailability":[{"scope":"all_models","status":"known","effectivePercentRemaining":50,"runway":{"status":"through_reset"}}]}}}\n' > "$MALFORMED" if err=$(call_choose --snapshot "$MALFORMED" --candidate claude:default 2>&1); then fail "malformed provider collection unexpectedly dispatched" From 9b5b36cbc74771429daead989c08ffa4285970dc Mon Sep 17 00:00:00 2001 From: keenvc Date: Sat, 19 Sep 2026 09:39:51 +0000 Subject: [PATCH 13/35] no-mistakes(review): keep recorded Claude seat across bare secondmate respawn --- bin/fm-spawn.sh | 20 ++++++++- tests/fm-spawn-dispatch-profile.test.sh | 57 +++++++++++++++++++++++++ 2 files changed, 76 insertions(+), 1 deletion(-) diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index a53db16083d..b4cb8506f4f 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -62,7 +62,10 @@ # single-store default, byte-identical to before this flag existed); a # --relaunch always reuses that recorded value and refuses a fresh # --claude-config-dir, so a relaunch can never silently move a task to a -# different seat. +# different seat. A bare `--secondmate` respawn of an existing secondmate - +# the shape bin/fm-bootstrap.sh's liveness sweep recovers with - reads the +# seat back out of that same record for the same reason, unless this spawn +# passes its own --claude-config-dir, which still wins. # --model and --effort are concrete profile # axes chosen by firstmate at intake. They are only threaded into harnesses whose # installed CLIs were verified to support that axis; unsupported axes are omitted @@ -2155,6 +2158,21 @@ elif [ -n "$CLAUDE_SEAT_ARG" ]; then } CLAUDE_SEAT_DIR=$(fm_claude_seat_validate "$CLAUDE_SEAT_ARG" "--claude-config-dir") || exit 1 CLAUDE_SEAT_RECORD=$CLAUDE_SEAT_DIR +elif [ "$KIND" = secondmate ]; then + # bin/fm-bootstrap.sh recovers a dead secondmate with a bare + # `fm-spawn.sh --secondmate` - no --relaunch and no flag - so the seat has + # to come back from this secondmate's own record the way its home already does, + # or the recovery would move a seated lane onto firstmate's ambient account and + # erase the record with it. A first-ever secondmate has no meta, so fm_meta_get + # yields nothing and this is a no-op. + CLAUDE_SEAT_RECORD=$(fm_meta_get "$STATE/$ID.meta" claude_config_dir) + if [ -n "$CLAUDE_SEAT_RECORD" ] && [ "$HARNESS" = claude ]; then + CLAUDE_SEAT_DIR=$(fm_claude_seat_validate "$CLAUDE_SEAT_RECORD" "this secondmate's recorded Claude config directory") || { + echo "hint: the seat is the one recorded in this secondmate's own meta; restore that directory, or pass --claude-config-dir to seat it somewhere else" >&2 + exit 1 + } + CLAUDE_SEAT_RECORD=$CLAUDE_SEAT_DIR + fi fi secondmate_registry_value() { diff --git a/tests/fm-spawn-dispatch-profile.test.sh b/tests/fm-spawn-dispatch-profile.test.sh index e982a54c7af..3c3f405b0e9 100755 --- a/tests/fm-spawn-dispatch-profile.test.sh +++ b/tests/fm-spawn-dispatch-profile.test.sh @@ -1127,6 +1127,61 @@ assert_attribution_policy() { # assert_contains "$launch" '"sessionUrl":false' "$what launch does not silence the session URL" } +# bin/fm-bootstrap.sh's liveness sweep recovers a dead secondmate with a bare +# `fm-spawn.sh --secondmate` - no --relaunch, no flag, home and identity +# taken from the existing record. The seat has to survive that the way the home +# does, or the recovery moves a seated lane onto firstmate's own account and +# erases the record, leaving a later relaunch nothing to restore. +test_bare_secondmate_respawn_keeps_the_recorded_claude_seat() { + local rec id sm seat out status launch + id=profile-secondmate-seat-respawn-z33 + rec=$(make_spawn_case profile-secondmate-seat-respawn claude "$id") + read_case_record "$rec" + sm="$CASE_DIR/secondmate-home" + make_seeded_secondmate_home "$sm" "$id" + seat=$(make_claude_seat "$CASE_DIR/seat") + + out=$(run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$sm" --secondmate --claude-config-dir "$seat") + status=$? + expect_code 0 "$status" "the seated secondmate's first spawn should succeed"$'\n'"$out" + assert_grep "claude_config_dir=$seat" "$HOME_DIR/state/$id.meta" \ + "the seated secondmate's first spawn did not record its seat" + + out=$(run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" --secondmate) + status=$? + expect_code 0 "$status" "the bare recovery respawn should succeed"$'\n'"$out" + assert_grep "claude_config_dir=$seat" "$HOME_DIR/state/$id.meta" \ + "the bare respawn dropped the secondmate's recorded seat from its meta" + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "CLAUDE_CONFIG_DIR='$seat'" \ + "the bare respawn launched on the single-store default instead of the recorded seat" + pass "a bare secondmate respawn keeps the seat recorded at creation, in its meta and its launch" +} + +test_bare_secondmate_respawn_refuses_a_recorded_seat_that_vanished() { + local rec id sm seat out status + id=profile-secondmate-seat-gone-z34 + rec=$(make_spawn_case profile-secondmate-seat-gone claude "$id") + read_case_record "$rec" + sm="$CASE_DIR/secondmate-home" + make_seeded_secondmate_home "$sm" "$id" + seat=$(make_claude_seat "$CASE_DIR/seat") + + out=$(run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$sm" --secondmate --claude-config-dir "$seat") + status=$? + expect_code 0 "$status" "the seated secondmate's first spawn should succeed"$'\n'"$out" + # The operator removed the seat between the creation and the recovery. + rm -f "$seat/.claude.json" + + out=$(run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" --secondmate) + status=$? + [ "$status" -ne 0 ] || fail "a respawn whose recorded seat is unusable must refuse"$'\n'"$out" + assert_contains "$out" "this secondmate's recorded Claude config directory '$seat'" \ + "the refusal should name the record the seat came from, not a flag the caller never passed" + [ ! -s "$LAUNCH_LOG" ] || fail "an unusable recorded seat must launch nothing (got: $(cat "$LAUNCH_LOG"))" + pass "a bare secondmate respawn refuses when its recorded seat is no longer usable" +} + test_claude_task_launch_carries_control_channel_authority() { local rec id out status launch id=profile-claude-control-channel-z21 @@ -1661,6 +1716,8 @@ test_claude_config_dir_not_a_directory_refuses test_claude_config_dir_without_config_refuses test_claude_config_dir_refused_for_non_claude_harness test_claude_config_dir_refused_for_a_remote_secondmate +test_bare_secondmate_respawn_keeps_the_recorded_claude_seat +test_bare_secondmate_respawn_refuses_a_recorded_seat_that_vanished test_claude_task_launch_carries_control_channel_authority test_claude_secondmate_launch_omits_task_control_channel_authority test_claude_crewmate_launch_carries_the_attribution_policy From 30e7a05124aafdf6370850f05da6360156b8e07b Mon Sep 17 00:00:00 2001 From: keenvc Date: Sat, 19 Sep 2026 10:12:29 +0000 Subject: [PATCH 14/35] no-mistakes(document): document remote seat refusal; correct stale seat doc claims --- .../harness-adapters/references/harness/claude.md | 4 ++-- bin/fm-spawn.sh | 14 ++++++++------ docs/verification/runtime-backends.md | 3 ++- 3 files changed, 12 insertions(+), 9 deletions(-) diff --git a/.agents/skills/harness-adapters/references/harness/claude.md b/.agents/skills/harness-adapters/references/harness/claude.md index 5427109b6ad..140662b79da 100644 --- a/.agents/skills/harness-adapters/references/harness/claude.md +++ b/.agents/skills/harness-adapters/references/harness/claude.md @@ -13,7 +13,7 @@ Busy hooks verified 2026-07-28 on Claude Code 2.1.220. | Model | `--model `; discover through the interactive `/model` picker, with alias or full-name shape documented by `claude --help`. | | Effort | `--effort `, verified on 2.1.196. | | Permissions | `--dangerously-skip-permissions` by default, or `--permission-mode auto` when `config/claude-permission-mode` is `auto`; the `auto` shape verified on 2.1.269, and `../../../../../docs/configuration.md` "Claude permission mode" owns the file. | -| Config seat | `--claude-config-dir ` on `fm-spawn.sh` picks the `CLAUDE_CONFIG_DIR` this one claude spawn's pane resolves into, validated and recorded in the task's own meta and reused on relaunch, so two claude lanes can sit on different accounts at once; `fm-spawn.sh --help` owns the exact contract. | +| Config seat | `--claude-config-dir ` on `fm-spawn.sh` picks the `CLAUDE_CONFIG_DIR` this one claude spawn's pane resolves into, validated and recorded in the task's own meta and reused whenever that same task spawns again, so two claude lanes can sit on different accounts at once; `fm-spawn.sh --help` owns the exact contract. | ## Workspace trust @@ -42,7 +42,7 @@ Never send Enter to that one either: it was observed rendering in the same shape Firstmate cannot move a selection with Enter, Escape, and C-c alone, so it cannot accept this dialog at all, and an operator accepts it once per machine instead. Inspect the pane to identify which dialog is on screen, and report it rather than answering it. A launch under `config/claude-permission-mode=auto` never meets the bypass confirmation, because it does not request bypass mode: on 2.1.269 `claude --permission-mode auto` reached the composer directly with the footer `⏵⏵ auto mode on (shift+tab to cycle)`, so a captain who refuses the bypass dialog selects `auto` there instead of accepting it. -That setting is fleet-wide rather than per seat - `config/claude-permission-mode` is read once per home and governs every claude launch from it, crewmates, scouts, secondmates, and relaunches alike - so choosing `auto` to avoid the dialog for one seat moves every claude lane in that home with it. +That setting is fleet-wide rather than per seat (the Permissions row above names its owner), so choosing `auto` to avoid the dialog for one seat moves every claude lane in that home with it. The workspace-trust dialog is unaffected by the permission mode and still needs the pre-registration above. ### Preparing a config seat diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index b4cb8506f4f..167fa867ba1 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -52,12 +52,14 @@ # into, letting two claude lanes run concurrently under different accounts # without moving every claude lane at once the way setting CLAUDE_CONFIG_DIR # in firstmate's own environment would. Refused when the resolved harness is -# not claude. Validated before any worktree or endpoint is created: -# must resolve to an existing directory holding a Claude config store -# (.claude.json), or the spawn refuses naming rather than launching a -# worker that would wedge. The resolved directory feeds both -# bin/fm-claude-trust.sh's pre-registration and the launch's own -# CLAUDE_CONFIG_DIR, so the two halves can never land in different stores. +# not claude, and on a remote secondmate, whose launch happens on another +# host where a local directory path names nothing. Validated before any +# worktree or endpoint is created: must resolve to an existing +# directory holding a Claude config store (.claude.json), or the spawn +# refuses naming rather than launching a worker that would wedge. +# The resolved directory feeds both bin/fm-claude-trust.sh's +# pre-registration and the launch's own CLAUDE_CONFIG_DIR, so the two +# halves can never land in different stores. # Recorded in the task's own meta as claude_config_dir= (absent means the # single-store default, byte-identical to before this flag existed); a # --relaunch always reuses that recorded value and refuses a fresh diff --git a/docs/verification/runtime-backends.md b/docs/verification/runtime-backends.md index f870d561b89..c45cf6b8c7b 100644 --- a/docs/verification/runtime-backends.md +++ b/docs/verification/runtime-backends.md @@ -460,7 +460,8 @@ That verification is point-in-time rather than a durable guarantee, because a co One limitation belongs beside that result. An intermediate arm run against an isolated `CLAUDE_CONFIG_DIR` holding only a copied `.claude.json` cleared the trust dialog but then surfaced the separate machine-scoped Bypass Permissions warning. That warning rendered in the same shape as the trust dialog, with the selection cursor on `No, exit` and the footer `Enter to confirm . Esc to cancel`, so a sent Enter would end that worker too. -That gate is not a production blocker, because a normal environment has already accepted it and the treatment arm above ran against the real config and saw neither dialog. +That gate is not a production blocker for a launch against the ambient store, because a normal environment has already accepted it and the treatment arm above ran against the real config and saw neither dialog. +It does reach production for a spawn seated on its own store with `bin/fm-spawn.sh --claude-config-dir`, whose preparation `.agents/skills/harness-adapters/references/harness/claude.md` owns under "Preparing a config seat". This change does not address that warning and does not claim to. ### Secondmate homes From 1f9beb7ec832b6f538c5d100c17e6bdb65fcd237 Mon Sep 17 00:00:00 2001 From: keenvc Date: Sun, 20 Sep 2026 00:48:20 +0000 Subject: [PATCH 15/35] feat: add human-text-discipline skill for human-read text --- .agents/skills/human-text-discipline/SKILL.md | 115 ++++++++++++++++++ AGENTS.md | 3 + docs/documentation-audiences.json | 4 + 3 files changed, 122 insertions(+) create mode 100644 .agents/skills/human-text-discipline/SKILL.md diff --git a/.agents/skills/human-text-discipline/SKILL.md b/.agents/skills/human-text-discipline/SKILL.md new file mode 100644 index 00000000000..10b14487e35 --- /dev/null +++ b/.agents/skills/human-text-discipline/SKILL.md @@ -0,0 +1,115 @@ +--- +name: human-text-discipline +description: >- + Agent-only writing discipline for text a human reads for its own sake. + Use before writing or editing a pull request body, a commit message, or a captain-facing chat message. + Owns the checkable list of AI tells and their fixes, plus the positive target that keeps corrected prose from reading as sterile. +user-invocable: false +metadata: + internal: true +--- + +# human-text-discipline + +Load this before writing or editing text a person reads directly: a pull request body, a commit message, or a captain-facing message. +Its companion is `agent-doc-discipline`, which disciplines documents an agent consumes to act; that skill's test is that a fresh session can act on the document, while this skill's test is that a person reads the text as written by a person for them. +It is not a second owner of `AGENTS.md` section 9. +Section 9 owns what a captain-facing message must contain and how internal terms are translated. +This skill owns how the prose reads once that content is decided. +Where they meet, section 9 decides substance and this skill decides wording, and neither restates the other. + +Not for documents an agent reads: those follow `agent-doc-discipline`. +Not for maintained project prose: those follow the audience owner in `docs/documentation-audiences.md`. + +## The other failure: sterile prose + +Removing tells is half the job. +Flat, voiceless text is as obviously machine-made as purple text, so the rewrite must also add a human signal. +Four moves carry most of it: + +- Hold an opinion instead of neutrally listing both sides. +- Vary sentence length: a short sentence, then a longer one that takes its time. +- Admit complexity, because "impressive and a little unsettling" beats "impressive". +- Be specific, because a name, a number, or a concrete detail is the strongest human signal there is. + +## The tells + +Each entry names the observable pattern and the fix. +Scan for the pattern, apply the fix, then confirm the fix did not flatten the sentence. + +### Content + +1. Puffery: "pivotal moment", "testament to", "evolving landscape", "setting the stage", "indelible mark". + Delete it and state what actually happened. +2. Promotional adjectives used as praise: "vibrant", "groundbreaking", "renowned", "seamless", "robust", "cutting-edge". + Replace with a neutral description or a number. +3. Superficial "-ing" tails: a comma followed by "highlighting", "ensuring", "showcasing", "reflecting", or "fostering" with no fact after it. + Cut the tail, or expand it into the concrete fact it gestures at. +4. Vague attribution: "experts believe", "industry reports suggest", "it is widely regarded". + Name the source or delete the claim. +5. Formulaic framing: "Despite the challenges, X continues to thrive". + Replace it with the specific facts behind the framing. +6. Generic conclusion: "The future looks bright", "This is a big step forward". + State the specific next step or result, or delete the sentence. + +### Diction + +7. Fancy ways to say "is": "serves as", "stands as", "boasts", "features", "represents". + Use "is" or "has". +8. "Not just X but Y", and its cousin "It is not merely X, it is Y". + State the point directly instead of staging it as a contrast. +9. AI vocabulary: additionally, crucial, delve, enhance, foster, garner, interplay, intricate, landscape (abstract), pivotal, showcase, tapestry, testament, underscore, vibrant. + Replace with the plain word a person would say. +10. Abstract metaphor nouns: substrate, wedge, vector, locus, nexus, primitive, harness (as metaphor), surface (as in "API surface"), bedrock, scaffolding, flywheel, north star. + Use the concrete word the metaphor stands for. +11. A feeling instead of a fact: "the database stays close at hand". + Name the mechanism or the number the reader can act on. +12. A weak verb propped up by an adverb: "significantly improves", "runs quickly". + Use a stronger verb or the measured number. +13. Hedging: "could potentially possibly", "it might be argued that". + Reduce it to the single honest word, such as "may". +14. Filler: "in order to", "due to the fact that", "it is important to note that". + Use "to", "because", or delete it. +15. Synonym cycling: protagonist, main character, central figure, hero all in one paragraph. + Pick one word and repeat it. +16. The plain word: "utilize" becomes "use", "leverage" becomes "use", "facilitate" becomes "help", "numerous" becomes "many", "in the event that" becomes "if". + The fancier synonym is rarely clearer. + +### Structure + +17. Forced rule of three: ideas padded or trimmed to arrive in threes. + Use the natural number, even when that is two or four. +18. False ranges: "from X to Y" where X and Y are not on a meaningful scale. + List the items directly. +19. Dense sentences the reader must backtrack to parse. + Split into two sentences or drop a clause, one idea per sentence. +20. Passive voice that hides a known actor: "queries are validated". + Name the actor: "the compiler validates queries". + Passive stays only when the actor is unknown or genuinely irrelevant. + +### Mechanics + +21. An em dash as punctuation. + End the sentence or use a comma, never a dash, an en dash, or parentheses as a substitute. +22. A colon as a mid-sentence connector: "If you are coming from X: instead, do Y". + Rewrite so the point stands on its own without the comparison framing. +23. Boldface on every proper noun or acronym. + Bold only what a scanner must find. +24. Inline-header list items whose bold label restates the line: "**Performance:** Performance improved". + Convert them to prose; a bold lead-in that names the item and is followed by genuinely new detail is fine, not a tell. +25. Title case headings. + Use sentence case. +26. Decorative emojis in headings or bullets. + Remove them. +27. Curly quotes. + Use straight quotes. +28. Chatbot phrases: "Of course!", "I hope this helps!", "Let me know if you need anything else". + Remove them. +29. Sycophantic openers: "Great question!", "You are absolutely right!". + Respond directly to the substance. + +## Before you send + +Ask one audit question: what still makes this obviously machine-written? +Walk the tells above once more against the answer. +Then confirm the positive side survived the edit: at least one opinion, at least one sentence that breaks the rhythm, and at least one specific name or number where generality was possible. diff --git a/AGENTS.md b/AGENTS.md index c3634216137..f6ada9de71b 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -525,6 +525,8 @@ Use plain chat for a yes-or-no decision and `lavish-axi` only when several optio Whenever a PR is mentioned, and for any review or merge ask, include the PR's full `https://...` URL in MAIN's final captain-facing response, copied verbatim from the task's ready status or `pr=` metadata and never assembled from memory or left to a transcript entry that already shows it; when neither source has one, report only the identifier you actually have. Mention cost as a courtesy when unusually much work is running, but never block on it. +Load `human-text-discipline` before writing a PR body, a commit message, or any captain-facing message; it owns the checkable AI-tell list for text a person reads, and this section stays the owner of what those messages must contain. + ## 10. Backlog contract The configured `tasks-axi` backend is the durable queue; the tracked default is `data/backlog.md`. @@ -589,6 +591,7 @@ These skills are not captain-invocable; load them only at their precise triggers - `fmx-respond` - load on an `x-mention ` `check:` wake to handle the mention, on an `x-mode-error ...` `check:` wake to report the Relay configuration blocker, on a `public-followup ...` `check:` wake or a startup-surfaced public commitment, and on any milestone or terminal wake for a Relay-linked task before posting its completion follow-up; relevant only when Relay is on. - `firstmate-codexapp` - load before coordinating a visible Codex Desktop thread, evaluating a Codex App backend request, or reconciling Codex Desktop host-tool smoke evidence for Firstmate work. - `firstmate-coding-guidelines` - load before changing firstmate's shared, tracked material, as defined by section 1's list, whether editing directly or briefing a crewmate for a firstmate-repo task. +- `human-text-discipline` - load before writing or editing a pull request body, a commit message, or a captain-facing chat message; it owns the checkable AI-tell list for text a human reads and does not restate section 9. ## 14. Relay diff --git a/docs/documentation-audiences.json b/docs/documentation-audiences.json index e459e95006a..a90329db4d3 100644 --- a/docs/documentation-audiences.json +++ b/docs/documentation-audiences.json @@ -224,6 +224,10 @@ "path": ".agents/skills/harness-adapters/references/harness/rovo.md", "audience": "agent-runtime" }, + { + "path": ".agents/skills/human-text-discipline/SKILL.md", + "audience": "agent-runtime" + }, { "path": ".agents/skills/process-event-sources/SKILL.md", "audience": "agent-runtime" From 83dad8080cd6d1f19a0792a629a020cc78d0de8b Mon Sep 17 00:00:00 2001 From: keenvc Date: Sun, 20 Sep 2026 00:51:50 +0000 Subject: [PATCH 16/35] feat: require a before/after evidence pair in the ship definition of done --- AGENTS.md | 1 + bin/fm-dod-lib.sh | 37 +++++++++++++++++++++++++++---------- docs/architecture.md | 2 +- tests/fm-brief.test.sh | 30 ++++++++++++++++++++++++++++++ 4 files changed, 59 insertions(+), 11 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index c3634216137..acd6c5ee9e3 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -367,6 +367,7 @@ The task worker that starts a no-mistakes run drives the pipeline and owns every Firstmate never invokes `no-mistakes axi respond` for a crew-owned run. When the captain adds or changes an ask mid-task, append the captain's words without added speaker labels or direct address to that brief's `## Captain's intent` and relay those words to the worker; Firstmate build constraints stay in `## Firstmate spec` or the steer. `bin/fm-dod-lib.sh` owns the worker-side `--intent` contract. +`bin/fm-dod-lib.sh` also owns the definition of done's before/after evidence-pair requirement for any change with an observable surface. Once validation starts, prefer routing new requirements to follow-up work rather than expanding the current task, unless a new requirement completely invalidates the work being validated; however, the smallest downstream changes needed to keep already accepted product or engineering behavior correct, add behavioral tests where an executable contract exists, or keep documentation accurate remain within the current task even when they touch files not named at intake, and corrections required to satisfy already accepted intent are not new requirements. Only a current, explicit captain instruction that completely invalidates the work being validated keeps the task with the same worker instead of routing it to follow-up work or handing it to a replacement. diff --git a/bin/fm-dod-lib.sh b/bin/fm-dod-lib.sh index b1cf1fd80d7..9b3a6bbb900 100755 --- a/bin/fm-dod-lib.sh +++ b/bin/fm-dod-lib.sh @@ -10,7 +10,11 @@ # mode is refused rather than silently rendered as the pipeline contract. # The block opens with the fixed machine-readable "Delivery contract: mode=" # line that bin/fm-spawn.sh checks a ship brief against. -# This file is the one owner of the no-mistakes `--intent` contract: only the +# This file is the one owner of the definition of done's before/after +# evidence-pair requirement, rendered by fm_dod_evidence_pair into every mode's +# block: any change with an observable surface needs a before/after pair taken +# with one stated methodology, and the before is captured at reproduction time. +# It also owns the no-mistakes `--intent` contract: only the # brief's `## Captain's intent` subsection plus later captain words, never # `## Firstmate spec` and never the worker's own tradeoffs. # Author the subsection body and later relays as the actual words, without @@ -232,13 +236,33 @@ fm_ask_user_escalation_block() { # EOF } +fm_dod_evidence_pair() { + cat <<'EOF' + +## Evidence pair (capture the before at reproduction) +Any change with an observable surface requires a before/after pair captured with one stated methodology, and the before is captured while reproducing the defect, before any fix, when it is cheapest. +The surface is observable when a person could see it or a number could move: a UI or rendered artifact, an API response, a measured value, or an output pair. +State the methodology once - the tool, the exact command or URL, the data set, the environment, and any device or viewport - and apply that same methodology to both captures, so they are comparable and a measurement that is wrong in both directions cannot pass unnoticed. +A change with no visible surface uses the same discipline with measured numbers or before/after output instead of screenshots; when nothing observable can move, say so and name what you verified instead. +Keep the pair with the task's own deliverable - in the PR body, the delivery path's evidence location, or the task report - and never upload it to a public host. +EOF +} + fm_dod_block() { # local mode=$1 id=$2 + case "$mode" in + direct-PR|local-only|no-mistakes) ;; + *) + echo "error: fm_dod_block: unknown delivery mode '$mode'" >&2 + return 1 ;; + esac + printf '# Definition of done\n' + printf 'Delivery contract: mode=%s\n' "$mode" + fm_dod_evidence_pair + printf '\n' case "$mode" in direct-PR) cat <&2 - return 1 ;; esac } diff --git a/docs/architecture.md b/docs/architecture.md index 57c2d05af76..86004136f63 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -326,7 +326,7 @@ Each task's mode and `yolo` merge posture are firstmate's decision at intake. The mode is passed explicitly to `bin/fm-brief.sh`, and both values are passed explicitly to `bin/fm-spawn.sh` and `bin/fm-promote.sh`; each command refuses to guess the values it consumes. A ship brief records its mode as a fixed machine-readable line and the spawn refuses to launch on a different one, so the worker's instructions and the recorded task delivery cannot diverge. `bin/fm-dod-lib.sh` is the one owner of that mode's definition of done, rendered into a generated ship brief, the ship instructions a promoted scout receives, and that scout's own `brief.md` so a later relaunch reads the same contract, so a promoted worker cannot be handed a weaker contract than a briefed one. -It is also the one owner of the no-mistakes `--intent` contract those workers follow. +It is also the one owner of the no-mistakes `--intent` contract those workers follow, and of the definition of done's before/after evidence-pair requirement. `data/projects.md` records each project's standing posture and optional `+yolo` merge flag as the captain's default and as context for that decision, including the conditional `no-mistakes-prod-only` policy; a ship spawn that drops below the registered rigor prints a deviation notice and continues. `bin/fm-project-mode.sh` remains the one registry parser for the mechanical consumers that have no task in hand: fleet sync's `local-only` skip and home seeding's refusal and no-mistakes initialization. When a selected delivery path calls for a diff, `bin/fm-review-diff.sh` refreshes the authoritative base and, when task meta records `pr=`, always fetches and compares against `refs/pull//head` by default (recorded `pr_head=` is only an offline fallback) before falling back to the local branch with a warning. diff --git a/tests/fm-brief.test.sh b/tests/fm-brief.test.sh index 88f4e566ff0..0a98c069bf5 100755 --- a/tests/fm-brief.test.sh +++ b/tests/fm-brief.test.sh @@ -376,6 +376,35 @@ test_no_mistakes_dod_wording() { pass "fm-brief.sh: no-mistakes DOD keeps its apostrophe prose and bans --yes outright" } +# Reproducing a defect is the cheapest point to capture its before state, and an +# after-only check cannot catch a measurement that was wrong in both directions: +# the same wrong ruler applied twice shows no movement. Every ship mode's +# definition of done must therefore require a before/after pair captured with one +# stated methodology, the before taken at reproduction time, including measured +# numbers and output pairs for non-UI work. +test_ship_dod_requires_evidence_pair() { + local home id mode brief + home="$TMP_ROOT/evidence-pair-home" + mkdir -p "$home/data" + for mode in no-mistakes direct-PR local-only; do + id="brief-evidence-$(printf '%s' "$mode" | tr '[:upper:]' '[:lower:]')" + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" "$id" some-proj --mode "$mode" >/dev/null 2>&1 \ + || fail "fm-brief.sh $id --mode $mode should scaffold" + brief="$home/data/$id/brief.md" + assert_grep "requires a before/after pair captured with one stated methodology" "$brief" \ + "$mode DOD must require a before/after pair with one stated methodology" + assert_grep "the before is captured while reproducing the defect, before any fix" "$brief" \ + "$mode DOD must require the before at reproduction time" + assert_grep "apply that same methodology to both captures" "$brief" \ + "$mode DOD must apply one methodology to both captures" + assert_grep "measured numbers or before/after output instead of screenshots" "$brief" \ + "$mode DOD must cover non-UI observable surfaces with measured or output pairs" + assert_grep "never upload it to a public host" "$brief" \ + "$mode DOD must keep evidence off a public upload host" + done + pass "fm-brief.sh: every ship DOD requires a before/after pair taken at reproduction" +} + test_ask_user_escalation_format() { local home id brief mode other_id other_brief home="$TMP_ROOT/ask-user-home" @@ -934,6 +963,7 @@ test_ship_mode_is_explicit_not_registry test_delivery_flags_are_refused_where_they_do_not_apply test_faster_paths_use_configured_authority_without_stacked_review test_no_mistakes_dod_wording +test_ship_dod_requires_evidence_pair test_ask_user_escalation_format test_ship_project_memory_wording test_herdr_lab_contract_is_explicit_and_complete From 2c1be1f89b519d248ad499955a8b45e1ee40b596 Mon Sep 17 00:00:00 2001 From: keenvc Date: Sun, 20 Sep 2026 01:06:04 +0000 Subject: [PATCH 17/35] feat(bin): cap live lanes per billing provider at spawn Count live crewmate, scout, and local secondmate lanes grouped by the billing provider a candidate actually draws on, expose the load for dispatch intake, and refuse a spawn that would push a provider past its configured cap. Provider identity comes from the resolved model string, not the harness name: a provider-qualified prefix or a model-id pattern decides the pool, and the harness table is only the fallback. That keeps two models on one pool counting together while a different pool stays separate. The mapping lives once in bin/fm-provider-lib.sh, whose fallback reuses the existing quota tables rather than restating them. A lane occupies a seat unless its recorded endpoint is provably dead or missing, so a cleared seat is visible before the next dispatch. The cap comes from providerCaps in config/crew-dispatch.json (per provider, else default), falling back to 4; bootstrap now rejects a malformed providerCaps instead of silently ignoring it. bin/fm-provider-load.sh prints the current per-provider used/cap for intake. Stranded-record detection stays with fm-lane-account-dead-records; this counter reads the current endpoint classifier. --- AGENTS.md | 1 + bin/fm-bootstrap.sh | 13 ++ bin/fm-provider-lib.sh | 183 ++++++++++++++++++++++++ bin/fm-provider-load.sh | 45 ++++++ bin/fm-spawn.sh | 18 +++ docs/configuration.md | 10 +- docs/examples/crew-dispatch.json | 3 +- docs/scripts.md | 2 + tests/fm-bootstrap.test.sh | 5 + tests/fm-provider-lane-cap.test.sh | 217 +++++++++++++++++++++++++++++ 10 files changed, 495 insertions(+), 2 deletions(-) create mode 100755 bin/fm-provider-lib.sh create mode 100755 bin/fm-provider-load.sh create mode 100755 tests/fm-provider-lane-cap.test.sh diff --git a/AGENTS.md b/AGENTS.md index c3634216137..da4daac96f6 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -230,6 +230,7 @@ Break genuine evidence ties without array-order or harness bias. `quota-axi` owns how model or product windows relate to bounding account windows and remains data-only. Load `quota-array-dispatch` before choosing among a matched profile array; that skill is the single owner of the TOON-first spendPriority selection procedure. Run `bin/fm-dispatch-resolve.sh` directly on the written brief in the same turn, with no preflight, and on `clear` pass its `profile:` line to `fm-spawn` unless you state a reason to override; `ambiguous`, `escalate`, `error`, and off all mean the intake above, unchanged (contract: `docs/configuration.md` "Typed dispatch resolution"). +Read current per-provider lane load with `bin/fm-provider-load.sh` before choosing a candidate, and treat a provider at its cap as a re-assignment trigger: `bin/fm-spawn.sh` refuses a dispatch that would exceed a provider's configured cap, and the model string decides which provider a candidate bills against. The generic effort fallback and its precedence are owned by `harness-adapters`: explicit captain and standing configured effort win; otherwise use low for well-understood explicit work, xhigh for ambiguous investigation or design, intermediate levels proportionally, and never max without explicit captain preference. Do not add model-specific versions of that policy. diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index 31792fa37ba..786ab93ac59 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -1132,6 +1132,18 @@ crew_dispatch_validate() { err=$(jq -r --argjson typed "$typed_active" --argjson verified_harnesses "$verified_harnesses" --arg provider_re "$FM_QUOTA_PROVIDER_ID_RE" ' def verified($h): $verified_harnesses | index($h); def provider_id($p): ($p | type) == "string" and ($p | test($provider_re)); + # providerCaps (docs/configuration.md "Crew dispatch profiles") bounds the + # live lanes one billing provider may carry; fm-provider-lib.sh enforces it + # at spawn. An invalid declaration must fail loudly here rather than be + # silently ignored, so every value must be a whole number of at least one + # and every key a provider id or the reserved `default`. + def provider_caps_bad: + (.providerCaps // null) as $c + | $c != null and ( + ($c | type) != "object" + or ([$c | keys[] | select(. != "default") | select(test($provider_re) | not)] | length > 0) + or ([$c[] | select((type != "number") or (. < 1) or (. != (. | floor)))] | length > 0) + ); def effort_ok($h; $m; $e): if $e == null then true elif ($e | type) != "string" then false @@ -1181,6 +1193,7 @@ crew_dispatch_validate() { | unique; if type != "object" then "top-level value must be an object" elif has("rules") and (.rules | type) != "array" then "rules must be an array" + elif provider_caps_bad then "providerCaps must map each provider id (or default) to a positive integer" elif [(.rules // [])[]? | select(type != "object")] | length > 0 then "each rule must be an object" elif [(.rules // [])[]? | select((.when? | type) != "string" or (.when | length) == 0)] | length > 0 then "each rule needs non-empty when" elif [(.rules // [])[]? | select((.use? | type) != "object" and (.use? | type) != "array")] | length > 0 then "each rule needs use" diff --git a/bin/fm-provider-lib.sh b/bin/fm-provider-lib.sh new file mode 100755 index 00000000000..9460764fc85 --- /dev/null +++ b/bin/fm-provider-lib.sh @@ -0,0 +1,183 @@ +#!/usr/bin/env bash +# shellcheck shell=bash +# fm-provider-lib.sh - single owner of the model -> provider identity mapping and +# the per-provider live-lane concurrency cap. +# +# Usage: +# . bin/fm-provider-lib.sh +# +# A provider is the billing pool a dispatch draws on, NOT the harness that +# launches it. The identity is taken from the resolved model string first - a +# provider-qualified `/` prefix, else a model-id pattern - and the +# harness table is consulted only when the model carries no signal. That is what +# makes `opencode-go/deepseek-v4.1-flash` and `opencode-go/deepseek-v4-pro` ONE +# pool while `pi/deepseek-v4p1-flash`, billed through Fireworks, is a different +# one. The harness fallback reuses the existing quota tables +# (fm_quota_provider_for_harness / fm_quota_single_provider_for_harness in +# bin/fm-quota-axi-lib.sh) rather than restating them; those tables are frozen +# for no-key quota routing and are never edited here. +# +# Functions: +# fm_provider_for_model +# Print the provider id the tuple bills against, or return 1 when the +# model and harness together name no known pool. +# fm_lane_provider +# Print fm_provider_for_model's answer, or the harness name as a +# last-resort bucket, or return 1 when even that is empty. +# fm_provider_cap_for [] +# Print the operator-configured cap: `providerCaps.` from +# config/crew-dispatch.json, else `providerCaps.default`, else +# FM_PROVIDER_LANE_CAP_DEFAULT. +# fm_provider_lane_counts [] [] +# Print ` ` for every provider carrying a live lane, +# sorted by provider. One line per provider, no header. +# fm_provider_cap_refuse [] +# Print an operator-facing refusal to stderr and return 1 when dispatching +# this tuple would push its provider past the cap; return 0 otherwise. +# +# Seat accounting: a lane occupies a seat unless its recorded endpoint is +# PROVABLY dead or missing (fm_backend_agent_state). A lane whose endpoint cannot +# be proven gone - alive, ambiguous, unreadable, or unverified (every backend +# except tmux and herdr) - keeps its seat, and a record with no endpoint target +# keeps it too. The count can therefore under-admit, never over-admit; a +# stranded record on an unverifiable backend is released by the stranded-record +# detector that retires it, not by this counter. +# +# Scope: the cap is per home. A remote secondmate's own lanes live on another +# host and are invisible here, so this counter governs only the lanes this +# home's state directory records. +set -u + +if [ -n "${FM_PROVIDER_LIB_SOURCED:-}" ]; then + return 0 +fi +FM_PROVIDER_LIB_SOURCED=1 + +FM_PROVIDER_LIB_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +# shellcheck source=bin/fm-backend.sh +. "$FM_PROVIDER_LIB_DIR/fm-backend.sh" +# shellcheck source=bin/fm-quota-axi-lib.sh +. "$FM_PROVIDER_LIB_DIR/fm-quota-axi-lib.sh" + +# Fallback cap when config/crew-dispatch.json is absent or declares no +# providerCaps entry. The operator overrides it per provider (or for every +# provider through providerCaps.default) in config/crew-dispatch.json. +FM_PROVIDER_LANE_CAP_DEFAULT=4 + +# fm_provider_for_model +fm_provider_for_model() { # + local harness=${1:-} model=${2:-} segment + case "$model" in + '' | default | -) ;; + */*) + segment=${model%%/*} + case "$segment" in + fireworks_ai | fireworks) printf 'fireworks\n'; return 0 ;; + openai-codex | openai) printf 'codex\n'; return 0 ;; + claude-bridge | anthropic) printf 'claude\n'; return 0 ;; + esac + # An unrecognized provider-qualified prefix is its own billing pool when + # it is a well-formed provider id (the same shape docs/configuration.md + # pins for the resolver's provider fields). + case "$segment" in + '' | -* | *- | *--* | *[!a-z0-9-]*) ;; + *) printf '%s\n' "$segment"; return 0 ;; + esac + ;; + esac + case "$model" in + '' | default | -) ;; + deepseek-v4p1*) printf 'fireworks\n'; return 0 ;; + deepseek*) printf 'deepseek\n'; return 0 ;; + composer* | cursor-*) printf 'cursor\n'; return 0 ;; + grok-*) printf 'grok\n'; return 0 ;; + claude-* | sonnet | haiku | opus | fable) printf 'claude\n'; return 0 ;; + kimi-*) printf 'kimi\n'; return 0 ;; + gemini-*) printf 'gemini\n'; return 0 ;; + muse*) printf 'meta\n'; return 0 ;; + codex* | gpt-* | o[0-9]*) printf 'codex\n'; return 0 ;; + esac + if fm_quota_provider_for_harness "$harness" 2>/dev/null; then + return 0 + fi + if fm_quota_single_provider_for_harness "$harness" 2>/dev/null; then + return 0 + fi + return 1 +} + +# fm_lane_provider +# The bucket a lane is counted against: the model-derived provider when one is +# known, else the harness name so unmapped lanes still count somewhere. +fm_lane_provider() { # + local harness=${1:-} + if fm_provider_for_model "$1" "${2:-}" 2>/dev/null; then + return 0 + fi + [ -n "$harness" ] || return 1 + printf '%s\n' "$harness" +} + +# fm_provider_cap_for [] +fm_provider_cap_for() { # [] + local provider=${1:-} config=${2:-} cap='' + if [ -n "$config" ] && [ -r "$config/crew-dispatch.json" ] && command -v jq >/dev/null 2>&1; then + cap=$(jq -r --arg p "$provider" ' + (.providerCaps // {}) as $c + | ($c[$p] // $c.default // empty) + | select(type == "number" and . >= 1 and . == floor) + ' "$config/crew-dispatch.json" 2>/dev/null || true) + fi + case "$cap" in + '' | *[!0-9]*) cap=$FM_PROVIDER_LANE_CAP_DEFAULT ;; + esac + printf '%s\n' "$cap" +} + +# fm_provider_lane_counts [] [] +fm_provider_lane_counts() { # [] [] + local state=${1:-} config=${2:-} exclude=${3:-} + local meta id harness model provider backend target endpoint_state + [ -n "$state" ] || return 0 + for meta in "$state"/*.meta; do + [ -e "$meta" ] || [ -L "$meta" ] || continue + id=${meta##*/} + id=${id%.meta} + [ -n "$exclude" ] && [ "$id" = "$exclude" ] && continue + harness=$(fm_meta_get "$meta" harness) + model=$(fm_meta_get "$meta" model) + provider=$(fm_lane_provider "$harness" "$model" 2>/dev/null) || continue + [ -n "$provider" ] || continue + backend=$(fm_backend_of_meta "$meta") + target=$(fm_backend_target_of_meta "$meta") + if [ -n "$target" ]; then + endpoint_state=$(fm_backend_agent_state "$backend" "$target" 2>/dev/null || printf 'unreadable') + case "$endpoint_state" in + dead | missing) continue ;; + esac + fi + printf '%s\n' "$provider" + done | LC_ALL=C sort | uniq -c | while read -r used provider; do + printf '%s %s %s\n' "$provider" "$used" "$(fm_provider_cap_for "$provider" "$config")" + done +} + +# fm_provider_cap_refuse [] +fm_provider_cap_refuse() { # [] + local state=${1:-} config=${2:-} harness=${3:-} model=${4:-} exclude=${5:-} + local provider used cap + provider=$(fm_lane_provider "$harness" "$model" 2>/dev/null) || return 0 + [ -n "$provider" ] || return 0 + cap=$(fm_provider_cap_for "$provider" "$config") + used=$(fm_provider_lane_counts "$state" "$config" "$exclude" | awk -v p="$provider" '$1 == p { print $2; exit }') + case "$used" in + '' | *[!0-9]*) used=0 ;; + esac + if [ "$used" -ge "$cap" ]; then + printf 'error: provider lane cap: %s already carries %s live lanes (cap %s), so dispatching harness=%s model=%s would exceed it; reassign to a provider with headroom, or raise providerCaps.%s in %s/crew-dispatch.json (fallback cap %s).\n' \ + "$provider" "$used" "$cap" "${harness:-unknown}" "${model:-default}" \ + "$provider" "${config:-}" "$FM_PROVIDER_LANE_CAP_DEFAULT" >&2 + return 1 + fi + return 0 +} diff --git a/bin/fm-provider-load.sh b/bin/fm-provider-load.sh new file mode 100755 index 00000000000..1d824b5088f --- /dev/null +++ b/bin/fm-provider-load.sh @@ -0,0 +1,45 @@ +#!/usr/bin/env bash +# fm-provider-load.sh - live crewmate/scout/secondmate lanes per billing provider +# against each provider's configured concurrency cap, for dispatch intake. +# +# Usage: +# fm-provider-load.sh +# +# Read-only: it acquires no lock, mutates nothing, and never starts, stops, or +# steers an agent. It reads this home's state/.meta records, resolves each +# lane's provider from its recorded harness and model (the model string wins; +# see bin/fm-provider-lib.sh), and prints one line per provider that currently +# carries a live lane: +# +# provider-load: / +# +# A lane whose recorded endpoint is provably dead or missing does not count, so +# a cleared seat is visible before the next dispatch. With no live lane it prints +# provider-load: no live lanes +# The cap is `providerCaps.` from config/crew-dispatch.json, else +# `providerCaps.default`, else the fallback in bin/fm-provider-lib.sh; that tool +# is the single owner of the counting and cap rules. +# +# Environment: FM_HOME, FM_STATE_OVERRIDE, and FM_CONFIG_OVERRIDE select the home +# and its state/config directories, exactly as the other bin/ scripts do. +set -u + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}}" +STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" +CONFIG="${FM_CONFIG_OVERRIDE:-$FM_HOME/config}" + +# shellcheck source=bin/fm-provider-lib.sh +. "$SCRIPT_DIR/fm-provider-lib.sh" + +counts=$(fm_provider_lane_counts "$STATE" "$CONFIG") +if [ -z "$counts" ]; then + printf 'provider-load: no live lanes\n' + exit 0 +fi +while read -r provider used cap; do + [ -n "$provider" ] || continue + printf 'provider-load: %s %s/%s\n' "$provider" "$used" "$cap" +done <", "model": "", "effort": "" } - ] + ], + "providerCaps": { "default": 4, "fireworks": 4 } } ``` @@ -461,6 +462,13 @@ The resolver supplies the fixed neutral Choice option `No listed rule applies to A rule `floor` names the quota-axi `provider` and `scope` whose `effectivePercentRemaining` must be at least `min_percent` for the rule's profiles to apply. A known percentage below it makes the tool resolve among `default` instead; an absent or unknown row or unmeasured provider makes the floor unverifiable and escalates without authorizing default routing. A profile `provider` optionally names the quota-axi provider family whose rows apply to that profile; when present, profile and rule-floor provider IDs must match the strict whole-string pattern `^[a-z0-9]+(-[a-z0-9]+)*\z`. +`providerCaps` is optional and bounds how many live lanes one billing provider may carry. +`providerCaps.` sets that one provider's cap and `providerCaps.default` sets the cap for every provider without its own entry; an absent entry, or a value below one, falls back to a cap of 4. +The provider a lane is counted against comes from the lane's recorded harness and model, with the model string deciding the identity, so two models on one pool count together even across harnesses while a different pool stays separate. +A lane whose recorded endpoint is provably dead or missing no longer occupies a seat, while a lane whose endpoint cannot be proven gone keeps it. +`bin/fm-provider-load.sh` prints the current per-provider `used/cap` for dispatch intake, and `bin/fm-spawn.sh` refuses a crewmate, scout, or local secondmate spawn that would push a provider past its cap (`bin/fm-provider-lib.sh` is the single owner of both rules). +The cap is per home: a remote secondmate's own lanes are recorded on its host and are not counted here. +Bootstrap rejects a malformed `providerCaps` - a non-object, a key that is neither a provider id nor `default`, or a value that is not a whole number of at least one - with the usual `CREW_DISPATCH:` diagnostic. Bootstrap validates resolver-only `approval`, `floor`, and present `provider` values only while typed resolution is active; without the key those inert fields and the pre-existing verified-harness baseline preserve bootstrap behavior. Typed resolution additively recognizes `gemini` because AGENTS.md section 4 verifies it for crewmate and scout dispatch. The opted-in resolver has authoritative single-provider mappings for `claude`, `codex`, `grok`, `kimi`, `cursor`, `agy`, and `muse`; every other verified harness must declare `provider` explicitly, including multi-provider `pi`, `pi-signed`, `omp`, and `opencode` and unmapped `gemini` and `rovo`. diff --git a/docs/examples/crew-dispatch.json b/docs/examples/crew-dispatch.json index 97c5ad38db1..07b8fe96013 100644 --- a/docs/examples/crew-dispatch.json +++ b/docs/examples/crew-dispatch.json @@ -22,5 +22,6 @@ "default": [ { "harness": "codex", "model": "gpt-5.5", "effort": "medium" }, { "harness": "pi", "model": "anthropic/claude-sonnet-5", "effort": "medium", "provider": "claude" } - ] + ], + "providerCaps": { "default": 4 } } diff --git a/docs/scripts.md b/docs/scripts.md index ef45d68dfa2..2ad239cc045 100644 --- a/docs/scripts.md +++ b/docs/scripts.md @@ -107,6 +107,8 @@ The shared no-mistakes gate refusal for fleet lifecycle entrypoints is summarize | `fm-backlog-transition-lib.sh` | Pair task-record changes with their backlog transitions and replay interrupted closes | | `fm-quota-axi-lib.sh` | Shared `quota-axi` compatibility floor and quota snapshot schema validation | | `fm-quota-choose.sh` | Choose the first candidate with known positive quota from an ordered harness:model list | +| `fm-provider-lib.sh` | Single owner of the model-to-provider identity mapping and the per-provider live-lane concurrency cap | +| `fm-provider-load.sh` | Print live lanes per billing provider against each provider's configured cap, for dispatch intake | | `fm-vendor-auth-probe.sh`| Run one hard-bounded, non-destructive authentication probe of a named vendor CLI and report the fact | | `fm-wake-drain.sh` | Present and acknowledge the current actor's claimed wake rows alongside status, outcome-backstop, decision, divergence, recovery, and supervision checks | | `fm-wake-grant.sh` | Serialize Pi supervision-branch wake-row claim activation, publication, release, and deactivation | diff --git a/tests/fm-bootstrap.test.sh b/tests/fm-bootstrap.test.sh index d8cc824f0dd..026901f97ca 100755 --- a/tests/fm-bootstrap.test.sh +++ b/tests/fm-bootstrap.test.sh @@ -1180,6 +1180,11 @@ default array profile without harness is flagged^{"default":[{"model":"gpt-5.5"} default array malformed effort is flagged^{"default":[{"harness":"codex","effort":3}]}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - default profile model and effort must be non-empty strings, and provider must match ^[a-z0-9]+(-[a-z0-9]+)*\z when present default profile floor without min_percent is flagged^{"default":[{"harness":"codex","floor":{"scope":"all_models"}}]}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - default profile floor needs scope and min_percent 0..100 default profile floor provider override is flagged^{"default":{"harness":"codex","floor":{"scope":"all_models","min_percent":50,"provider":"claude"}}}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - default profile floor needs scope and min_percent 0..100 +provider caps are accepted^{"rules":[],"providerCaps":{"default":4,"fireworks":3}}^empty^ +provider caps zero is flagged^{"rules":[],"providerCaps":{"fireworks":0}}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - providerCaps must map each provider id (or default) to a positive integer +provider caps fractional is flagged^{"rules":[],"providerCaps":{"fireworks":2.5}}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - providerCaps must map each provider id (or default) to a positive integer +provider caps non-object is flagged^{"rules":[],"providerCaps":4}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - providerCaps must map each provider id (or default) to a positive integer +provider caps bad key is flagged^{"rules":[],"providerCaps":{"Fireworks":2}}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - providerCaps must map each provider id (or default) to a positive integer ROWS case_dir="$TMP_ROOT/dispatch-opt-in-gate" diff --git a/tests/fm-provider-lane-cap.test.sh b/tests/fm-provider-lane-cap.test.sh new file mode 100755 index 00000000000..96eb90f65a9 --- /dev/null +++ b/tests/fm-provider-lane-cap.test.sh @@ -0,0 +1,217 @@ +#!/usr/bin/env bash +# Behavior tests for the per-provider lane cap: bin/fm-spawn.sh refuses a +# dispatch that would push a billing provider past its configured cap, and +# bin/fm-provider-load.sh reports the current load for intake. +# +# Every case drives the real spawn CLI with a fake tmux pane and a real +# isolated git worktree, seeds real state/.meta fixtures at known provider +# loads, and asserts the actual accept/refuse outcome and the reported load. +set -u + +# shellcheck source=tests/fixtures.sh +. "$(dirname "${BASH_SOURCE[0]}")/fixtures.sh" + +LOAD="$ROOT/bin/fm-provider-load.sh" +TMP_ROOT=$(fm_test_tmproot fm-provider-lane-cap) + +FIREWORKS_MODEL='fireworks_ai/accounts/fireworks/models/deepseek-v4p1-flash' +DEEPSEEK_MODEL='deepseek-v4.1-flash' +DEEPSEEK_PRO_MODEL='deepseek-v4-pro' + +# make_case [brief-id...] +# Builds home+project+worktree+fakebin plus a brief per id, and echoes +# "||||". +make_case() { + local name=$1 case_dir home proj wt fakebin id + shift + case_dir="$TMP_ROOT/$name" + home="$case_dir/home" + proj="$case_dir/project" + wt="$case_dir/wt" + fakebin=$(fm_test_make_spawn_fakebin "$case_dir/fake") + fm_test_spawn_home "$home" + fm_git_worktree "$proj" "$wt" "wt-$name" + for id in "$@"; do fm_test_spawn_brief "$home" "$id"; done + printf '%s\n' "$case_dir|$home|$proj|$wt|$fakebin" +} + +read_case() { + IFS='|' read -r _ HOME_DIR PROJ_DIR WT_DIR FAKEBIN_DIR < [] +# A minimal real task record: harness and model decide the provider, and an +# optional window target gives the counter an endpoint to classify (a target +# the fake tmux never lists reads provably missing, freeing the seat). +seed_lane() { + local home=$1 id=$2 harness=$3 model=$4 window=${5:-} + { + printf 'harness=%s\n' "$harness" + printf 'model=%s\n' "$model" + printf 'kind=ship\n' + [ -n "$window" ] && printf 'window=%s\n' "$window" + } > "$home/state/$id.meta" +} + +write_caps() { # + mkdir -p "$1/config" + printf '{"rules":[],"providerCaps":%s}\n' "$2" > "$1/config/crew-dispatch.json" +} + +run_ship_spawn() { # [extra...] + local home=$1 wt=$2 fakebin=$3 id=$4 proj=$5 + shift 5 + fm_test_run_spawn "$home" "$wt" "$fakebin" "$id" "$proj" --mode direct-PR --yolo off "$@" +} + +run_load() { # + FM_ROOT_OVERRIDE='' FM_HOME="$1" FM_STATE_OVERRIDE="$1/state" FM_CONFIG_OVERRIDE="$1/config" \ + "$LOAD" 2>&1 +} + +test_under_cap_accepts() { + local rec out status + rec=$(make_case under-cap spawn-under-cap) + read_case "$rec" + seed_lane "$HOME_DIR" lane-fw-1 opencode "$FIREWORKS_MODEL" + seed_lane "$HOME_DIR" lane-fw-2 opencode "$FIREWORKS_MODEL" + seed_lane "$HOME_DIR" lane-fw-3 opencode "$FIREWORKS_MODEL" + + out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" spawn-under-cap "$PROJ_DIR" \ + --harness opencode --model "$FIREWORKS_MODEL") + status=$? + expect_code 0 "$status" "a third lane against a cap of four should spawn" + assert_contains "$out" "spawned spawn-under-cap harness=opencode" "spawn did not report success" + pass "a dispatch under the provider cap is accepted" +} + +test_at_cap_refuses_with_provider_and_load() { + local rec out status + rec=$(make_case at-cap spawn-at-cap) + read_case "$rec" + seed_lane "$HOME_DIR" lane-fw-1 opencode "$FIREWORKS_MODEL" + seed_lane "$HOME_DIR" lane-fw-2 opencode "$FIREWORKS_MODEL" + seed_lane "$HOME_DIR" lane-fw-3 opencode "$FIREWORKS_MODEL" + seed_lane "$HOME_DIR" lane-fw-4 opencode "$FIREWORKS_MODEL" + + out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" spawn-at-cap "$PROJ_DIR" \ + --harness opencode --model "$FIREWORKS_MODEL") + status=$? + expect_code 1 "$status" "the fifth lane against a cap of four should refuse" + assert_contains "$out" "provider lane cap: fireworks already carries 4 live lanes (cap 4)" \ + "refusal did not name the provider and its load" + assert_absent "$HOME_DIR/state/spawn-at-cap.meta" "a refused spawn wrote a task record" + pass "a dispatch at the provider cap is refused before any record is written" +} + +test_one_pool_counts_models_together_and_a_different_pool_stays_separate() { + local rec out status + rec=$(make_case pool-spread spawn-pool-spread) + read_case "$rec" + seed_lane "$HOME_DIR" lane-ds-1 opencode "$DEEPSEEK_MODEL" + seed_lane "$HOME_DIR" lane-ds-2 opencode "$DEEPSEEK_MODEL" + seed_lane "$HOME_DIR" lane-ds-3 opencode "$DEEPSEEK_PRO_MODEL" + seed_lane "$HOME_DIR" lane-ds-4 opencode "$DEEPSEEK_PRO_MODEL" + + out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" spawn-pool-spread "$PROJ_DIR" \ + --harness opencode --model "$DEEPSEEK_PRO_MODEL") + status=$? + expect_code 1 "$status" "two models on one pool must share the cap" + assert_contains "$out" "provider lane cap: deepseek already carries 4 live lanes (cap 4)" \ + "the shared pool was not reported as full" + + rec=$(make_case pool-separate spawn-pool-separate) + read_case "$rec" + seed_lane "$HOME_DIR" lane-ds-1 opencode "$DEEPSEEK_MODEL" + seed_lane "$HOME_DIR" lane-ds-2 opencode "$DEEPSEEK_MODEL" + seed_lane "$HOME_DIR" lane-ds-3 opencode "$DEEPSEEK_PRO_MODEL" + seed_lane "$HOME_DIR" lane-ds-4 opencode "$DEEPSEEK_PRO_MODEL" + + out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" spawn-pool-separate "$PROJ_DIR" \ + --harness opencode --model "$FIREWORKS_MODEL") + status=$? + expect_code 0 "$status" "Fireworks is a different pool and should keep its headroom" + assert_contains "$out" "spawned spawn-pool-separate harness=opencode" \ + "the separate pool did not accept the dispatch" + pass "one pool's models count together while a different pool stays separate" +} + +test_dead_endpoint_frees_a_seat() { + local rec out status + rec=$(make_case dead-seat spawn-dead-seat) + read_case "$rec" + seed_lane "$HOME_DIR" lane-ds-1 opencode "$DEEPSEEK_MODEL" + seed_lane "$HOME_DIR" lane-ds-2 opencode "$DEEPSEEK_MODEL" + seed_lane "$HOME_DIR" lane-ds-3 opencode "$DEEPSEEK_PRO_MODEL" + seed_lane "$HOME_DIR" lane-ds-4 opencode "$DEEPSEEK_PRO_MODEL" 'firstmate:gone' + + out=$(run_load "$HOME_DIR") + assert_contains "$out" 'provider-load: deepseek 3/4' \ + "a provably missing endpoint must not keep occupying a seat" + + out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" spawn-dead-seat "$PROJ_DIR" \ + --harness opencode --model "$DEEPSEEK_MODEL") + status=$? + expect_code 0 "$status" "a freed seat should admit the next dispatch" + assert_contains "$out" "spawned spawn-dead-seat harness=opencode" \ + "the freed seat did not admit the dispatch" + pass "a lane whose endpoint is provably gone no longer holds a seat" +} + +test_operator_cap_config_is_honoured() { + local rec out status + rec=$(make_case cap-config spawn-cap-config) + read_case "$rec" + write_caps "$HOME_DIR" '{"default":1}' + seed_lane "$HOME_DIR" lane-fw-1 opencode "$FIREWORKS_MODEL" + + out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" spawn-cap-config "$PROJ_DIR" \ + --harness opencode --model "$FIREWORKS_MODEL") + status=$? + expect_code 1 "$status" "providerCaps.default of one should refuse a second lane" + assert_contains "$out" "provider lane cap: fireworks already carries 1 live lanes (cap 1)" \ + "the configured default cap was not applied" + + rec=$(make_case cap-raise spawn-cap-raise) + read_case "$rec" + write_caps "$HOME_DIR" '{"fireworks":5}' + seed_lane "$HOME_DIR" lane-fw-1 opencode "$FIREWORKS_MODEL" + seed_lane "$HOME_DIR" lane-fw-2 opencode "$FIREWORKS_MODEL" + seed_lane "$HOME_DIR" lane-fw-3 opencode "$FIREWORKS_MODEL" + seed_lane "$HOME_DIR" lane-fw-4 opencode "$FIREWORKS_MODEL" + + out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" spawn-cap-raise "$PROJ_DIR" \ + --harness opencode --model "$FIREWORKS_MODEL") + status=$? + expect_code 0 "$status" "a raised per-provider cap should admit a fifth lane" + assert_contains "$out" "spawned spawn-cap-raise harness=opencode" \ + "the raised cap did not admit the dispatch" + pass "the cap is operator-editable configuration, not a code constant" +} + +test_load_command_reports_provider_load() { + local rec out + rec=$(make_case load-report) + read_case "$rec" + out=$(run_load "$HOME_DIR") + assert_contains "$out" 'provider-load: no live lanes' "an empty home should report no lanes" + + seed_lane "$HOME_DIR" lane-ds-1 opencode "$DEEPSEEK_MODEL" + seed_lane "$HOME_DIR" lane-ds-2 opencode "$DEEPSEEK_MODEL" + seed_lane "$HOME_DIR" lane-fw-1 opencode "$FIREWORKS_MODEL" + out=$(run_load "$HOME_DIR") + assert_contains "$out" 'provider-load: deepseek 2/4' "the load command did not count the deepseek lane" + assert_contains "$out" 'provider-load: fireworks 1/4' "the load command did not count the fireworks lane" + pass "the intake load command reports live lanes per provider against their caps" +} + +test_under_cap_accepts +test_at_cap_refuses_with_provider_and_load +test_one_pool_counts_models_together_and_a_different_pool_stays_separate +test_dead_endpoint_frees_a_seat +test_operator_cap_config_is_honoured +test_load_command_reports_provider_load + +echo "# all fm-provider-lane-cap tests passed" From 32386dae3a42c5ee8fd36c7d61d9cdffe8c59424 Mon Sep 17 00:00:00 2001 From: keenvc Date: Sun, 20 Sep 2026 01:06:37 +0000 Subject: [PATCH 18/35] feat(bin): add recurring re-verification of aged captain holds Deliver bin/fm-hold-reverify.sh, an armed watcher check that re-checks each captain hold past an age threshold against shipped reality and reports it as dead, still_live, not_a_decision, or unestablishable - the reconciliation vocabulary captain-hold-lifecycle already owns. It reports only: it never calls answer and never closes or annotates a call, so only the captain's own words or an explicit evidence-backed reconciliation can resolve one. Dead is never inferred from absence or an unreadable source. Aged holds come from the canonical local backlog projection (fm-fleet-snapshot.sh --contribution-input); recorded pull requests are read through fm-pr-lib.sh. Each sweep writes a docket and prints one line only when the finding set changes, with a report record keyed on that set. --- AGENTS.md | 3 + bin/fm-hold-reverify.sh | 622 +++++++++++++++++++++++++++++++++ docs/configuration.md | 29 ++ tests/fm-hold-reverify.test.sh | 437 +++++++++++++++++++++++ 4 files changed, 1091 insertions(+) create mode 100755 bin/fm-hold-reverify.sh create mode 100755 tests/fm-hold-reverify.test.sh diff --git a/AGENTS.md b/AGENTS.md index c3634216137..9744ba02473 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -125,6 +125,9 @@ state/ runtime records and signals; gitignored x-watch.check.sh generated Relay poll shim; present only when opted in (section 14) tool-updates.check.sh generated watched-tool update poll shim and its .check-trust binding; present only after bin/fm-tool-update-check.sh arm; its report record .tool-updates is what keeps one pending update from being reported on every poll mail.check.sh generated received-mail poll shim and its .check-trust binding; present only after bin/fm-mail-check.sh arm; report record .mail-check (mail schema: docs/configuration.md "Mail plane") + hold-reverify.check.sh generated aged captain-hold re-verification shim and its .check-trust binding; present only after bin/fm-hold-reverify.sh arm; report record .hold-reverify keeps one finding set from being reported on every sweep + hold-reverify/ generated docket.json: the last sweep's per-hold dead/still_live/not_a_decision/unestablishable findings with their evidence; written only by bin/fm-hold-reverify.sh, which never closes a captain call (docs/configuration.md "Captain-hold re-verification") + .hold-reverify re-verification sweep cadence epoch and finding-set digest; written only by bin/fm-hold-reverify.sh .mail-seen .mail-woken .mail-retry .mail-retry-pos .mail-turn .mail-seen.lock mail-plane poll cursor, emission journal, transient-fetch retry set, retry-scan position, contended-slot turn flag, and overlapping-poll lock; written only by bin/fm-mail.sh (mail schema: docs/configuration.md "Mail plane") pending-replies/ parent-owned secondmate pending-reply records (correlation id, delivery vs reply, recovery, escalation); fm-pending-reply-lib.sh procevent/ registered process-to-event sources, one private record per canonical source id; written only by bin/fm-procevent.sh, and their presence alone keeps supervision required (section 13) diff --git a/bin/fm-hold-reverify.sh b/bin/fm-hold-reverify.sh new file mode 100755 index 00000000000..b6f14d522ad --- /dev/null +++ b/bin/fm-hold-reverify.sh @@ -0,0 +1,622 @@ +#!/usr/bin/env bash +# fm-hold-reverify.sh - recurring re-verification of aged captain-held tasks. +# +# Usage: +# fm-hold-reverify.sh [check] run one bounded re-verification sweep +# fm-hold-reverify.sh classify print the verdict for one hold's facts +# fm-hold-reverify.sh arm write and register the standing check +# fm-hold-reverify.sh disarm remove the standing check +# fm-hold-reverify.sh --help print this help +# +# WHY THIS EXISTS +# A captain call is an ordinary backlog task held for the captain (identity: the +# task id), owned by bin/fm-captain-hold.sh and the captain-hold-lifecycle skill. +# Holds accumulate age: across the fleet hundreds sat behind a hold that nobody +# had ever re-checked, so the captain's list was mostly ghosts and every count of +# remaining work was wrong. The rot concentrates in age. +# +# This script re-checks each aged captain hold against shipped reality and reports +# it in the SAME reconciliation vocabulary captain-hold-lifecycle already owns: +# dead / still_live / not_a_decision / unestablishable. It reports only. It never +# calls `answer` and never calls `reconcile close`/`reconcile note`, so it can +# never close a captain call; only the captain's own words or an explicit +# evidence-backed reconciliation may do that. The point is that the list the +# captain reads is true, not that it is short. +# +# SCHEDULING +# `check` is a plain custom watcher check, not a process-event source. The +# process-event `when` adapter explicitly excludes "an action whose right form +# depends on what the condition finds", and this sweep both classifies each hold +# differently and defers every close to a human, so it stays in the +# check-fires-then-firstmate-decides flow that the process-event-sources skill +# names as the correct home for a plain custom check. `arm` writes +# state/hold-reverify.check.sh and binds its bytes with fm-check-register.sh, so +# the watcher dispatches it on its normal FM_CHECK_INTERVAL cadence and turns its +# one line into a `check:` wake. Session-start-only scanning was rejected: a home +# that never restarts would keep its rot, which is the exact failure being fixed. +# +# THE REPORT IS THE DELIVERABLE +# A sweep writes state/hold-reverify/docket.json (schema fm-hold-reverify-docket.v1) +# listing every examined hold with its verdict, structured evidence, and a short +# reason, and prints ONE line (the wake) only when the finding set changes. +# state/.hold-reverify stores the last sweep's epoch and a digest of the +# {id:verdict} set, mirroring state/.tool-updates, so a new or changed finding +# wakes once while an unchanged sweep stays silent. A sweep killed by the +# watcher's FM_CHECK_TIMEOUT writes no record and is retried. +# +# VERDICT RULES (decided only from structured fields, never from prose) +# not_a_decision the row does not carry a live captain question: it is already +# Done, or it records no hold reason. This is the closed or +# superseded call that still carries the hold annotation. +# dead shipped reality resolves the subject: the row records a +# `merged` completion, or its recorded pull request is merged. +# dead is NEVER inferred from absence or from an unreadable +# source; it requires positive resolution evidence. +# still_live the subject is provably still open: the recorded pull request +# is open. +# unestablishable everything else: no recorded subject, the forge could not be +# read or authenticated, a non-GitHub provider, or a closed +# (unmerged) pull request whose premise is ambiguous. +# The four buckets are total and mutually exclusive, and every result is a +# proposal for reconciliation, not a closure. +# +# WHAT IT READS +# Aged holds come from the canonical local backlog projection rather than a second +# parser: `fm-fleet-snapshot.sh --contribution-input` reuses the canonical backlog +# parser WITHOUT observing workers or other homes, so the sweep stays local and +# bounded. A hold's recorded pull request is read through bin/fm-pr-lib.sh, which +# is the same gh-then-gh-axi path every other surface uses. A redundant local +# origin/main fetch is deliberately NOT performed: the forge merge state and the +# row's own recorded completion are the authoritative landing signals, and a clone +# fetch would add cost and a second source of truth without new signal. +# +# BOUNDS +# AGE FM_HOLD_REVERIFY_AGE_DAYS default 14 (whole days, matching +# FM_SNAPSHOT_UNDATED_HOLD_AGE_DAYS) +# CADENCE FM_HOLD_REVERIFY_INTERVAL default 21600, 0 disables the gate, +# otherwise 60..604800 seconds +# SWEEP FM_HOLD_REVERIFY_BUDGET_SECS default 20, cut to fit FM_CHECK_TIMEOUT +# PROBE FM_HOLD_REVERIFY_PROBE_SECS default 8, valid 1..30 +# COUNT FM_HOLD_REVERIFY_MAX_HOLDS default 12 (whole holds examined per sweep) +set -u +export LC_ALL=C +# A forge read must fail inside its bound rather than stop for credentials. +export GIT_TERMINAL_PROMPT=0 + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}}" +STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" +DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" +CONFIG="${FM_CONFIG_OVERRIDE:-$FM_HOME/config}" +DOCKET_DIR="$STATE/hold-reverify" +DOCKET="$DOCKET_DIR/docket.json" +RECORD="$STATE/.hold-reverify" +CHECK_ID='hold-reverify' +CHECK_SHIM="$STATE/$CHECK_ID.check.sh" +CHECK_TRUST="$STATE/$CHECK_ID.check-trust" +REGISTER_BIN="$SCRIPT_DIR/fm-check-register.sh" +SNAPSHOT_BIN="${FM_HOLD_REVERIFY_SNAPSHOT_BIN:-$SCRIPT_DIR/fm-fleet-snapshot.sh}" +RECORD_SCHEMA=fm-hold-reverify-v1 +DOCKET_SCHEMA=fm-hold-reverify-docket.v1 +MAX_LINE=520 + +# shellcheck source=bin/fm-timeout-lib.sh +. "$SCRIPT_DIR/fm-timeout-lib.sh" +# shellcheck source=bin/fm-pr-lib.sh +. "$SCRIPT_DIR/fm-pr-lib.sh" +# shellcheck source=bin/fm-line-cap-lib.sh +. "$SCRIPT_DIR/fm-line-cap-lib.sh" +# shellcheck source=bin/fm-check-lib.sh +. "$SCRIPT_DIR/fm-check-lib.sh" + +usage() { + cat <<'EOF' +Usage: + fm-hold-reverify.sh [check] run one bounded re-verification sweep + fm-hold-reverify.sh classify print the verdict for one hold's facts + fm-hold-reverify.sh arm write and register state/hold-reverify.check.sh + fm-hold-reverify.sh disarm remove the standing check, its trust binding, and the record + fm-hold-reverify.sh --help print this help + +A sweep re-checks aged captain holds against shipped reality and reports them as +dead, still_live, not_a_decision, or unestablishable. It never closes a captain +call. The docket is written to state/hold-reverify/docket.json. +EOF +} + +die_usage() { + printf 'fm-hold-reverify: %s\n' "$1" >&2 + usage >&2 + exit 2 +} + +# --- configuration ---------------------------------------------------------- + +AGE_DAYS=${FM_HOLD_REVERIFY_AGE_DAYS:-14} +case "$AGE_DAYS" in + ''|*[!0-9]*) die_usage "FM_HOLD_REVERIFY_AGE_DAYS must be a whole number of days" ;; +esac + +INTERVAL=${FM_HOLD_REVERIFY_INTERVAL:-21600} +case "$INTERVAL" in + ''|*[!0-9]*) die_usage "FM_HOLD_REVERIFY_INTERVAL must be 0 or a whole number from 60 to 604800" ;; +esac +if [ "$INTERVAL" -ne 0 ] && { [ "$INTERVAL" -lt 60 ] || [ "$INTERVAL" -gt 604800 ]; }; then + die_usage "FM_HOLD_REVERIFY_INTERVAL must be 0 or a whole number from 60 to 604800" +fi + +BUDGET_SECS=${FM_HOLD_REVERIFY_BUDGET_SECS:-20} +case "$BUDGET_SECS" in + ''|*[!0-9]*|0) die_usage "FM_HOLD_REVERIFY_BUDGET_SECS must be a whole number from 1 to 120" ;; +esac +if [ "$BUDGET_SECS" -gt 120 ]; then + die_usage "FM_HOLD_REVERIFY_BUDGET_SECS must be a whole number from 1 to 120" +fi + +PROBE_SECS=${FM_HOLD_REVERIFY_PROBE_SECS:-8} +case "$PROBE_SECS" in + ''|*[!0-9]*|0) die_usage "FM_HOLD_REVERIFY_PROBE_SECS must be a whole number from 1 to 30" ;; +esac +if [ "$PROBE_SECS" -gt 30 ]; then + die_usage "FM_HOLD_REVERIFY_PROBE_SECS must be a whole number from 1 to 30" +fi + +MAX_HOLDS=${FM_HOLD_REVERIFY_MAX_HOLDS:-12} +case "$MAX_HOLDS" in + ''|*[!0-9]*|0) die_usage "FM_HOLD_REVERIFY_MAX_HOLDS must be a positive whole number" ;; +esac + +# The watcher's per check bound, read from this check's own environment, since the +# watcher runs the check as a direct child. Keep the sweep inside it so a killed +# check does not repeat its silence every cycle. +CHECK_TIMEOUT=${FM_CHECK_TIMEOUT:-30} +case "$CHECK_TIMEOUT" in + ''|*[!0-9]*|0) CHECK_TIMEOUT=30 ;; +esac +PROBE_MIN_SECS=1 +CLOCK_ROUNDING_SECS=1 +KILL_GRACE_SECS=1 +BUDGET_MAX=$((CHECK_TIMEOUT - PROBE_MIN_SECS - CLOCK_ROUNDING_SECS - KILL_GRACE_SECS)) +[ "$BUDGET_MAX" -ge 1 ] || BUDGET_MAX=1 +BUDGET_CUT_FROM= +if [ "$BUDGET_SECS" -gt "$BUDGET_MAX" ]; then + BUDGET_CUT_FROM=$BUDGET_SECS + BUDGET_SECS=$BUDGET_MAX +fi +# The local projection is a fast bounded child of the same sweep budget, so it +# can never consume more than the sweep has left. +SNAPSHOT_BOUND=5 +[ "$SNAPSHOT_BOUND" -le "$BUDGET_SECS" ] || SNAPSHOT_BOUND=$BUDGET_SECS + +# --- small helpers ---------------------------------------------------------- + +# The record epoch is overridable so a test can drive the cadence gate; the +# sweep budget always uses real time so a frozen epoch cannot disable it. +record_epoch_now() { + case "${FM_HOLD_REVERIFY_NOW:-}" in + ''|*[!0-9]*) date +%s ;; + *) printf '%s\n' "$FM_HOLD_REVERIFY_NOW" ;; + esac +} + +real_epoch() { date +%s; } + +digest_of() { + local text=$1 + if command -v shasum >/dev/null 2>&1; then + printf '%s' "$text" | shasum -a 256 | awk '{print $1}' + elif command -v sha256sum >/dev/null 2>&1; then + printf '%s' "$text" | sha256sum | awk '{print $1}' + else + printf '%s' "$text" | cksum | awk '{print $1}' + fi +} + +utc_now() { date -u +%Y-%m-%dT%H:%M:%SZ; } + +# --- report record ---------------------------------------------------------- + +RECORD_EPOCH=0 +RECORD_DIGEST= + +record_read() { + local line first=1 + RECORD_EPOCH=0 + RECORD_DIGEST= + [ -f "$RECORD" ] || return 0 + while IFS= read -r line; do + if [ "$first" = 1 ]; then + first=0 + [ "$line" = "$RECORD_SCHEMA" ] || return 0 + continue + fi + case "$line" in + epoch=*) + line=${line#epoch=} + case "$line" in + ''|*[!0-9]*) RECORD_EPOCH=0 ;; + *) RECORD_EPOCH=$line ;; + esac + ;; + findings=*) RECORD_DIGEST=${line#findings=} ;; + esac + done < "$RECORD" + return 0 +} + +record_write() { + local digest=$1 tmp + tmp=$(mktemp "$RECORD.XXXXXX" 2>/dev/null) || return 1 + chmod 0600 "$tmp" 2>/dev/null || { rm -f -- "$tmp"; return 1; } + { + printf '%s\n' "$RECORD_SCHEMA" + printf 'epoch=%s\n' "$(record_epoch_now)" + printf 'findings=%s\n' "$digest" + } > "$tmp" || { rm -f -- "$tmp"; return 1; } + mv -f -- "$tmp" "$RECORD" || { rm -f -- "$tmp"; return 1; } + return 0 +} + +# --- classifier ------------------------------------------------------------- + +# classify_facts : print the one verdict for a hold's structured +# facts. Pure: no clock, no network, no filesystem. This is the seam the tests +# drive, and action_check drives it too, so both paths classify identically. +classify_facts() { + printf '%s\n' "$1" | jq -r ' + if (.state == "done") or ((.hold_reason // "") == "") then "not_a_decision" + elif (.completion_merged == true) or (.pr_state == "merged") then "dead" + elif (.pr_state == "open") then "still_live" + else "unestablishable" end' +} + +# read_record_bounded : print " " +# from bin/fm-pr-lib.sh under a hard bound, or nothing. The bounded child sources +# the lib itself so the global readout survives the process boundary. +read_record_bounded() { + local owner=$1 repo=$2 number=$3 bound=$4 + # shellcheck disable=SC2016 # The child sources the lib and expands its own positionals. + fm_run_timed "$bound" bash -c ' + . "$1" || exit 1 + fm_pr_github_read_record "$2" "$3" "$4" || exit 1 + printf "%s %s\n" "$FM_PR_RECORD_STATE" "$FM_PR_RECORD_MERGED" + ' _ "$SCRIPT_DIR/fm-pr-lib.sh" "$owner" "$repo" "$number" +} + +# gather_facts : emit one facts object for a selected hold. +gather_facts() { + local hold=$1 id state reason pr_url merged pr_state=none owner repo number out record_state + id=$(printf '%s\n' "$hold" | jq -r '.id // ""') + state=$(printf '%s\n' "$hold" | jq -r '.state // ""') + reason=$(printf '%s\n' "$hold" | jq -r '.hold_reason // ""') + pr_url=$(printf '%s\n' "$hold" | jq -r '.pr_url // ""') + merged=$(printf '%s\n' "$hold" | jq -r 'if .completion_merged == true then "true" else "false" end') + if [ -n "$pr_url" ]; then + if fm_pr_url_parse "$pr_url" && [ "$FM_PR_PROVIDER" = github ] \ + && [ -n "$FM_PR_OWNER" ] && [ -n "$FM_PR_REPO" ] && [ -n "$FM_PR_NUMBER" ]; then + owner=$FM_PR_OWNER + repo=$FM_PR_REPO + number=$FM_PR_NUMBER + if out=$(read_record_bounded "$owner" "$repo" "$number" "$PROBE_SECS"); then + record_state=${out%% *} + case "$record_state" in + MERGED) pr_state=merged ;; + OPEN) pr_state=open ;; + CLOSED) pr_state=closed ;; + *) pr_state=unreadable ;; + esac + else + pr_state=unreadable + fi + else + pr_state=unreadable + fi + fi + jq -cn \ + --arg id "$id" \ + --arg state "$state" \ + --arg hold_reason "$reason" \ + --arg pr_url "$pr_url" \ + --arg pr_state "$pr_state" \ + --argjson completion_merged "$merged" \ + '{id:$id,state:$state,hold_reason:$hold_reason,pr_url:$pr_url, + pr_state:$pr_state,completion_merged:$completion_merged}' +} + +# --- the sweep -------------------------------------------------------------- + +FINDINGS_FILE= +EXAMINED=0 +DEFERRED=0 +DEADLINE=0 +SNAPSHOT_ERROR= + +budget_exhausted() { [ "$(real_epoch)" -ge "$DEADLINE" ]; } + +sweep_cleanup() { + [ -z "$FINDINGS_FILE" ] || rm -f -- "$FINDINGS_FILE" + FINDINGS_FILE= +} + +# snapshot_holds: print one compact JSON object per aged captain hold, or nothing. +snapshot_holds() { + local snapshot + snapshot=$(FM_HOME="$FM_HOME" FM_STATE_OVERRIDE="$STATE" FM_DATA_OVERRIDE="$DATA" \ + FM_CONFIG_OVERRIDE="$CONFIG" \ + fm_run_timed "$SNAPSHOT_BOUND" "$SNAPSHOT_BIN" --contribution-input 2>/dev/null) || return 1 + [ -n "$snapshot" ] || return 1 + printf '%s\n' "$snapshot" | jq -c --argjson age "$AGE_DAYS" ' + (.backlog.records // [])[] + | select(.structured == true) + | select(.hold_kind == "captain") + | select(.hold_age_days != null and .hold_age_days >= $age) + | {id, title, state, hold_reason, hold_age_days, pr_url, + completion_merged: (.completion.verb == "merged")}' || return 1 +} + +action_check() { + [ -d "$STATE" ] || return 0 + record_read + local now + now=$(record_epoch_now) + if [ "$INTERVAL" -ne 0 ] && [ "$RECORD_EPOCH" -gt 0 ] \ + && [ "$now" -ge "$RECORD_EPOCH" ] && [ $((now - RECORD_EPOCH)) -lt "$INTERVAL" ]; then + return 0 + fi + + DEADLINE=$(( $(real_epoch) + BUDGET_SECS )) + FINDINGS_FILE=$(mktemp "${TMPDIR:-/tmp}/fm-hold-reverify.XXXXXX") || return 0 + : > "$FINDINGS_FILE" + EXAMINED=0 + DEFERRED=0 + SNAPSHOT_ERROR= + + local holds='' hold facts verdict + if ! holds=$(snapshot_holds); then + SNAPSHOT_ERROR="could not read the aged-hold projection" + else + while IFS= read -r hold; do + [ -n "$hold" ] || continue + if [ "$EXAMINED" -ge "$MAX_HOLDS" ] || budget_exhausted; then + DEFERRED=$((DEFERRED + 1)) + continue + fi + EXAMINED=$((EXAMINED + 1)) + facts=$(gather_facts "$hold") + verdict=$(classify_facts "$facts") + printf '%s\n' "$facts" \ + | jq -c --arg v "$verdict" '. + {verdict:$v}' >> "$FINDINGS_FILE" + done </dev/null) || counts='{}' + dead=$(printf '%s\n' "$counts" | jq -r '.dead // 0') + live=$(printf '%s\n' "$counts" | jq -r '.still_live // 0') + notdec=$(printf '%s\n' "$counts" | jq -r '.not_a_decision // 0') + unest=$(printf '%s\n' "$counts" | jq -r '.unestablishable // 0') + examined=$EXAMINED + deferred=$DEFERRED + + if [ -d "$DOCKET_DIR" ] || mkdir -p "$DOCKET_DIR"; then + if jq -s --arg generated "$generated" --arg schema "$DOCKET_SCHEMA" \ + --argjson age "$AGE_DAYS" --argjson examined "$examined" --argjson deferred "$deferred" \ + '{schema:$schema, generated:$generated, threshold_days:$age, + examined:$examined, deferred:$deferred, + counts:{dead:([.[]|select(.verdict=="dead")]|length), + still_live:([.[]|select(.verdict=="still_live")]|length), + not_a_decision:([.[]|select(.verdict=="not_a_decision")]|length), + unestablishable:([.[]|select(.verdict=="unestablishable")]|length)}, + findings:(sort_by(.id))}' \ + "$FINDINGS_FILE" > "$DOCKET.tmp" 2>/dev/null; then + mv -f -- "$DOCKET.tmp" "$DOCKET" || rm -f -- "$DOCKET.tmp" + else + rm -f -- "$DOCKET.tmp" + fi + fi + + # The digest keys the record on the finding set, so a new hold aging in or a + # verdict changing is news while an unchanged sweep stays silent. + digest=$(sort "$FINDINGS_FILE" | jq -sc '[.[] | .id + ":" + .verdict] | sort | join(",")' 2>/dev/null) + digest=$(digest_of "${digest:-}") + + summary= + line= + if [ -n "$SNAPSHOT_ERROR" ]; then + summary="hold re-verify: $SNAPSHOT_ERROR" + digest=$(digest_of "error:$SNAPSHOT_ERROR") + elif [ "$examined" -gt 0 ] || [ "$deferred" -gt 0 ]; then + summary="hold re-verify: $dead dead, $live still live, $notdec not-a-decision, $unest unestablishable among $examined aged captain holds" + [ "$deferred" -eq 0 ] || summary="$summary ($deferred deferred)" + summary="$summary; docket state/hold-reverify/docket.json" + fi + if [ -n "$summary" ]; then + fm_cap_line_var "$summary" "$MAX_LINE" + line=$FM_LINE_CAP_LINE + fi + + if [ -n "$BUDGET_CUT_FROM" ] && [ -n "$line" ]; then + line="$line [budget ${BUDGET_CUT_FROM}s cut to ${BUDGET_SECS}s]" + fi + + if [ -n "$line" ] && [ "$digest" != "$RECORD_DIGEST" ]; then + printf '%s\n' "$line" + fi + record_write "$digest" || true +} + +# --- arming ----------------------------------------------------------------- + +# The home is embedded already resolved, because the watcher runs the shim from +# its own working directory and a relative spelling would send the check to a +# different home, or to none at all. +shim_content() { + local home=$1 + printf '%s\n' \ + '#!/usr/bin/env bash' \ + '# Auto-generated by fm-hold-reverify.sh - aged captain-hold re-verification.' \ + '# The watcher validates these bytes, then dispatches the trusted check script.' \ + "export FM_HOME=$(printf '%q' "$home")" \ + "exec $(printf '%q' "$SCRIPT_DIR/fm-hold-reverify.sh") check" +} + +SHIM_WRITE_TMP= +ARM_BACKUP= + +shim_write() { + local want=$1 device tmp + [ -d "$STATE" ] && [ ! -L "$STATE" ] || return 1 + device=$(fm_pr_file_device "$STATE") || return 1 + [ -n "$device" ] || return 1 + fm_pr_regular_destination_on_device_or_absent "$CHECK_SHIM" "$device" || return 1 + if [ -e "$CHECK_SHIM" ] && [ "$(fm_pr_file_mode "$CHECK_SHIM")" = 700 ] \ + && [ "$(cat "$CHECK_SHIM" 2>/dev/null)" = "$want" ]; then + return 0 + fi + tmp=$(umask 077; mktemp "$STATE/.fm-hold-reverify-check.XXXXXX" 2>/dev/null) || return 1 + SHIM_WRITE_TMP=$tmp + if ! printf '%s\n' "$want" > "$tmp" \ + || ! chmod 0700 "$tmp" \ + || ! fm_pr_private_file_valid "$tmp" 700 "$device"; then + rm -f -- "$tmp" + SHIM_WRITE_TMP= + return 1 + fi + if ! fm_pr_regular_destination_on_device_or_absent "$CHECK_SHIM" "$device" \ + || ! mv -f -- "$tmp" "$CHECK_SHIM"; then + rm -f -- "$tmp" + SHIM_WRITE_TMP= + return 1 + fi + SHIM_WRITE_TMP= + fm_pr_private_file_valid "$CHECK_SHIM" 700 "$device" +} + +shim_backup() { + local device tmp + device=$(fm_pr_file_device "$STATE") || return 1 + [ -n "$device" ] || return 1 + tmp=$(umask 077; mktemp "$STATE/.fm-hold-reverify-check.XXXXXX" 2>/dev/null) || return 1 + if ! cat "$CHECK_SHIM" > "$tmp" 2>/dev/null \ + || ! chmod 0700 "$tmp" \ + || ! fm_pr_private_file_valid "$tmp" 700 "$device"; then + rm -f -- "$tmp" + return 1 + fi + printf '%s\n' "$tmp" +} + +# An unregistered shim is not inert: the watcher rejects it every cycle and wakes +# about unauthenticated state checks. After a failed or interrupted arm the home +# must never hold a shim without a matching trust binding. +arm_rollback() { + [ -z "$SHIM_WRITE_TMP" ] || rm -f -- "$SHIM_WRITE_TMP" + SHIM_WRITE_TMP= + if [ -n "$ARM_BACKUP" ]; then + mv -f -- "$ARM_BACKUP" "$CHECK_SHIM" 2>/dev/null || rm -f -- "$ARM_BACKUP" + ARM_BACKUP= + if fm_custom_check_registered "$STATE" "$CHECK_ID"; then + return 0 + fi + fi + rm -f -- "$CHECK_SHIM" +} + +# shellcheck disable=SC2329 # Registered by action_arm's signal trap. +arm_interrupted() { + arm_rollback + printf 'fm-hold-reverify: arming was interrupted, so state/%s.check.sh is not armed\n' "$CHECK_ID" >&2 + exit 1 +} + +action_arm() { + local want home + mkdir -p "$STATE" || return 1 + case "$FM_HOME" in + /*) home=$FM_HOME ;; + *) + home=$(CDPATH='' cd -- "$FM_HOME" 2>/dev/null && pwd -P) || { + printf 'fm-hold-reverify: cannot resolve FM_HOME %s\n' "$FM_HOME" >&2 + return 1 + } + ;; + esac + want=$(shim_content "$home") + ARM_BACKUP= + if [ -f "$CHECK_SHIM" ] && [ ! -L "$CHECK_SHIM" ]; then + ARM_BACKUP=$(shim_backup) || { + printf 'fm-hold-reverify: could not save the existing %s\n' "$CHECK_SHIM" >&2 + return 1 + } + fi + trap arm_interrupted HUP INT TERM + if ! shim_write "$want"; then + trap - HUP INT TERM + arm_rollback + printf 'fm-hold-reverify: could not write %s\n' "$CHECK_SHIM" >&2 + return 1 + fi + if ! FM_HOME="$home" "$REGISTER_BIN" "$CHECK_ID" >/dev/null; then + trap - HUP INT TERM + arm_rollback + printf 'fm-hold-reverify: could not register %s\n' "$CHECK_SHIM" >&2 + return 1 + fi + trap - HUP INT TERM + [ -z "$ARM_BACKUP" ] || rm -f -- "$ARM_BACKUP" + ARM_BACKUP= + printf 'armed: state/%s.check.sh\n' "$CHECK_ID" + return 0 +} + +action_disarm() { + rm -f -- "$CHECK_SHIM" "$CHECK_TRUST" "$RECORD" + printf 'disarmed: state/%s.check.sh\n' "$CHECK_ID" + return 0 +} + +# --- dispatch --------------------------------------------------------------- + +case "${1:-check}" in + check) + [ "$#" -le 1 ] || die_usage "check takes no arguments" + action_check + ;; + classify) + [ "$#" -eq 2 ] || die_usage "classify requires one facts JSON file" + [ -f "$2" ] && [ ! -L "$2" ] || die_usage "classify facts file is unavailable: $2" + classify_facts "$(cat "$2")" + ;; + arm) + [ "$#" -eq 1 ] || die_usage "arm takes no arguments" + action_arm + ;; + disarm) + [ "$#" -eq 1 ] || die_usage "disarm takes no arguments" + action_disarm + ;; + -h|--help) + usage + ;; + *) + die_usage "unknown command: $1" + ;; +esac diff --git a/docs/configuration.md b/docs/configuration.md index 808ee716baa..e36e794851f 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -623,6 +623,29 @@ The sweep must finish inside `FM_CHECK_TIMEOUT` (default 30), because a run the So a budget larger than that timeout allows is cut down to what fits instead of being refused, and the cut is reported in the report line. A budget that is not a whole number from 1 to 120 is still refused outright. +## Captain-hold re-verification + +A captain call is an ordinary backlog task held for the captain, and its list rots with age: hundreds of holds had never been re-checked, so the captain's list was mostly ghosts and every count of remaining work was wrong. +`bin/fm-hold-reverify.sh` re-checks each aged hold against shipped reality and reports it in the reconciliation vocabulary `captain-hold-lifecycle` already owns: `dead`, `still_live`, `not_a_decision`, or `unestablishable`. +It reports only. +It never calls `answer` and never closes or annotates a call, so only the captain's own words or an explicit evidence-backed reconciliation can resolve one. +A hold is `dead` when shipped reality resolves the subject - its recorded pull request is merged, or its row records a merged completion. +It is `still_live` when the recorded pull request is open, `not_a_decision` when the row carries no live captain question (already Done, or no hold reason), and `unestablishable` otherwise. +`dead` is never inferred from absence or from an unreadable source, and a closed-unmerged pull request stays `unestablishable` rather than reading as dead. +Aged holds come from the canonical local backlog projection (`fm-fleet-snapshot.sh --contribution-input`), and a recorded pull request is read through `bin/fm-pr-lib.sh`; no second backlog parser and no redundant `origin/main` clone fetch are involved. + +`check` is a plain custom watcher check, so it stays in the check-fires-then-firstmate-decides flow that the process-event `when` adapter explicitly excludes for an action whose right form depends on what the condition finds. +Arm it once per home with `bin/fm-hold-reverify.sh arm`, which writes `state/hold-reverify.check.sh` and binds its bytes with `bin/fm-check-register.sh` so the watcher dispatches it on its normal cadence and turns its one line into a `check:` wake. +`disarm` removes the shim, its trust binding, and the report record. +Each sweep writes `state/hold-reverify/docket.json` (schema `fm-hold-reverify-docket.v1`) with every examined hold's verdict, evidence, and reason, and prints one line only when the finding set changes. +`state/.hold-reverify` records the sweep epoch and a digest of the `{id: verdict}` set, so a new or changed finding is reported once while an unchanged sweep stays silent. +A sweep the watcher kills writes no record and is retried. + +`FM_HOLD_REVERIFY_AGE_DAYS` (default 14, matching `FM_SNAPSHOT_UNDATED_HOLD_AGE_DAYS`) sets the age at which a hold is re-verified. +`FM_HOLD_REVERIFY_INTERVAL` (default 21600 seconds, `0` to sweep on every watcher cycle) gates how often a sweep actually runs. +`FM_HOLD_REVERIFY_BUDGET_SECS` (default 20) bounds a whole sweep and is cut to fit `FM_CHECK_TIMEOUT`, with the cut reported in the wake line. +`FM_HOLD_REVERIFY_PROBE_SECS` (default 8) bounds one forge read, and `FM_HOLD_REVERIFY_MAX_HOLDS` (default 12) caps the holds examined per sweep, deferring the rest and disclosing the count. + ## Mail plane (.env) The mail plane (bin/fm-mail.sh) reads unseen IMAP messages and sends one SMTP message. @@ -1077,6 +1100,12 @@ FM_TOOL_UPDATE_INTERVAL=900 # seconds between watched-tool probe sweeps; 0 pro FM_TOOL_UPDATE_PROBE_SECS=5 # 1..30 seconds allowed for one version or git probe FM_TOOL_UPDATE_BUDGET_SECS=20 # 1..120 seconds allowed for a whole watched-tool sweep; cut to fit FM_CHECK_TIMEOUT, and the cut is reported FM_TOOL_UPDATE_NOW= # test override for the watched-tool sweep clock; the sweep budget still uses real time +FM_HOLD_REVERIFY_AGE_DAYS=14 # floored elapsed-day age at which a captain hold is re-verified against shipped reality; 0 re-verifies every hold with a non-negative age +FM_HOLD_REVERIFY_INTERVAL=21600 # seconds between re-verification sweeps; 0 sweeps every watcher cycle, other values must be 60..604800 +FM_HOLD_REVERIFY_BUDGET_SECS=20 # 1..120 seconds allowed for a whole sweep; cut to fit FM_CHECK_TIMEOUT, and the cut is reported +FM_HOLD_REVERIFY_PROBE_SECS=8 # 1..30 seconds allowed for one forge read +FM_HOLD_REVERIFY_MAX_HOLDS=12 # whole holds examined per sweep; the remainder is deferred and its count disclosed +FM_HOLD_REVERIFY_NOW= # test override for the re-verification cadence clock; the sweep budget still uses real time FM_PROCEVENT_MAX_OUTPUT_BYTES=1048576 # bound on one captured process-to-event result FM_PROCEVENT_CLAIM_ROOT= # machine-wide source claim root; default $XDG_STATE_HOME/firstmate/procevent-claims FM_PROCEVENT_OWNER_LEASE_SECONDS=600 # how long a source runner keeps going with no activity in its owning home; 1..86400 diff --git a/tests/fm-hold-reverify.test.sh b/tests/fm-hold-reverify.test.sh new file mode 100755 index 00000000000..42928de4125 --- /dev/null +++ b/tests/fm-hold-reverify.test.sh @@ -0,0 +1,437 @@ +#!/usr/bin/env bash +# Behavior tests for bin/fm-hold-reverify.sh, the recurring re-verification of +# aged captain-held backlog tasks. +# +# Every case drives the executable interface: it builds a fixture home whose +# data/backlog.md holds aged captain rows with a known ground truth, fakes the +# forge (a PATH `gh` answering `api graphql` from a per-home state file) so no +# case ever contacts a network, and runs the real `check`/`classify`/`arm`/ +# `disarm` commands. Assertions read the resulting docket and the printed wake +# line, never any implementation source byte, so a rewrite that keeps the +# behavior passes. +# +# The ground truths the classifier must separate: +# a merged pull request -> dead +# a still-open question (open PR) -> still_live +# a superseded/closed finding -> not_a_decision +# no readable subject -> unestablishable +set -u + +# shellcheck source=tests/lib.sh +# shellcheck disable=SC1091 +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +CHECK="$ROOT/bin/fm-hold-reverify.sh" +TMP_ROOT=$(fm_test_tmproot fm-hold-reverify) +FIXED_NOW=2026-09-20T00:00:00Z +OLD_HOLD_SET=2026-01-01T00:00:00Z +YOUNG_HOLD_SET=2026-09-19T00:00:00Z + +command -v jq >/dev/null 2>&1 || { echo "skip: jq not found"; exit 0; } + +# make_home : a scratch home with an empty backlog and a forge fake. +make_home() { + local name=$1 home fakebin + home="$TMP_ROOT/$name" + mkdir -p "$home/data" "$home/state" "$home/config" "$home/fakebin" "$home/forge" + cp "$ROOT/.tasks.toml" "$home/.tasks.toml" + cat > "$home/data/backlog.md" <<'EOF' +## In flight + +## Queued + +## Done +EOF + fakebin="$home/fakebin" + # A fake forge: the state of pull request is read from $FM_TEST_FORGE_DIR/ + # as two fields " ". A first field of FAIL makes the read fail, + # standing in for an unauthenticated or unreachable forge. Applied only when + # invoked as `gh api graphql -F number=`. + cat > "$fakebin/gh" <<'SH' +#!/usr/bin/env bash +case "${1:-} ${2:-}" in + "api graphql") + number= prev= + for arg in "$@"; do + if [ "$prev" = "-F" ]; then + case "$arg" in number=*) number=${arg#number=} ;; esac + fi + prev=$arg + done + [ -n "$number" ] || exit 1 + fixture="${FM_TEST_FORGE_DIR:-}/$number" + [ -f "$fixture" ] || exit 1 + read -r pr_state merged < "$fixture" + [ "$pr_state" != FAIL ] || exit 1 + printf 'state=%s\nmerged=%s\n' "$pr_state" "$merged" + ;; + *) exit 1 ;; +esac +SH + chmod +x "$fakebin/gh" + printf '%s\n' "$home" +} + +# write_backlog : the whole backlog comes from stdin, so a case can name +# exactly the rows and ground truth it needs. +write_backlog() { + local home=$1 + cat > "$home/data/backlog.md" +} + +# set_pr : the faked forge answer for one PR. +set_pr() { + local home=$1 number=$2 state=$3 merged=$4 + printf '%s %s\n' "$state" "$merged" > "$home/forge/$number" +} + +# run_check [extra env KEY=VAL...]: run one sweep with a pinned +# snapshot clock, no cadence gate, and the fixture forge. +run_check() { + local home=$1 out=$2 status=0 + shift 2 + env FM_CHECK_TIMEOUT=30 \ + FM_SNAPSHOT_NOW="$FIXED_NOW" \ + FM_HOLD_REVERIFY_INTERVAL=0 \ + FM_TEST_FORGE_DIR="$home/forge" \ + FM_HOME="$home" \ + PATH="$home/fakebin:$PATH" \ + "$@" "$CHECK" check >"$out" 2>&1 || status=$? + printf '%s\n' "$status" +} + +# docket_verdict : the verdict recorded for one hold. +docket_verdict() { + jq -r --arg id "$2" '.findings[] | select(.id == $id) | .verdict' \ + "$1/state/hold-reverify/docket.json" 2>/dev/null +} + +# --- classifier: four ground truths ----------------------------------------- + +test_merged_pull_request_reports_dead() { + local home out + home=$(make_home merged) + set_pr "$home" 101 MERGED true + write_backlog "$home" </dev/null \ + || fail "a young hold must not be examined" + pass "fm-hold-reverify: a hold younger than the threshold is ignored" +} + +# --- reporting contract ------------------------------------------------------ + +test_repeat_is_silent_and_change_wakes() { + local home out status + home=$(make_home repeat) + set_pr "$home" 105 MERGED true + write_backlog "$home" < "$out" + status=$(run_check "$home" "$out") + expect_code 0 "$status" "repeat sweep exit" + [ ! -s "$out" ] || fail "an unchanged finding set must stay silent: $(cat "$out")" + + # The subject opens again: a changed verdict is news. + set_pr "$home" 105 OPEN false + : > "$out" + status=$(run_check "$home" "$out") + expect_code 0 "$status" "changed sweep exit" + assert_contains "$(cat "$out")" "1 still live" "a changed verdict wakes once" + pass "fm-hold-reverify: an unchanged sweep is silent and a changed verdict wakes" +} + +test_cadence_gate_suppresses_until_the_interval_elapses() { + local home out status + home=$(make_home cadence) + set_pr "$home" 106 MERGED true + write_backlog "$home" < "$out" + status=$(run_check "$home" "$out" FM_HOLD_REVERIFY_INTERVAL=21600) + expect_code 0 "$status" "cadence immediate sweep exit" + [ ! -s "$out" ] || fail "a sweep inside the cadence interval must stay silent: $(cat "$out")" + + # Past the interval the gate opens and the changed finding is reported. + : > "$out" + status=$(run_check "$home" "$out" FM_HOLD_REVERIFY_INTERVAL=21600 \ + FM_HOLD_REVERIFY_NOW=$(( $(date +%s) + 1000000 ))) + expect_code 0 "$status" "cadence advanced sweep exit" + assert_contains "$(cat "$out")" "1 still live" "past the interval the changed finding wakes" + pass "fm-hold-reverify: the cadence gate suppresses sweeps until the interval elapses" +} + +test_sweep_defers_beyond_the_hold_cap() { + local home out i + home=$(make_home cap) + { + printf '## In flight\n\n## Queued\n\n' + for i in 1 2 3; do + printf -- '- [ ] h-cap%s - A question with no artifact (repo: sample) (kind: captain) (since 2026-01-01) (hold: pick) (hold-kind: captain)\n Captain hold set: %s\n' "$i" "$OLD_HOLD_SET" + done + printf '\n## Done\n' + } > "$home/data/backlog.md" + out="$home/out" + expect_code 0 "$(run_check "$home" "$out" FM_HOLD_REVERIFY_MAX_HOLDS=2)" "capped sweep exit" + assert_equals 2 "$(jq -r '.examined' "$home/state/hold-reverify/docket.json")" \ + "the sweep examines no more than the cap" + assert_equals 1 "$(jq -r '.deferred' "$home/state/hold-reverify/docket.json")" \ + "the remainder is recorded as deferred, not silently dropped" + assert_contains "$(cat "$out")" "1 deferred" "the wake line discloses the deferred remainder" + pass "fm-hold-reverify: the hold cap bounds the sweep and discloses what it deferred" +} + +test_unreadable_projection_reports_once() { + local home out broken status + home=$(make_home broken) + broken="$home/broken-snapshot.sh" + cat > "$broken" <<'SH' +#!/usr/bin/env bash +exit 1 +SH + chmod +x "$broken" + out="$home/out" + status=$(run_check "$home" "$out" FM_HOLD_REVERIFY_SNAPSHOT_BIN="$broken") + expect_code 0 "$status" "broken projection sweep exit" + assert_contains "$(cat "$out")" "could not read the aged-hold projection" \ + "an unreadable projection is reported rather than silently passing" + + : > "$out" + run_check "$home" "$out" FM_HOLD_REVERIFY_SNAPSHOT_BIN="$broken" >/dev/null + [ ! -s "$out" ] || fail "the same projection failure must not repeat every sweep: $(cat "$out")" + pass "fm-hold-reverify: an unreadable projection is reported once" +} + +# --- classify seam ----------------------------------------------------------- + +test_classify_prints_the_verdict_for_facts() { + local home facts + home=$(make_home classify) + facts="$home/facts.json" + + printf '%s\n' '{"state":"queued","hold_reason":"q","pr_state":"merged","completion_merged":false}' > "$facts" + assert_equals dead "$("$CHECK" classify "$facts")" "merged facts classify dead" + + printf '%s\n' '{"state":"queued","hold_reason":"q","pr_state":"open","completion_merged":false}' > "$facts" + assert_equals still_live "$("$CHECK" classify "$facts")" "open facts classify still_live" + + printf '%s\n' '{"state":"queued","hold_reason":"","pr_state":"none","completion_merged":false}' > "$facts" + assert_equals not_a_decision "$("$CHECK" classify "$facts")" "a questionless record is not a decision" + + printf '%s\n' '{"state":"done","hold_reason":"q","pr_state":"none","completion_merged":false}' > "$facts" + assert_equals not_a_decision "$("$CHECK" classify "$facts")" "a closed record is not a live decision" + + printf '%s\n' '{"state":"queued","hold_reason":"q","pr_state":"closed","completion_merged":true}' > "$facts" + assert_equals dead "$("$CHECK" classify "$facts")" "a recorded merged completion is dead" + + printf '%s\n' '{"state":"queued","hold_reason":"q","pr_state":"none","completion_merged":false}' > "$facts" + assert_equals unestablishable "$("$CHECK" classify "$facts")" "facts with no evidence are unestablishable" + pass "fm-hold-reverify: classify reports the right verdict for each fact shape" +} + +# --- arming ------------------------------------------------------------------ + +test_arm_writes_and_registers_and_disarm_removes() { + local home status + home=$(make_home arm) + status=0 + FM_HOME="$home" FM_HOLD_REVERIFY_AGE_DAYS=14 "$CHECK" arm >/dev/null 2>&1 || status=$? + expect_code 0 "$status" "arm exit" + assert_present "$home/state/hold-reverify.check.sh" "arm writes the check shim" + assert_present "$home/state/hold-reverify.check-trust" "arm binds the shim bytes" + bash -c ' + . "$1/bin/fm-pr-lib.sh" + . "$1/bin/fm-check-lib.sh" + fm_custom_check_registered "$2" hold-reverify + ' _ "$ROOT" "$home/state" || fail "arm must register the shim with a matching trust binding" + + status=0 + FM_HOME="$home" "$CHECK" disarm >/dev/null 2>&1 || status=$? + expect_code 0 "$status" "disarm exit" + assert_absent "$home/state/hold-reverify.check.sh" "disarm removes the check shim" + assert_absent "$home/state/hold-reverify.check-trust" "disarm removes the trust binding" + pass "fm-hold-reverify: arm writes and binds the standing check and disarm removes it" +} + +test_help_and_usage() { + local status=0 + "$CHECK" --help >/dev/null 2>&1 || status=$? + expect_code 0 "$status" "help exit" + status=0 + "$CHECK" bogus >/dev/null 2>&1 || status=$? + expect_code 2 "$status" "unknown command exit" + pass "fm-hold-reverify: help prints and an unknown command is refused" +} + +test_merged_pull_request_reports_dead +test_open_question_reports_still_live +test_superseded_finding_reports_not_a_decision +test_no_subject_and_unreadable_forge_are_unestablishable +test_closed_unmerged_is_unestablishable_never_dead +test_young_hold_is_not_examined +test_repeat_is_silent_and_change_wakes +test_cadence_gate_suppresses_until_the_interval_elapses +test_sweep_defers_beyond_the_hold_cap +test_unreadable_projection_reports_once +test_classify_prints_the_verdict_for_facts +test_arm_writes_and_registers_and_disarm_removes +test_help_and_usage From e79860694e1217d4aed6f7ef70a8d7128e2a089f Mon Sep 17 00:00:00 2001 From: keenvc Date: Sun, 20 Sep 2026 01:07:50 +0000 Subject: [PATCH 19/35] fix(bin): bound herdr CLI probes so a hung read cannot wedge a supervisor --- bin/backends/herdr.sh | 46 ++++-- bin/fm-test-run.sh | 1 + docs/configuration.md | 1 + docs/herdr-backend.md | 5 + tests/fm-backend-herdr-probe-timeout.test.sh | 157 +++++++++++++++++++ 5 files changed, 200 insertions(+), 10 deletions(-) create mode 100755 tests/fm-backend-herdr-probe-timeout.test.sh diff --git a/bin/backends/herdr.sh b/bin/backends/herdr.sh index dee97f7866c..ef807c67250 100644 --- a/bin/backends/herdr.sh +++ b/bin/backends/herdr.sh @@ -93,6 +93,12 @@ FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" # shellcheck source=bin/fm-agent-process-lib.sh . "$FM_BACKEND_HERDR_ROOT/bin/fm-agent-process-lib.sh" +# The repo-wide bounded runner (bin/fm-timeout-lib.sh), the single owner of how +# this repo bounds a subprocess. Every synchronous Herdr CLI call below runs +# through it; see fm_backend_herdr_bounded. +# shellcheck source=bin/fm-timeout-lib.sh +. "$FM_BACKEND_HERDR_ROOT/bin/fm-timeout-lib.sh" + FM_BACKEND_HERDR_MIN_PROTOCOL=14 # events.subscribe (the native pane.agent_status_changed push stream) and its # subscription_event schema first shipped at protocol 16 (verified: herdr @@ -369,6 +375,23 @@ fm_backend_herdr_workspace_label() { printf 'firstmate' } +# FM_BACKEND_HERDR_CLI_TIMEOUT: the hard per-call bound in whole seconds on +# every synchronous Herdr CLI read or write. Without it a wedged server or a +# hung pane read blocks the supervisor that made the call forever and leaks the +# shell it ran in, one per probe. The long-lived `server` launch is exempt: its +# whole purpose is to outlive the call. Invalid or zero values fall back to 10. +FM_BACKEND_HERDR_CLI_TIMEOUT=${FM_BACKEND_HERDR_CLI_TIMEOUT:-10} + +# fm_backend_herdr_bounded: run under the adapter's hard bound via +# the repo-wide bounded runner (bin/fm-timeout-lib.sh). Returns the command's +# own status, or 124 when the bound fires, and kills the whole child process +# group so a hung herdr and anything it spawned cannot outlive the call. +fm_backend_herdr_bounded() { # + local bound=$FM_BACKEND_HERDR_CLI_TIMEOUT + case "$bound" in ''|*[!0-9]*|0) bound=10 ;; esac + fm_run_timed "$bound" "$@" +} + # fm_backend_herdr_cli: run `herdr ` scoped to , setting # BOTH the HERDR_SESSION env var AND appending a trailing `--session ` # CLI flag. Verified empirically (docs/herdr-backend.md "Session targeting: the @@ -393,21 +416,24 @@ fm_backend_herdr_cli() { # # stderr is buffered (stdout streams untouched) so a protocol_mismatch # refusal can be recognized and retried once on a compatible client; see # "client selection" below. A failed command's stderr is replayed verbatim. - # The long-lived `server` launch is exec'd straight through: buffering its - # stderr would hold this call open for the server's whole lifetime. + # Every call below runs under fm_backend_herdr_bounded (the + # FM_BACKEND_HERDR_CLI_TIMEOUT contract) so a hung server cannot wedge the + # caller. The long-lived `server` launch is the one exemption: it is exec'd + # straight through, because buffering its stderr would hold this call open + # for the server's whole lifetime and a bound would kill the server. if [ "${1:-}" = server ]; then HERDR_SESSION="$session" "$client_bin" "$@" --session "$session" return $? fi failed_bin=$client_bin - { err=$(HERDR_SESSION="$session" "$failed_bin" "$@" --session "$session" 2>&1 1>&3 3>&-) || rc=$?; } 3>&1 + { err=$(fm_backend_herdr_bounded env HERDR_SESSION="$session" "$failed_bin" "$@" --session "$session" 2>&1 1>&3 3>&-) || rc=$?; } 3>&1 if [ "$rc" -ne 0 ]; then case "$err" in *protocol_mismatch*) fm_backend_herdr_client_select "$session" force selected_bin=$(fm_backend_herdr_bin) if [ "$selected_bin" != "$failed_bin" ]; then - HERDR_SESSION="$session" "$selected_bin" "$@" --session "$session" + fm_backend_herdr_bounded env HERDR_SESSION="$session" "$selected_bin" "$@" --session "$session" return $? fi ;; @@ -468,7 +494,7 @@ fm_backend_herdr_client_candidates() { # client did not report. Never fails. fm_backend_herdr_client_status() { # local bin=$1 session=$2 out - out=$(HERDR_SESSION="$session" "$bin" status --json --session "$session" 2>/dev/null) || out= + out=$(fm_backend_herdr_bounded env HERDR_SESSION="$session" "$bin" status --json --session "$session" 2>/dev/null) || out= printf '%s' "$out" | jq -r ' [ (if (.server | type) == "object" and .server.running != null then (.server.running | tostring) else "" end), (if (.server | type) == "object" and (.server | has("compatible")) @@ -520,7 +546,7 @@ fm_backend_herdr_tool_check() { fm_backend_herdr_version_check() { fm_backend_herdr_tool_check || return 1 local status protocol version - status=$(herdr status --json 2>/dev/null) || { echo "error: 'herdr status --json' failed; is herdr installed correctly?" >&2; return 1; } + status=$(fm_backend_herdr_bounded herdr status --json 2>/dev/null) || { echo "error: 'herdr status --json' failed; is herdr installed correctly?" >&2; return 1; } protocol=$(printf '%s' "$status" | jq -r '.client.protocol // empty' 2>/dev/null) version=$(printf '%s' "$status" | jq -r '.client.version // empty' 2>/dev/null) case "$protocol" in @@ -3525,7 +3551,7 @@ fm_backend_herdr_pane_for_tab() { # # normally carry meta), best-effort. fm_backend_herdr_resolve_bare_selector() { # local name=$1 sessions session tabs tab_id wsid pane_id - sessions=$(herdr session list --json 2>/dev/null | jq -r '.sessions[]? | select(.running == true) | .name' 2>/dev/null) + sessions=$(fm_backend_herdr_bounded herdr session list --json 2>/dev/null | jq -r '.sessions[]? | select(.running == true) | .name' 2>/dev/null) while IFS= read -r session; do [ -n "$session" ] || continue tabs=$(fm_backend_herdr_cli "$session" tab list 2>/dev/null) || continue @@ -3589,7 +3615,7 @@ fm_backend_herdr_list_live() { # # ~/.config/herdr/sessions//herdr.sock). Empty on any failure. fm_backend_herdr_socket_path() { # local session=$1 - herdr session list --json 2>/dev/null \ + fm_backend_herdr_bounded herdr session list --json 2>/dev/null \ | jq -r --arg name "$session" '.sessions[]? | select(.name == $name) | .socket_path // empty' 2>/dev/null \ | head -1 } @@ -3613,10 +3639,10 @@ fm_backend_herdr_events_capable() { # if [ -z "${FM_BACKEND_HERDR_EVENT_READER:-}" ]; then command -v python3 >/dev/null 2>&1 || return 1 fi - protocol=$(herdr status --json 2>/dev/null | jq -r '.client.protocol // empty' 2>/dev/null) + protocol=$(fm_backend_herdr_bounded herdr status --json 2>/dev/null | jq -r '.client.protocol // empty' 2>/dev/null) case "$protocol" in ''|*[!0-9]*) return 1 ;; esac [ "$protocol" -ge "$FM_BACKEND_HERDR_MIN_EVENTS_PROTOCOL" ] || return 1 - schema=$(herdr api schema --json 2>/dev/null) || return 1 + schema=$(fm_backend_herdr_bounded herdr api schema --json 2>/dev/null) || return 1 printf '%s' "$schema" | grep -Fq 'events.subscribe' || return 1 printf '%s' "$schema" | grep -Fq 'pane.agent_status_changed' || return 1 return 0 diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 44e8da93bcc..1923bfeb0f9 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -365,6 +365,7 @@ family_for_basename() { printf '%s\n' live-harness-optin ;; fm-backend-herdr.test.sh|fm-backend-tmux-smoke.test.sh|fm-backend.test.sh|\ + fm-backend-herdr-probe-timeout.test.sh|\ fm-tmux-agent-liveness.test.sh|\ fm-control.test.sh|fm-control-relaunch.test.sh|\ fm-herdr-session-cleanup.test.sh|fm-send-resolve-key.test.sh|fm-send-strict.test.sh|\ diff --git a/docs/configuration.md b/docs/configuration.md index 808ee716baa..d748bb00dbd 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -1040,6 +1040,7 @@ FM_TASK_ID= # internal task-worker marker fm-spawn.sh exports into s HERDR_SESSION=default # herdr-only: named session for normal backend ops; not enough for destructive cleanup (docs/herdr-backend.md) FM_BACKEND_HERDR_SUBMIT_POLLS=6 # herdr-only: agent-state samples spread across each Enter attempt's budget when confirming a submit (docs/herdr-backend.md "Current transport behavior") FM_BACKEND_HERDR_SUBMIT_MIN_SLEEP=0.6 # herdr-only: minimum per-Enter confirmation budget before polling agent-state after an idle baseline +FM_BACKEND_HERDR_CLI_TIMEOUT=10 # herdr-only: whole-second hard bound on every synchronous herdr CLI read/write, so a hung probe cannot block a supervisor or leak its shell; invalid or zero values fall back to 10, and the long-lived `herdr server` launch is exempt (docs/herdr-backend.md "Current transport behavior") FM_ZELLIJ_SESSION=firstmate # zellij-only: named session for normal backend ops and test isolation (docs/zellij-backend.md) CMUX_SOCKET_PASSWORD= # cmux-only: socket password fallback when config/cmux-socket-password is absent (docs/cmux-backend.md) FM_SESSION_START_STATUS_TAIL=5 # state/*.status lines printed per task in the session-start digest; each line is capped by bin/fm-line-cap-lib.sh diff --git a/docs/herdr-backend.md b/docs/herdr-backend.md index 1199bd8142d..c092bacfa70 100644 --- a/docs/herdr-backend.md +++ b/docs/herdr-backend.md @@ -225,6 +225,11 @@ Herdr passes its server startup environment to every later pane, so retaining th An already-running server is reused without restart or environment changes. Explicit named-session routing and unrelated launch environment remain intact. +Every synchronous Herdr CLI read or write runs under a hard per-call bound (`FM_BACKEND_HERDR_CLI_TIMEOUT`, default 10 seconds) through the repo-wide bounded runner in `bin/fm-timeout-lib.sh`, so a wedged server or a hung pane read cannot block a supervisor indefinitely or leak the shell that made the call. +The bound kills the whole child process group and reports Herdr timeout as exit 124. +The long-lived `herdr server` launch is the one exemption, because its purpose is to outlive the call and a bound would kill the server. +`tests/fm-backend-herdr-probe-timeout.test.sh` pins the bound, the process reaping, and the server exemption against a TERM-ignoring fake herdr. + Literal text and Enter are separate operations on `fm-send.sh`'s typed plane; ordinary local text steers instead use the durable steering inbox and send only its best-effort constant doorbell through this adapter. Spawn-time fixed commands may use Herdr's atomic run primitive. Enter, Escape, and Ctrl-C are supported. diff --git a/tests/fm-backend-herdr-probe-timeout.test.sh b/tests/fm-backend-herdr-probe-timeout.test.sh new file mode 100755 index 00000000000..af6bab15db9 --- /dev/null +++ b/tests/fm-backend-herdr-probe-timeout.test.sh @@ -0,0 +1,157 @@ +#!/usr/bin/env bash +# tests/fm-backend-herdr-probe-timeout.test.sh - proves every synchronous herdr +# CLI read in bin/backends/herdr.sh runs under a real process-level bound, and +# that a hung probe's process is gone once that bound fires. The property is the +# adapter's own timeout discipline, so it is pinned with a fake herdr that +# ignores TERM and never answers a read; no real herdr installation is needed. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +command -v jq >/dev/null 2>&1 || { echo "skip: jq not found (required by the herdr adapter)"; exit 0; } + +TMP_ROOT=$(fm_test_tmproot fm-backend-herdr-probe-timeout) +HANG_PIDS="$TMP_ROOT/hang-pids" +: > "$HANG_PIDS" + +# A herdr stub that answers the server-state liveness read so target_ready +# passes, records its own pid, and then never returns from a real read. It +# ignores TERM (with a self-deadline so a broken adapter cannot hang the suite +# forever), so only the runner's KILL escalation can reap it. +make_hanging_herdr_fakebin() { # -> echoes fakebin dir + local dir=$1 fb="$1/fakebin" + mkdir -p "$fb" + cat > "$fb/herdr" <<'SH' +#!/usr/bin/env bash +set -u +printf '%s\n' "$$" >> "$FM_HANG_PIDS" +if [ "${1:-}" = status ] && [ "${2:-}" = --json ]; then + printf '{"client":{"protocol":22},"server":{"running":true}}\n' + exit 0 +fi +trap '' TERM +deadline=$((SECONDS + 20)) +while [ "$SECONDS" -lt "$deadline" ]; do + sleep 1 +done +exit 0 +SH + chmod +x "$fb/herdr" + printf '%s\n' "$fb" +} + +# A fake whose server launch is a short, normal-lived command: it exits 0 after +# a delay LONGER than the bound under test, so a wrongly bounded server call +# would be killed (124) instead of completing. +make_server_launch_fakebin() { # -> echoes fakebin dir + local dir=$1 nap=$2 fb="$1/fakebin-server" + mkdir -p "$fb" + cat > "$fb/herdr" </dev/null || true + done < "$HANG_PIDS" +} +trap 'reap_hung_fakes; fm_test_cleanup' EXIT + +# every_fake_is_gone: true only when each recorded pid is gone (or a zombie the +# init reaper is about to collect), after a short bounded settle. A still-live +# process after the grace window is a leaked shell and fails the case. +every_fake_is_gone() { + local attempt=0 pid state + while [ "$attempt" -lt 30 ]; do + local all_gone=1 + while IFS= read -r pid; do + case "$pid" in ''|*[!0-9]*) continue ;; esac + if kill -0 "$pid" 2>/dev/null; then + state=$(ps -o stat= -p "$pid" 2>/dev/null | tr -d ' ') + case "$state" in + ''|Z*) ;; + *) all_gone=0 ;; + esac + fi + done < "$HANG_PIDS" + [ "$all_gone" -eq 1 ] && return 0 + attempt=$((attempt + 1)) + sleep 0.2 + done + return 1 +} + +run_adapter_snippet() { # + local fb=$1 bound=$2 snippet=$3 + PATH="$fb:$PATH" FM_HANG_PIDS="$HANG_PIDS" FM_BACKEND_HERDR_CLI_TIMEOUT="$bound" \ + FM_HOME="$TMP_ROOT/ambient-home" \ + bash -c ". \"\$0/bin/backends/herdr.sh\"; $snippet" "$ROOT" +} + +test_capture_probe_is_bounded_and_reaped() { + local dir fb start elapsed out rc + dir="$TMP_ROOT/capture"; mkdir -p "$dir" + fb=$(make_hanging_herdr_fakebin "$dir") + start=$SECONDS + out=$(run_adapter_snippet "$fb" 1 'fm_backend_herdr_capture fmtest:w1:p2 40' 2>/dev/null) + rc=$? + elapsed=$((SECONDS - start)) + [ "$rc" -ne 0 ] || fail "a hung capture read must fail rather than return success (out='$out')" + [ "$elapsed" -lt 15 ] || fail "a hung capture read ignored the bound and ran ${elapsed}s" + every_fake_is_gone || fail "a hung capture read leaked its herdr process past the bound: $(tr '\n' ' ' < "$HANG_PIDS")" + pass "capture read: a TERM-ignoring hung herdr is bounded and its process is reaped" +} + +test_composer_state_probe_is_bounded_and_reaped() { + local dir fb start elapsed out + dir="$TMP_ROOT/composer"; mkdir -p "$dir" + fb=$(make_hanging_herdr_fakebin "$dir") + start=$SECONDS + out=$(run_adapter_snippet "$fb" 1 'fm_backend_herdr_composer_state fmtest:w1:p2' 2>/dev/null) + elapsed=$((SECONDS - start)) + [ "$out" = unknown ] || fail "a hung composer probe must read unknown, got '$out'" + [ "$elapsed" -lt 20 ] || fail "a hung composer probe ignored the bound and ran ${elapsed}s" + every_fake_is_gone || fail "a hung composer probe leaked its herdr process past the bound: $(tr '\n' ' ' < "$HANG_PIDS")" + pass "composer_state probe: a TERM-ignoring hung herdr is bounded and its process is reaped" +} + +test_generic_cli_read_is_bounded_and_reaped() { + local dir fb start elapsed rc + dir="$TMP_ROOT/cli"; mkdir -p "$dir" + fb=$(make_hanging_herdr_fakebin "$dir") + start=$SECONDS + run_adapter_snippet "$fb" 1 'fm_backend_herdr_cli fmtest pane read w1:p2 --source recent --lines 200' >/dev/null 2>&1 + rc=$? + elapsed=$((SECONDS - start)) + [ "$rc" -eq 124 ] || fail "a hung fm_backend_herdr_cli read must return 124 (the bound), got $rc" + [ "$elapsed" -lt 10 ] || fail "a hung fm_backend_herdr_cli read ignored the bound and ran ${elapsed}s" + every_fake_is_gone || fail "a hung cli read leaked its herdr process past the bound: $(tr '\n' ' ' < "$HANG_PIDS")" + pass "fm_backend_herdr_cli: the shared read owner bounds and reaps a hung herdr" +} + +test_server_launch_is_exempt_from_the_bound() { + local dir fb rc + dir="$TMP_ROOT/server"; mkdir -p "$dir" + fb=$(make_server_launch_fakebin "$dir" 2) + run_adapter_snippet "$fb" 1 'fm_backend_herdr_cli fmtest server' >/dev/null 2>&1 + rc=$? + [ "$rc" -eq 0 ] \ + || fail "the long-lived server launch must not be bounded (got rc=$rc; a 1s bound would kill it)" + pass "fm_backend_herdr_cli: the long-lived server launch stays exempt from the bound" +} + +test_capture_probe_is_bounded_and_reaped +test_composer_state_probe_is_bounded_and_reaped +test_generic_cli_read_is_bounded_and_reaped +test_server_launch_is_exempt_from_the_bound From 344485aa63660e750c25034230cb42b3bef07299 Mon Sep 17 00:00:00 2001 From: keenvc Date: Sun, 20 Sep 2026 01:08:41 +0000 Subject: [PATCH 20/35] fix(bin): account for every registered secondmate in liveness sweep --- .agents/skills/bootstrap-diagnostics/SKILL.md | 2 +- .../skills/secondmate-provisioning/SKILL.md | 1 + AGENTS.md | 2 +- bin/fm-bootstrap.sh | 126 ++++++++++++++---- docs/configuration.md | 3 +- tests/fm-secondmate-liveness.test.sh | 115 ++++++++++++++++ 6 files changed, 220 insertions(+), 29 deletions(-) diff --git a/.agents/skills/bootstrap-diagnostics/SKILL.md b/.agents/skills/bootstrap-diagnostics/SKILL.md index 1d49f4b8312..cd73ea9b55f 100644 --- a/.agents/skills/bootstrap-diagnostics/SKILL.md +++ b/.agents/skills/bootstrap-diagnostics/SKILL.md @@ -65,7 +65,7 @@ When any diagnostic needs captain attention, report the plain consequence and re Neither copy is a safe winner: union-merge them into this home's file by task id, resolve each conflicting id to its most recent real transition, check this home's archive before treating a missing Done row as lost, and verify the merged id set equals the union of both inputs before installing it. Then move the code-root file aside rather than deleting it, tell the captain which rows were recovered, and run every later backlog command through `bin/fm-tasks-axi.sh`; re-linking the code-root copy is never the fix, because the next cwd-relative tasks-axi write replaces the link again. - `SECONDMATE_SYNC: secondmate : skipped: ` - secondmate convergence left a live home on its existing checkout because the home was dirty, diverged, unsafe, on the wrong branch, missing its placement-specific target commit, unreachable, or otherwise not fast-forwardable, or because inherited local-material propagation failed; bootstrap continued, but inspect the reason because the secondmate's tracked instructions, inherited settings, or shared captain preferences may be stale after a primary update. -- `SECONDMATE_LIVENESS: secondmate : skipped: |respawn failed after : ` - the session-start liveness sweep could not guarantee that the registered secondmate is running a real agent process. +- `SECONDMATE_LIVENESS: secondmate : skipped: |respawn failed after : |gap: ` - the session-start liveness sweep could not guarantee that the registered secondmate is running a real agent process; a `gap:` line means a registered secondmate had no usable task record and relaunching it from the registry also failed. Investigate the reason because that secondmate is not guaranteed live. - `SECONDMATE_HANDOFF: secondmate : pending delivery: item(s)` - queued work has already left the main dispatchable backlog and remains safe in the named remote route's backlog-format outbox because backlog receipt or local outbox cleanup has not completed; [`bin/fm-backlog-handoff.sh`](../../../bin/fm-backlog-handoff.sh) owns the release contract. Preserve that outbox and rerun `bin/fm-backlog-handoff.sh --resume-pending` after the route, receipt, or cleanup problem is resolved; never re-add or dispatch the items from the main backlog. diff --git a/.agents/skills/secondmate-provisioning/SKILL.md b/.agents/skills/secondmate-provisioning/SKILL.md index f716d5e960c..d512ad9193b 100644 --- a/.agents/skills/secondmate-provisioning/SKILL.md +++ b/.agents/skills/secondmate-provisioning/SKILL.md @@ -219,6 +219,7 @@ bin/fm-spawn.sh --secondmate Use the recorded `home=` in meta. If meta is missing but `data/secondmates.md` still registers the secondmate, respawn from the registry entry and its persistent home. +The locked session-start liveness sweep performs exactly this recovery for every secondmate registered in `data/secondmates.md`, including one with no `state/.meta` record or a record with no endpoint; when it cannot relaunch one, it names that secondmate as an explicit `SECONDMATE_LIVENESS:` gap line rather than passing over it. For a remote route, the same command probes and relaunches only on the configured host. An SSH transport failure or unreadable remote endpoint remains unknown and must be reconciled on that host; never launch a local replacement. `stuck-crewmate-recovery`'s remote-secondmate note owns why the endpoint-dead and send-failed verdicts that seem to justify this are themselves unreliable. diff --git a/AGENTS.md b/AGENTS.md index c3634216137..6e6a5eefc22 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -185,7 +185,7 @@ When that section reports its checks still in progress it names exactly what is 2. **Bootstrap** - detect-only checks (tool/version problems, the worktree-tangle check, harness override, dispatch-profile validation, backlog-backend status) always run, but routine confirmations stay silent by default. When the lock could not be acquired, the worktree-tangle check uses read-only advisory wording without a checkout repair command. Home-local stale Herdr projection cleanup and the six bootstrap MUTATING sweeps - same-home backlog reconciliation, fleet sync, secondmate convergence, secondmate liveness, pending remote handoff retry, and Relay artifact writes - run only when this session actually holds the lock from step 1; the four network ones among them run in the deferred stage rather than in this section. - The secondmate liveness sweep deterministically accounts for every registered secondmate: it relaunches only from the recovery-grade `dead` or `missing` states, preserves ambiguous, unreadable, or unreachable remote targets, and reports skipped or failed guarantees as `SECONDMATE_LIVENESS:` lines (`bin/fm-bootstrap.sh`; `bin/fm-backend.sh`'s `fm_backend_agent_state`; `docs/remote-secondmates.md`). + The secondmate liveness sweep deterministically accounts for every registered secondmate: it walks `data/secondmates.md` as well as `state/.meta`, so a secondmate whose record is missing or has no endpoint is relaunched from its registry entry rather than silently passed over; it relaunches from the recovery-grade `dead` or `missing` states, preserves ambiguous, unreadable, or unreachable remote targets, and reports skipped, failed, or unrecoverable guarantees as `SECONDMATE_LIVENESS:` lines, naming an unrecoverable registered secondmate as an explicit gap (`bin/fm-bootstrap.sh`; `bin/fm-backend.sh`'s `fm_backend_agent_state`; `docs/remote-secondmates.md`). 3. **Wake queue** - when locked, drains and presents the durable wake queue without running the inactive-outcome scan inline, and prints the raw records prominently as this turn's first work queue; a clearly labeled status-event annotation may follow a valid `signal` record and includes every status line still unread at the presentation cursor, but never replaces the raw record or current-state reconciliation, and a lapsed watcher chain still surfaces here via the same guard alarm. Presented records remain durable until the handling turn runs the generation-bound acknowledgement printed by the drain. Every locked drain also prints a bounded fleet-wide `OPEN DECISIONS` section when durable decision records remain open, including when the queue itself is empty; reconcile those entries before continuing. diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index 31792fa37ba..772a866d328 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -20,7 +20,7 @@ # "SECONDMATE_SYNC: secondmate : skipped: ", # "NUDGE_SECONDMATES: secondmate : send failed: ", # "BOOTSTRAP_INFO: nudged fm- with ''", -# "SECONDMATE_LIVENESS: secondmate : skipped: |respawn failed after : ", +# "SECONDMATE_LIVENESS: secondmate : skipped: |respawn failed after : |gap: ", # "SECONDMATE_HANDOFF: secondmate : pending delivery: item(s)", # "FMX: X mode on ..." or "FMX: X mode off ...". # When a RUNNING secondmate home is fast-forwarded, its target is @@ -48,6 +48,11 @@ # fm_backend_agent_state: skipped distinguishes an existing ambiguous # process, an unreadable target, and an unverified backend; respawn # failed names whether the endpoint was missing or agent-less. +# The sweep accounts for every secondmate registered in +# data/secondmates.md, not only those with a state/.meta record: a +# registered secondmate with no record, or a record with no endpoint, +# is relaunched from the registry, and one that cannot be recovered is +# named with an explicit `gap:` line rather than omitted. # Already-live and successfully relaunched secondmates are silent # unless FM_BOOTSTRAP_VERBOSE_FACTS=1 requests BOOTSTRAP_INFO facts. # A TANGLE line means the firstmate primary checkout (FM_ROOT) is stranded @@ -683,61 +688,130 @@ report_relaunch() { # echo "BOOTSTRAP_INFO: secondmate $1 relaunched after $2 ($3)" } +# Registered secondmate ids from data/secondmates.md, in file order. The +# registry is the durable authority for WHICH secondmates exist; state/.meta +# is only the endpoint record for one that is currently running. +secondmate_registered_ids() { # + local reg=$1 line id + [ -f "$reg" ] && [ ! -L "$reg" ] || return 0 + while IFS= read -r line || [ -n "$line" ]; do + case "$line" in + '- '*) ;; + *) continue ;; + esac + id=${line#- } + id=${id%% *} + case "$id" in '' | *[!A-Za-z0-9._-]*) continue ;; esac + printf '%s\n' "$id" + done < "$reg" +} + +# Every id the sweep must account for: registered secondmates first, then any +# kind=secondmate endpoint record not already covered. The caller deduplicates, +# so a running registered secondmate is probed exactly once. +secondmate_liveness_ids() { # + local state=$1 registry=$2 meta id + secondmate_registered_ids "$registry" + [ -d "$state" ] || return 0 + for meta in "$state"/*.meta; do + [ -f "$meta" ] || continue + grep -q '^kind=secondmate$' "$meta" 2>/dev/null || continue + id=$(basename "$meta" .meta) + printf '%s\n' "$id" + done +} + secondmate_liveness_sweep() { - # Idempotent secondmate liveness guarantee - SESSION START ONLY. The detailed - # state machine and its only recovery-authorizing states are owned by - # fm_backend_agent_state. A missing tmux pane is not enough: tmux must prove - # the window or session absent. This preserves duplicate prevention for + # Idempotent secondmate liveness guarantee - SESSION START ONLY. Every + # REGISTERED secondmate is accounted for: the walk covers data/secondmates.md + # plus state/.meta, never only the meta records, so a secondmate whose + # record is missing or incomplete is recovered rather than silently passed + # over. The detailed state machine and its recovery-authorizing states are + # owned by fm_backend_agent_state. A missing tmux pane is not enough: tmux must + # prove the window or session absent. This preserves duplicate prevention for # existing ambiguous processes and every transiently unreadable target while # adding the missing-session path the original bare-shell and Herdr-husk sweep # lacked. - # A meta with no window remains owned by secondmate-provisioning recovery. - # Secondmate homes never contain kind=secondmate meta, so this is naturally a - # primary-only no-op there. Mid-session liveness remains explicitly out of - # scope and requires a separate periodic signal. + # A registered secondmate with no record, or a record with no endpoint, is a + # recoverable state handled by secondmate_liveness_recover_from_registry; one + # that cannot be recovered is named as an explicit `gap:` line rather than + # omitted. + # Secondmate homes never contain kind=secondmate meta AND never register + # secondmates, so this is naturally a primary-only no-op there. Mid-session + # liveness remains explicitly out of scope and requires a separate periodic + # signal. [ -d "$STATE" ] || return 0 - local meta id remote_host label __fm_timing_stamp parallel=0 + local meta id remote_host label parallel=0 SECONDMATE_RESPAWNED_IDS="" if bootstrap_parallel_begin; then parallel=1 fi - for meta in "$STATE"/*.meta; do - [ -f "$meta" ] || continue - grep -q '^kind=secondmate$' "$meta" 2>/dev/null || continue - # Identity for the timing record is read here, in the loop, so the per-meta - # body below keeps its single-exit-per-outcome shape. - id=$(basename "$meta" .meta) - remote_host=$(fm_meta_get "$meta" remote_host) + while IFS= read -r id; do + [ -n "$id" ] || continue + meta="$STATE/$id.meta" + if [ -f "$meta" ]; then + grep -q '^kind=secondmate$' "$meta" 2>/dev/null || meta= + else + meta= + fi label=$id - [ -z "$remote_host" ] || label="$id@$remote_host" + if [ -n "$meta" ]; then + remote_host=$(fm_meta_get "$meta" remote_host) + [ -z "$remote_host" ] || label="$id@$remote_host" + fi if [ "$parallel" -eq 1 ]; then - bootstrap_parallel_spawn secondmate_liveness_one_timed "$meta" "$id" "$label" + bootstrap_parallel_spawn secondmate_liveness_one_timed "$id" "$meta" "$label" else - secondmate_liveness_one_timed "$meta" "$id" "$label" + secondmate_liveness_one_timed "$id" "$meta" "$label" fi - done + done < <(secondmate_liveness_ids "$STATE" "$DATA/secondmates.md" | awk '!seen[$0]++') [ "$parallel" -eq 0 ] || bootstrap_parallel_finish return 0 } -secondmate_liveness_one_timed() { #