diff --git a/.claude/mods/firstmate-calm/lib/fm-calm-presentation.ts b/.claude/mods/firstmate-calm/lib/fm-calm-presentation.ts index c8a4e556a79..f2ed8d349aa 100644 --- a/.claude/mods/firstmate-calm/lib/fm-calm-presentation.ts +++ b/.claude/mods/firstmate-calm/lib/fm-calm-presentation.ts @@ -9,6 +9,12 @@ // captain-facing contract and docs/configuration.md // the persisted preference schema. Everything here is pure so tests run it under Node. import { classifyFirstmateOperationalText } from "./fm-operational-input.ts"; +import { + CALM_PRESERVE_MIN_CHARS, + calmTextIsSubstantive, +} from "./fm-calm-preservation.ts"; + +export { CALM_PRESERVE_MIN_CHARS } from "./fm-calm-preservation.ts"; /** The environment variables that select the effective Firstmate home, as the mod reads them. */ export type CalmHomeEnvironment = { @@ -67,18 +73,6 @@ export type CalmStepOutcome = { readonly toolUses: readonly unknown[]; }; -/** - * Single-line narration in session history topped out around 215 characters, while - * substantive single-line content began around 270; every multi-line message was - * substantive, so this empirical boundary stays deliberately tunable. - */ -export const CALM_PRESERVE_MIN_CHARS = 240; - -/** Whether text is substantive enough to preserve despite ending alongside a tool call. */ -function shouldPreserveMidTurnText(text: string): boolean { - const trimmedText = text.trim(); - return text.includes("\n") || trimmedText.length >= CALM_PRESERVE_MIN_CHARS; -} /** * Whether text from a model step is a mid-turn working note: the model did not end @@ -88,7 +82,7 @@ function shouldPreserveMidTurnText(text: string): boolean { */ export function stepTextIsWorkingNote(step: CalmStepOutcome, text: string): boolean { const midTurn = step.stopReason === "tool_use" || (step.stopReason === "max_tokens" && step.toolUses.length > 0); - return midTurn && !shouldPreserveMidTurnText(text); + return midTurn && !calmTextIsSubstantive(text); } /** A trimmed text key that retains whether the raw row contained a newline. */ @@ -130,7 +124,7 @@ export function classifyRestoredTranscript(rows: readonly CalmSessionRow[]): { break; } } - if (followedByToolCall && shouldPreserveMidTurnText(row.text)) finalReplies.add(key); + if (followedByToolCall && calmTextIsSubstantive(row.text)) finalReplies.add(key); else if (followedByToolCall) notes.add(key); else finalReplies.add(key); } diff --git a/.claude/mods/firstmate-calm/lib/fm-calm-preservation.ts b/.claude/mods/firstmate-calm/lib/fm-calm-preservation.ts new file mode 100644 index 00000000000..1b1619a4805 --- /dev/null +++ b/.claude/mods/firstmate-calm/lib/fm-calm-preservation.ts @@ -0,0 +1,11 @@ +// Shared Calm policy for deciding whether mid-turn assistant text is substantive. +// Claude Code imports this file directly, while the Pi extension reaches the same +// implementation through its tracked symlink so both harnesses keep one threshold and rule. + +/** The minimum trimmed text length preserved from a mid-turn assistant message. */ +export const CALM_PRESERVE_MIN_CHARS = 240; + +/** Whether mid-turn assistant text is substantive enough to remain visible. */ +export function calmTextIsSubstantive(text: string): boolean { + return text.includes("\n") || text.trim().length >= CALM_PRESERVE_MIN_CHARS; +} diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 24ade79a846..a68f49cdd02 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -450,6 +450,18 @@ jobs: exit 1 } + # Same shape for the watcher's churn-deferral regression: an already- + # marked churn window expands an empty array that only stock Bash + # treats as an unbound variable under set -u. + churn_output=$(FM_TEST_ONLY=test_turn_ended_churn_existing_marker_absorbed \ + /bin/bash tests/fm-watch-triage.test.sh) + printf '%s\n' "$churn_output" + churn_count=$(printf '%s\n' "$churn_output" | grep -c '^ok - ') + [ "$churn_count" -eq 1 ] || { + echo "::error::expected 1 watcher churn-deferral bash 3.2 regression, got $churn_count" + exit 1 + } + invariants: name: Repo invariants runs-on: ubuntu-latest diff --git a/.pi/extensions/lib/fm-calm-assistant-layout.ts b/.pi/extensions/lib/fm-calm-assistant-layout.ts index e2f00af52bc..a337d4535b1 100644 --- a/.pi/extensions/lib/fm-calm-assistant-layout.ts +++ b/.pi/extensions/lib/fm-calm-assistant-layout.ts @@ -2,12 +2,14 @@ // updateContent method. installCalmAssistantLayout() probes that exact method and throws // if it is missing; fm-calm.ts catches that and skips only this adapter with a diagnostic // instead of blocking Calm or Pi. -// This layout removes collapsed thinking and the mid-turn assistant text blocks -// classified as "assistant-working-note" from a shallow presentation copy. The message +// This layout removes collapsed thinking and short mid-turn assistant text blocks +// classified as "assistant-working-note" from a shallow presentation copy. Substantive +// mid-turn text is preserved. The message // itself, model context, session storage, and export rendering are never touched. // ./fm-calm-visibility.ts owns which classes Calm hides. import type { AssistantMessageComponent as PiAssistantMessageComponent } from "@earendil-works/pi-coding-agent"; import * as PiCodingAgent from "@earendil-works/pi-coding-agent"; +import { calmTextIsSubstantive } from "./fm-calm-preservation.ts"; import { calmPresentationHides } from "./fm-calm-visibility.ts"; type AssistantMessage = Parameters[0]; @@ -75,7 +77,11 @@ export function installCalmAssistantLayout(): void { state.hideThinkingBlock && patch.hidesThinking(); const hideWorkingNote = - patch.hidesWorkingNote() && isMidTurnAssistantMessage(message); + patch.hidesWorkingNote() && + isMidTurnAssistantMessage(message) && + message.content.some( + (block) => block.type === "text" && !calmTextIsSubstantive(block.text), + ); const presentationMessage = hideThinking || hideWorkingNote ? { @@ -83,7 +89,11 @@ export function installCalmAssistantLayout(): void { content: message.content.filter( (block) => !(hideThinking && block.type === "thinking") && - !(hideWorkingNote && block.type === "text"), + !( + hideWorkingNote && + block.type === "text" && + !calmTextIsSubstantive(block.text) + ), ), } : message; diff --git a/.pi/extensions/lib/fm-calm-preservation.ts b/.pi/extensions/lib/fm-calm-preservation.ts new file mode 120000 index 00000000000..93dd8e02938 --- /dev/null +++ b/.pi/extensions/lib/fm-calm-preservation.ts @@ -0,0 +1 @@ +../../../.claude/mods/firstmate-calm/lib/fm-calm-preservation.ts \ No newline at end of file diff --git a/AGENTS.md b/AGENTS.md index b63cda234ee..e17d4b9979a 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -477,6 +477,10 @@ For the full `stuck-crewmate-recovery` trigger, including a live worker claiming **Talk in outcomes, not mechanics.** Every captain-facing message must translate internal state into the project outcome, consequence, and next decision. +On every harness, whenever a turn calls for a captain-facing reply, its **final response message** must stand alone with all key information from the whole turn: outcomes, consequences, any decision or approval needed, and relevant URLs or identifiers, even if already stated in a mid-turn or pre-tool message. +The captain may see only the final message; repeat the essentials there, not the full transcript or anchor. +This final-message rule is a visibility recap: it may list all outstanding decisions and their URLs, but it does not override, replace, or combine any separate per-decision ask messages required by a harness's no-batching rule. +Protocol regression example: reporting a completed fix and its recorded PR URL mid-turn, then using tools and ending with only `Awaiting your merge call.`, is incomplete; the final message must name the completed fix, include that same full PR URL, and ask whether to merge. Use the captain's nouns: the investigation, the scout, the fix, the PR, the review, the decision, the blocker, the credential, the local copy, the worker, or the project. Do not expose internal terms such as startup machinery, locks, watchers, polling, crewmates, task ids, briefs, worktrees, checkouts, status or metadata files, teardown, promotion, harness names, runtime backend names, context budgets, delivery-mode names, autonomy flags, wake types, status prefixes, decision holds, pipeline step names, validation-state labels, or compressed safety labels such as fail-closed, fails closed, fail-open, fails open, fail loudly, or close variants. Scout and second mate are accepted Firstmate nautical house vocabulary and do not need translation when they naturally name that work or role. @@ -513,10 +517,12 @@ Reach the captain immediately for: In a secondmate home, reaching the captain means appending the outcome to the parent channel your charter names; a captain-facing sentence in that home's chat has not been sent, and [`docs/secondmate-parent-channel.md`](docs/secondmate-parent-channel.md) owns which outcomes the home's own scripts deliver there without you. Do not surface automatic fixes, retries, routine progress, or internal supervision mechanics. -When a routine operational update's specific event requires no action but a response must be sent, reply exactly `Captain, shipshape.` without characterizing the visible session's unrelated decisions. +Reply exactly `Captain, shipshape.` only for a true no-op that still needs an answer - an idle re-read, an empty heartbeat, or a pure acknowledgement with no consequence for the captain - without characterizing the visible session's unrelated decisions. +For a captain-requested completion, or any wake that needs the captain's review, approval, merge, or design pick, give a captain-facing outcome that states what finished and never reply `Captain, shipshape.`; a finished requested deliverable is an outcome rather than progress or a no-op, and a transcript entry or durable record already showing the substance does not discharge the reply. +Ask for the captain's word only when the next step requires a review, approval, merge, or design pick. Batch non-urgent updates into the next natural reply. Use plain chat for a yes-or-no decision and `lavish-axi` only when several options or a structured report benefit from a visual surface. -Whenever a PR is mentioned, include its full `https://...` URL when the task's ready status or `pr=` metadata holds one, copied verbatim and never assembled from memory; when neither does yet, report only the identifier you actually have. +Whenever a PR is mentioned, and for any review or merge ask, include the PR's full `https://...` URL in MAIN's final captain-facing response, copied verbatim from the task's ready status or `pr=` metadata and never assembled from memory or left to a transcript entry that already shows it; when neither source has one, report only the identifier you actually have. Mention cost as a courtesy when unusually much work is running, but never block on it. ## 10. Backlog contract diff --git a/GROK_BOT.md b/GROK_BOT.md index f823d1e9c15..3cb2a16a346 100644 --- a/GROK_BOT.md +++ b/GROK_BOT.md @@ -27,3 +27,5 @@ Speak in outcomes and consequences, not internal mechanics. When you bring a decision to the captain, send one message per decision. Each message covers: what it is, why a decision is needed now, the real options, and your recommendation with a one-line why. Put the options on a choice card so they can tap one. One card at a time. Do not batch unrelated decisions into one list. Keep it simple for the captain. Focus on communicating outcomes, not mechanics. They scale by talking only to you; protect that. + +Read and follow [AGENTS.md section 9](AGENTS.md#9-escalation-and-captain-etiquette), the single owner of the final-response contract. diff --git a/bin/fm-backlog-handoff.sh b/bin/fm-backlog-handoff.sh index 3c9c97d71db..882be0b367f 100755 --- a/bin/fm-backlog-handoff.sh +++ b/bin/fm-backlog-handoff.sh @@ -804,7 +804,7 @@ remote_handoff() { # echo " nothing new was staged." >&2 return 1 fi - for key in "${to_move[@]}"; do + for key in "${to_move[@]+"${to_move[@]}"}"; do while IFS= read -r line; do printf 'error: refusing to hand off %s: non-2-space continuation line: %s\n' "$key" "$line" >&2 return 1 diff --git a/bin/fm-crew-state.sh b/bin/fm-crew-state.sh index 8ef77cf25dd..160c729ed67 100755 --- a/bin/fm-crew-state.sh +++ b/bin/fm-crew-state.sh @@ -51,10 +51,10 @@ # before it having ended at exactly this worktree's head - so an active fix # round never reads as an older failed run (rule owned by # fm_nm_runs_status_for_worktree in bin/fm-nm-run-lib.sh). -# More than one recorded run can bind to this worktree at once, and -# bin/fm-nm-run-lib.sh also owns which of them wins: a LIVE run always -# outranks a terminal one, so a terminal answer here is provisional until -# the ledger has been asked whether a live sibling run exists. +# fm_nm_select_run in bin/fm-nm-run-lib.sh owns complete run selection +# and ambiguity reporting. The selected run's id-addressed status must +# agree on id, branch, and live/terminal class before attribution; +# disagreement reports unknown with available candidate ids. # The run-step is AUTHORITATIVE: running/fixing -> working, ci -> working, # awaiting_approval/fix_review -> parked (with gate findings), terminal # passed/checks-passed -> done, failed/cancelled -> failed. EXCEPT: while @@ -86,8 +86,9 @@ # running/fixing with recent reported activity: a killed or timed-out drive # call is not daemon death, so that claim is answered by steering the crew # to reattach, not by escalating. -# 4. No run for this crew (pre-validation, or kind=scout): fall back to the -# recorded backend's pane busy state, then the resolved status declaration +# 4. No current run for this crew (pre-validation, uninitialized repository, +# proven historical head, or kind=scout): fall back to the recorded +# backend's pane busy state, then the resolved status declaration # when its verb maps to a recognized run-state. Decision-only events such as # `resolved` never become current state or detail. # 5. Missing meta or torn-down worktree: report unknown · none. If no run is @@ -134,9 +135,9 @@ LOG=${FM_CREW_STATE_STATUS_OVERRIDE:-"$STATE/$ID.status"} NM_TIMEOUT=${FM_CREW_STATE_NM_TIMEOUT:-10} case "$NM_TIMEOUT" in ''|*[!0-9]*) NM_TIMEOUT=10 ;; esac # How many of the most recent `no-mistakes runs` rows each ledger read -# (fm_nm_runs_status_for_worktree in bin/fm-nm-run-lib.sh) scans, whether it is -# the cross-branch fallback or the live-sibling probe behind a terminal `axi -# status` answer (docs/configuration.md owns the setting). Generous enough to +# (fm_nm_runs_status_for_worktree in bin/fm-nm-run-lib.sh) scans for the legacy +# fallback or an unfetched-head continuation (docs/configuration.md owns the +# setting). Generous enough to # still find a branch's own run on a busy multi-crew fleet without listing the # entire history every call. FM_CREW_STATE_RUNS_LIMIT=${FM_CREW_STATE_RUNS_LIMIT:-200} @@ -654,13 +655,13 @@ nm_ci_checks_state() { # has no runs-listing subcommand; tests/fm-crew-state.test.sh owns the # 2026-07-02 dead-code incident history this fallback replaced). # fm_nm_runs_status_for_worktree in bin/fm-nm-run-lib.sh is the ONE owner of -# the ledger format, the newest-row-decides rule, its live-over-terminal -# exception, and the anchored pipeline-continuation recognition +# the ledger format, the newest-row-decides rule, and the anchored +# pipeline-continuation recognition # (model-routing-benchmark-hardening: an active fix round whose head object the # task copy never fetched used to be rejected here, letting the older failed row # answer as current), so both attribution routes share one rule. -# The same reader is also consulted when `axi status` DID bind this branch's run -# but that run is terminal, to find a live sibling run for this worktree. +# The same reader checks for conflicting run records when the AXI overview +# cannot identify this branch's run. nm_runs_list() { nm_run runs --limit "$FM_CREW_STATE_RUNS_LIMIT" } @@ -687,50 +688,106 @@ HAVE_RUN=0 # the TOON field parsing entirely for this crew. RUN_SOURCE=full COARSE_STATUS="" +SELECTED_RUN_ID="" # Scouts and secondmates never drive a no-mistakes validation of their own # worktree, so skip the lookup for them and read state from pane/log directly. if [ "$KIND" = ship ] && [ -n "$CREW_BRANCH" ] && command -v no-mistakes >/dev/null 2>&1; then RUN_OUT=$(nm_run axi status) + if [ "$(strip_quotes "$(printf '%s\n' "$RUN_OUT" | sed -n 's/^error: //p')")" = "repo not initialized (run 'no-mistakes init' first)" ]; then + RUN_OUT="" + fi if [ -n "$RUN_OUT" ]; then - run_branch=$(strip_quotes "$(nm_field branch)") - # Head equality, or the pipeline-owned-active exemption: while the - # pipeline owns this branch, the daemon's own branch attribution is - # authoritative and the lane head need not be a git object here - # (fm_nm_run_is_pipeline_owned_active in bin/fm-nm-run-lib.sh). - if [ -n "$run_branch" ] && [ "$run_branch" = "$CREW_BRANCH" ] \ - && { nm_run_head_matches_worktree || fm_nm_run_is_pipeline_owned_active "$RUN_OUT"; }; then - HAVE_RUN=1 - # Live-over-terminal (bin/fm-nm-run-lib.sh). Bare `axi status` answers - # with the most-recently-touched run, which after a pipeline crash is the - # dead run sitting at this worktree's exact commit while the live run - # that replaced it validates a descendant commit on the same branch. Both - # bind, so a terminal answer is provisional until the ledger has been - # asked whether this worktree also has a live run. Only a live word - # displaces it: a terminal run with no live sibling keeps its full - # `axi status` step and gate detail rather than degrading to the ledger. - if ! fm_nm_run_is_active "$RUN_OUT"; then - live_status=$(fm_nm_runs_status_for_worktree "$WT" "$CREW_BRANCH" "$(nm_runs_list)") - if [ "$(fm_nm_run_status_class "$live_status")" = live ]; then - COARSE_STATUS=$live_status - RUN_SOURCE=coarse + # The overview includes run ids and creation order, which the plain runs + # listing omits. Keep the primary empty-call bound above: a nonresponding + # CLI is not retried. Older CLI surfaces without the table retain the + # coarse fallback below, but cannot turn a replacement into a vague live + # verdict when its identity and gate cannot be read. + overview_ok=1 + run_overview=$(fm_nm_run_checked "$WT" "$NM_TIMEOUT" axi) || overview_ok=0 + [ -n "$run_overview" ] || emit unknown run-step "run inventory unavailable; run id: $(strip_quotes "$(nm_field id)")" + run_choice=$(fm_nm_select_run "$CREW_BRANCH" "$run_overview" "$WT") + [ "$overview_ok" = 1 ] || emit unknown run-step "run inventory unreadable; run ids: $(strip_quotes "$(nm_field id)"), ${run_choice##*|}" + case "$run_choice" in + unknown\|*) + known_run_id="" + if [ "$(strip_quotes "$(nm_field branch)")" = "$CREW_BRANCH" ]; then + known_run_id=$(strip_quotes "$(nm_field id)") fi - fi - else - # The active-or-most-recent run is for another branch, or it names this - # branch with a head this copy cannot verify (a pipeline-advanced fix - # round, or a rewritten tip). Deliberately nested inside - # `[ -n "$RUN_OUT" ]`: an empty/timed-out primary call means the CLI - # itself did not respond, so retrying it immediately with a second - # bounded call would just double the wait for no better answer. - COARSE_STATUS=$(fm_nm_runs_status_for_worktree "$WT" "$CREW_BRANCH" "$(nm_runs_list)") - if [ -n "$COARSE_STATUS" ]; then + emit unknown run-step "${run_choice#*|}${known_run_id:+; last reported run id: $known_run_id}" + ;; + selected\|*) + IFS='|' read -r _ selected_id selected_status candidate_ids <<< "$run_choice" + RUN_OUT=$(fm_nm_run_checked "$WT" "$NM_TIMEOUT" axi status --run "$selected_id") \ + || emit unknown run-step "selected run unreadable; run ids: $candidate_ids" + if [ "$(strip_quotes "$(nm_field id)")" != "$selected_id" ] \ + || [ "$(strip_quotes "$(nm_field branch)")" != "$CREW_BRANCH" ]; then + emit unknown run-step "selected run unavailable or mismatched; run ids: $candidate_ids" + fi + case "$(strip_quotes "$(nm_field status)")" in + pending|running|fixing|ci|awaiting_approval|fix_review|completed|failed|cancelled) ;; + *) emit unknown run-step "selected run status unverified; run ids: $candidate_ids" ;; + esac + if fm_nm_run_is_active "$RUN_OUT"; then current_class=live; else current_class=terminal; fi + if [ "$(fm_nm_run_status_class "$selected_status")" != "$current_class" ]; then + emit unknown run-step "selected run status disagrees with inventory; run ids: $candidate_ids" + fi + if nm_run_head_matches_worktree || fm_nm_run_is_pipeline_owned_active "$RUN_OUT"; then + HAVE_RUN=1 + elif [ -z "$(fm_nm_resolve_commit "$WT" "$(strip_quotes "$(nm_field head)")")" ]; then + if fm_nm_run_is_active "$RUN_OUT" \ + && [ "$(fm_nm_runs_status_for_worktree "$WT" "$CREW_BRANCH" "$(nm_runs_list)" "$(strip_quotes "$(nm_field head)")")" = running ]; then + HAVE_RUN=1 + else + emit unknown run-step "selected run code identity unverified; run ids: $candidate_ids" + fi + fi + SELECTED_RUN_ID=$selected_id + ;; + esac + if [ "$HAVE_RUN" = 0 ] && [ -z "$SELECTED_RUN_ID" ]; then + run_branch=$(strip_quotes "$(nm_field branch)") + # Head equality, or the pipeline-owned-active exemption: while the + # pipeline owns this branch, the daemon's own branch attribution is + # authoritative and the lane head need not be a git object here + # (fm_nm_run_is_pipeline_owned_active in bin/fm-nm-run-lib.sh). + if [ -n "$run_branch" ] && [ "$run_branch" = "$CREW_BRANCH" ] \ + && { nm_run_head_matches_worktree || fm_nm_run_is_pipeline_owned_active "$RUN_OUT"; }; then HAVE_RUN=1 - # A branch-matching answer the strict rule rejected is this branch's - # own current run once the ledger proves the pipeline-owned - # continuation, so its axi TOON is the authoritative run detail - # (RUN_SOURCE stays full); only a foreign-branch answer leaves - # coarse status-word detail. - [ "$run_branch" = "$CREW_BRANCH" ] || RUN_SOURCE=coarse + # Without run ids, contradictory liveness cannot prove precedence. + # A live replacement also needs an id-addressed status read: a bare + # "running" row cannot tell working from waiting at a gate. + ledger_status=$(fm_nm_runs_status_for_worktree "$WT" "$CREW_BRANCH" "$(nm_runs_list)") + if fm_nm_run_is_active "$RUN_OUT"; then + if [ "$(fm_nm_run_status_class "$ledger_status")" = terminal ]; then + emit unknown run-step "run records disagree; run ids: $(strip_quotes "$(nm_field id)"), competing identity unavailable" + fi + else + if [ "$(fm_nm_run_status_class "$ledger_status")" = live ]; then + emit unknown run-step "replacement run identity unavailable; run ids: $(strip_quotes "$(nm_field id)"), replacement unavailable" + elif [ -n "$ledger_status" ] \ + && [ "$ledger_status" != "$(strip_quotes "$(nm_field status)")" ] \ + && [ "$ledger_status" != "$(strip_quotes "$(nm_field outcome)")" ]; then + COARSE_STATUS=$ledger_status + RUN_SOURCE=coarse + fi + fi + else + # The active-or-most-recent run is for another branch, or it names this + # branch with a head this copy cannot verify (a pipeline-advanced fix + # round, or a rewritten tip). Deliberately nested inside + # `[ -n "$RUN_OUT" ]`: an empty/timed-out primary call means the CLI + # itself did not respond, so retrying it immediately with a second + # bounded call would just double the wait for no better answer. + COARSE_STATUS=$(fm_nm_runs_status_for_worktree "$WT" "$CREW_BRANCH" "$(nm_runs_list)") + if [ -n "$COARSE_STATUS" ]; then + HAVE_RUN=1 + # A branch-matching answer the strict rule rejected is this branch's + # own current run once the ledger proves the pipeline-owned + # continuation, so its axi TOON is the authoritative run detail + # (RUN_SOURCE stays full); only a foreign-branch answer leaves + # coarse status-word detail. + [ "$run_branch" = "$CREW_BRANCH" ] || RUN_SOURCE=coarse + fi fi fi fi @@ -746,13 +803,9 @@ if [ "$HAVE_RUN" = 1 ]; then RUN_STATUS="" if [ "$RUN_SOURCE" = coarse ]; then # No step/gate detail is available from the plain runs list - only ever - # true/working, done, or failed. A crew genuinely parked at a gate still - # gets full detail once `axi status` reports its own branch again (e.g. - # once its own step is the most-recently-touched one), and its own - # needs-decision/blocked status-log append (a captain-relevant VERB) is - # surfaced by each supervisor's span classification (fm-classify-lib.sh's - # status_span_first_actionable) regardless of this coarse-vs-full - # distinction, so a real gate is never silently missed. + # working, done, failed, or unknown. Gate detail requires the identity-aware + # read above. The status event span remains independently available to the + # supervisor through fm-classify-lib.sh's status_span_first_actionable. case "$COARSE_STATUS" in running) RUN_STATE=working; RUN_DETAIL="validating (background run)" ;; completed) RUN_STATE="done"; RUN_DETAIL="run completed" ;; @@ -893,6 +946,7 @@ if [ "$HAVE_RUN" = 1 ]; then ;; esac + [ -z "$SELECTED_RUN_ID" ] || RUN_DETAIL="$RUN_DETAIL${SEP}run: $SELECTED_RUN_ID" emit "$RUN_STATE" run-step "$RUN_DETAIL" fi diff --git a/bin/fm-nm-run-lib.sh b/bin/fm-nm-run-lib.sh index ed71d315fd8..3377dffa4cf 100644 --- a/bin/fm-nm-run-lib.sh +++ b/bin/fm-nm-run-lib.sh @@ -86,15 +86,9 @@ fm_nm_resolve_commit() { # # the ancestor rule (observed 2026-08: a crashed validation daemon left a failed # run at the worktree's own commit while the live run that replaced it validated # a descendant commit on the same branch). -# When several runs bind, a LIVE run always outranks a terminal one, whichever -# match rule each one used, because a terminal run can be the corpse of a -# crashed attempt while the live one is what is actually validating this code. -# Within one liveness class the selecting caller's existing precedence is -# unchanged - for the runs ledger, fm_nm_runs_status_for_worktree's -# newest-row-decides rule below. -# fm_nm_run_status_class next classifies a recorded status word for that -# comparison, and a word it cannot classify keeps the caller's own precedence -# rather than being held back for a live row to displace. +# Head compatibility alone does not establish precedence between runs. +# fm_nm_select_run below owns identity-aware selection for current-state reads; +# fm_nm_runs_status_for_worktree owns the coarse ledger fallback. fm_nm_head_matches_worktree() { # local wt=$1 run_head=$2 local_full run_full [ -n "$run_head" ] || return 1 @@ -105,19 +99,177 @@ fm_nm_head_matches_worktree() { # git -C "$wt" merge-base --is-ancestor "$local_full" "$run_full" 2>/dev/null } -# Liveness class of a recorded run's status word, echoed as "terminal", "live", -# or "unknown", for the live-over-terminal selection rule above. -# The coarse `no-mistakes runs` ledger emits exactly these four status words; an +# Liveness class of a recorded ledger status word. +# The coarse `no-mistakes runs` ledger emits database status words; an # `axi status` run object reports its terminal result through its own outcome # field as well, which fm_nm_run_is_active below checks directly. fm_nm_run_status_class() { # case "${1:-}" in completed|failed|cancelled) printf 'terminal' ;; - running) printf 'live' ;; + pending|running) printf 'live' ;; *) printf 'unknown' ;; esac } +# Select from a complete `no-mistakes axi` overview with the existing awk +# toolchain. A capped overview requires an optional Python 3 sqlite3 reader +# for a read-only same-branch query of NM_HOME/state.sqlite (default: +# ~/.no-mistakes/state.sqlite; relative NM_HOME resolves from the worktree). +# If that reader or inventory is unavailable, report unknown with available +# candidate ids rather than treating the displayed window as complete. +# Structural completeness applies to the whole table; semantic validation +# applies only to the requested branch, after complete identity lookup when +# capped. Branch names are matched exactly without a character whitelist. +# Its rows are ordered by creation time descending (not last update), then id. +# The newest same-branch row is the candidate regardless of outcome: an older +# live run must not hide a newer failure. If the newest is live and another +# same-branch live run exists, neither has exclusive authority: report all +# candidate ids as unknown. A newer live row can replace cancelled history, +# but the caller must fetch its full status BY ID and prove branch/head or +# active pipeline custody before using its steps. Never reuse another run's +# gate detail. This is a read-only selection, not teardown authorization. +# +# Prints selected|id|status|candidate-ids, unknown|reason, absent (no row +# for this branch), or unavailable (CLI has no overview table). Malformed or +# structurally truncated tables report unknown, retaining every readable +# same-branch candidate id. +fm_nm_select_run() { # + local selection inventory available_ids + selection=$(printf '%s\n' "$2" | awk -v branch="$1" ' + function scalar(s) { + sub(/^[ \t]+/, "", s); sub(/[ \t]+$/, "", s) + if (s ~ /^".*"$/) s = substr(s, 2, length(s)-2) + return s + } + function row_fields(s, f, i, ch, n, quoted, escaped) { + for (i in f) delete f[i] + n = 1; f[n] = "" + for (i = 1; i <= length(s); i++) { + ch = substr(s, i, 1) + if (escaped) { f[n] = f[n] ch; escaped = 0 } + else if (quoted && ch == "\\") escaped = 1 + else if (ch == "\"") quoted = !quoted + else if (!quoted && ch == ",") { n++; f[n] = "" } + else f[n] = f[n] ch + } + if (quoted || escaped) return 0 + for (i = 1; i <= n; i++) { + sub(/^[ \t]+/, "", f[i]); sub(/[ \t]+$/, "", f[i]) + } + return n + } + /^count: / { + if (counts++) bad = 1 + count = scalar(substr($0, 8)) + if (count !~ /^[0-9]+ of [0-9]+ total$/) bad = 1 + split(count, c, " "); shown = c[1]; total = c[3] + } + /^runs\[[0-9]+\]\{id,branch,status,head,pr\}:$/ { + if (found++) bad = 1 + expected = $0; sub(/^runs\[/, "", expected); sub(/\].*$/, "", expected) + inrows = 1; next + } + /^runs\[/ { bad = 1; found = 1 } + inrows && /^[ \t]+/ { + seen++ + n = row_fields($0, f) + if (n != 5) bad = 1 + id = f[1]; br = f[2]; st = f[3]; head = f[4] + if (br != branch) next + if (id ~ /^[A-Za-z0-9_-]+$/) { + if (known[id]++) invalid_run = 1 + else ids = ids (ids == "" ? "" : ", ") id + } + if (n != 5) next + if (id !~ /^[A-Za-z0-9_-]+$/ || + st !~ /^[a-z_-]+$/ || head !~ /^[a-fA-F0-9]+$/ || length(head) < 7 || length(head) > 40) { + invalid_run = 1; next + } + if (first == "") { first = id; first_status = st } + if (st == "running" || st == "pending") live++ + if (st !~ /^(pending|running|completed|failed|cancelled)$/) unknown_status = 1 + next + } + inrows { inrows = 0 } + END { + if (!found) print "unavailable" + else if (bad || counts != 1 || seen != expected || seen != shown || total < shown) + print "unknown|unreadable runs table; run ids: " ids + else if (shown < total) print "incomplete|" ids + else if (invalid_run) print "unknown|unreadable runs table; run ids: " ids + else if (unknown_status) print "unknown|unrecognized run status; run ids: " ids + else if (first == "") print "absent" + else if ((first_status == "running" || first_status == "pending") && live > 1) + print "unknown|competing live runs; run ids: " ids + else print "selected|" first "|" first_status "|" ids + } + ') + case "$selection" in + incomplete\|*) available_ids=${selection#*|} ;; + *) printf '%s\n' "$selection"; return ;; + esac + if ! inventory=$(python3 - "$1" "$2" "$3" "$available_ids" 2>/dev/null <<'PY' +import json +import os +import re +import sqlite3 +import sys +from contextlib import closing +from pathlib import Path + +branch, overview, worktree, available_ids = sys.argv[1:] +ids = available_ids.split(", ") if available_ids else [] +try: + repos = [line[6:].strip() for line in overview.splitlines() if line.startswith("repo: ")] + if len(repos) != 1: + raise ValueError + repo_path = json.loads(repos[0]) if repos[0].startswith('"') else repos[0] + if not isinstance(repo_path, str) or not os.path.isabs(repo_path): + raise ValueError + root = Path(os.environ.get("NM_HOME") or Path.home() / ".no-mistakes") + if not root.is_absolute(): + root = Path(worktree) / root + with closing(sqlite3.connect((root / "state.sqlite").as_uri() + "?mode=ro", uri=True, timeout=1)) as db: + db.execute("BEGIN") + repo = db.execute("SELECT id FROM repos WHERE working_path = ?", (repo_path,)).fetchall() + if len(repo) != 1: + raise ValueError + rows = db.execute( + "SELECT id, branch, status, head_sha FROM runs WHERE repo_id = ? AND branch = ? " + "ORDER BY created_at DESC, id DESC", (repo[0][0], branch) + ).fetchall() + displayed_ids = set(ids) + for row in rows: + if isinstance(row[0], str) and re.fullmatch(r"[A-Za-z0-9_-]+", row[0]) and row[0] not in ids: + ids.append(row[0]) + if not displayed_ids.issubset(row[0] for row in rows): + raise ValueError + for row in rows: + if (not all(isinstance(value, str) for value in row) + or not re.fullmatch(r"[A-Za-z0-9_-]+", row[0]) or row[1] != branch + or not re.fullmatch(r"[a-z_-]+", row[2]) or not re.fullmatch(r"[a-fA-F0-9]{7,40}", row[3])): + raise ValueError + print("count: %d of %d total" % (len(rows), len(rows))) + print("runs[%d]{id,branch,status,head,pr}:" % len(rows)) + for row in rows: + print(" " + ",".join(json.dumps(value, ensure_ascii=False) for value in row) + ',""') +except (ValueError, OSError, sqlite3.Error): + print("unknown|complete same-branch run inventory unreadable; run ids: " + ", ".join(ids)) +PY + ); then + printf 'unknown|complete same-branch run inventory reader unavailable; run ids: %s\n' "$available_ids" + return + fi + case "$inventory" in + unknown\|*) selection=$inventory ;; + *) selection=$(fm_nm_select_run "$1" "$inventory" "$3") ;; + esac + case "$selection" in + selected\|*|unknown\|*|absent) printf '%s\n' "$selection" ;; + *) printf 'unknown|complete same-branch run inventory unreadable; run ids: %s\n' "$available_ids" ;; + esac +} + # branch_sync.state from captured `axi status` TOON $1: the scalar directly # under the top-level `branch_sync:` block. The first `state:` inside the # block is the direct child (the nested local/pipeline/target/remote @@ -179,32 +331,12 @@ fm_nm_run_is_pipeline_owned_active() { # # printed. Anything else (no anchor row, an anchor that is merely an # ancestor, a terminal unresolvable row) prints nothing, so branch-name # coincidence, arbitrary remote state, and other tasks' runs never match. -# The one exception to newest-row-decides is the live-over-terminal rule stated -# with fm_nm_head_matches_worktree above, and it only ever replaces a TERMINAL -# answer with a LIVE one: when the newest row binds but is terminal, the older -# rows are scanned for a live row that ALSO binds to this worktree, and that -# row's status word is printed instead. A live row whose head resolves in this -# copy binds by fm_nm_head_matches_worktree. A live row whose head does NOT -# resolve (the routine shape: the pipeline's fix-round commits live only in the -# gate repo) binds ONLY when the held terminal row sits at EXACTLY the worktree -# HEAD - the same exact-equality anchor the pipeline-continuation rule above -# requires, so branch-name coincidence and other tasks' runs still never -# match. A terminal newest row is the corpse of a crashed attempt whenever a -# live run for the same worktree is still on the ledger, so it is not the -# present. Nothing else widens: a newest row that does not bind still ends the -# scan, a newest row whose class is live or unclassifiable is still answered -# as-is, the anchored pipeline-continuation path is untouched, and with no live -# sibling the newest terminal word is still what is printed. +# An older live row never displaces a newer terminal result. # Read-only: git reads resolve objects in place; custody never changes. fm_nm_runs_status_for_worktree() { # [expected-head] local wt=$1 branch=$2 list=$3 expected_head=${4:-} local local_full row_full row st br sha day clock pr extra year_num month_num day_num max_day pending_st='' - # Set only by the newest binding row when its status classifies terminal, and - # printed when the scan ends without finding a live row for this worktree. It - # is the sole reason the scan continues past the newest row, and every exit - # below leaves the loop rather than returning, so a malformed older row can - # never swallow an answer the newest row had already decided. - local decided='' decided_exact='' + local decided='' local_full=$(git -C "$wt" rev-parse HEAD 2>/dev/null) || return 0 [ -n "$list" ] || return 0 while IFS= read -r row; do @@ -237,20 +369,6 @@ fm_nm_runs_status_for_worktree() { # [ex esac [ "$day_num" -ge 1 ] && [ "$day_num" -le "$max_day" ] || break [ "$br" = "$branch" ] || continue - if [ -n "$decided" ]; then - # Live-over-terminal: the newest row bound to this worktree but is a - # terminal record, so the older rows are searched for a live run that - # binds to the same worktree by the same head rule. Only such a row - # displaces the held terminal word; anything else leaves it standing. - [ "$(fm_nm_run_status_class "$st")" = live ] || continue - if [ -n "$(fm_nm_resolve_commit "$wt" "$sha")" ]; then - fm_nm_head_matches_worktree "$wt" "$sha" || continue - else - [ -n "$decided_exact" ] || continue - fi - decided=$st - break - fi if [ -n "$pending_st" ]; then # This is the row immediately older than the active unresolvable row: # the only admissible anchor, and only exact head equality proves the @@ -272,12 +390,6 @@ fm_nm_runs_status_for_worktree() { # [ex if [ -n "$row_full" ]; then if fm_nm_head_matches_worktree "$wt" "$sha"; then decided=$st - # A live or unclassifiable word is this worktree's current answer and - # ends the scan; only a terminal one keeps looking for a live sibling. - if [ "$(fm_nm_run_status_class "$st")" = terminal ]; then - [ "$row_full" != "$local_full" ] || decided_exact=1 - continue - fi fi break fi diff --git a/bin/fm-promote.sh b/bin/fm-promote.sh index 7f4118d776b..c76a6a63134 100755 --- a/bin/fm-promote.sh +++ b/bin/fm-promote.sh @@ -223,6 +223,7 @@ EOF ## Firstmate spec $PROMOTION_SHIP_SPEC + EOF promote_delivery_contract } > "$TMP" || { echo "error: could not render ship instructions for mode=$MODE" >&2; exit 1; } diff --git a/bin/fm-session-lock-lib.sh b/bin/fm-session-lock-lib.sh index 91c901f820b..7dec38a73a0 100644 --- a/bin/fm-session-lock-lib.sh +++ b/bin/fm-session-lock-lib.sh @@ -181,3 +181,29 @@ $pids EOF return 1 } + +# True when state dir $1 records a live verified harness outside this process's +# contiguous harness ancestry. Sets FM_SESSION_LOCK_FOREIGN_OWNER_PID for a +# diagnostic caller. Malformed, missing, dead, and ancestry-uncertain locks are +# not foreign-owner evidence. +# shellcheck disable=SC2034 # Output global, read by the sourcing guard caller. +FM_SESSION_LOCK_FOREIGN_OWNER_PID= +fm_session_lock_foreign_owner_live() { + local state=$1 lock_pid pids pid + FM_SESSION_LOCK_FOREIGN_OWNER_PID= + [ -f "$state/.lock" ] && [ ! -L "$state/.lock" ] || return 1 + lock_pid=$(cat "$state/.lock" 2>/dev/null || true) + case "$lock_pid" in + ''|*[!0-9]*) return 1 ;; + esac + fm_harness_pid_alive "$lock_pid" || return 1 + pids=$(fm_harness_ancestry_pids) || return 1 + while IFS= read -r pid; do + [ "$pid" = "$lock_pid" ] && return 1 + done < ... return 1 fi done - for key in "${missing_keys[@]}"; do + for key in "${missing_keys[@]+"${missing_keys[@]}"}"; do marker="$STATE/.churn-since-$key" if (set -C; printf '%s' "$now_s" > "$marker") 2>/dev/null; then created_keys+=("$key") continue fi - for created in "${created_keys[@]}"; do + for created in "${created_keys[@]+"${created_keys[@]}"}"; do rm -f "$STATE/.churn-since-$created" done return 1 done for key in "${churned_keys[@]}"; do if ! rm -f "$STATE/.stale-$key" "$STATE/.wedge-escalations-$key"; then - for created in "${created_keys[@]}"; do + for created in "${created_keys[@]+"${created_keys[@]}"}"; do rm -f "$STATE/.churn-since-$created" done return 1 diff --git a/docs/architecture.md b/docs/architecture.md index 1296396218e..8c7aa8ea762 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -88,9 +88,8 @@ A turn-ended-only queue row omits its historical status annotation when that sta Any direct or remaining historical annotation prints every status line unread at the presentation cursor instead of replaying only the latest line. `bin/fm-crew-state.sh ` is the cheap current-state read for an actionable heartbeat review: it attributes an active or terminal no-mistakes run under the shared run-attribution contract, then keeps that run-step authoritative even if the pane has closed, except that a `blocked:` event reporting a refused or missing daemon socket outranks a potentially stale active run record only while that socket-down declaration is itself the log's latest recognized event, since any later event, including another `blocked:` one, means the crew moved on. For other daemon, timeout, or unreachability claims, a running or fixing run with recent pipeline-reported activity supersedes the event and names reattachment as the recovery instead of surfacing a false block. -[`bin/fm-nm-run-lib.sh`](../bin/fm-nm-run-lib.sh)'s header owns the exact branch, head, pipeline-custody, and newest-first attribution rules. -It also owns which binding run wins when more than one recorded run binds to the same worktree: a live run outranks a terminal one, so a crashed run sitting at the worktree's own commit never reports a healthy task as failed while its live successor is still validating. -A run head the task copy cannot resolve locally is attributed only when the pipeline's own runs ledger proves it is an active continuation of the submitted head, so a pipeline fix round never reads as an older failed run. +[`bin/fm-nm-run-lib.sh`](../bin/fm-nm-run-lib.sh) owns branch, head, and pipeline-custody attribution, plus complete same-branch run selection, optional inventory lookup, and ambiguity reporting. +[`tests/fm-crew-state.test.sh`](../tests/fm-crew-state.test.sh) covers run selection; its [capture provenance and live-evidence limits](../tests/captures/no-mistakes-v1.70.1/README.md) distinguish recorded inputs from composed scenarios. During no-mistakes' `ci` monitor phase, it also reads the ci step log tail because `axi status` reports both "still waiting on checks" and "checks green, waiting on merge" as `ci,running`. The most recent recognized ci log marker wins, so checks-green monitoring reports done while a later re-arm, failed-check, or issue marker returns the crew to working. `bin/fm-crew-state.sh` owns the evidence guard that recognizes ended CI monitors after green checks, including cancelled runs and skipped rebase steps; a passed run alone never proves a forge merge. diff --git a/docs/calm-mode-feasibility.md b/docs/calm-mode-feasibility.md index f67f8c12dc5..68dab0cdc50 100644 --- a/docs/calm-mode-feasibility.md +++ b/docs/calm-mode-feasibility.md @@ -224,7 +224,7 @@ The test fixture enumerates every class below through the centralized policy, an | --- | --- | --- | | `genuine-user-prompt` | `UserMessageComponent` | Visible, including every tested operational near miss. | | `genuine-agent-response` | Assistant text in `AssistantMessageComponent` | Visible. | -| `assistant-working-note` | Assistant text in an `AssistantMessageComponent` message the model did not end its response with, identified by its own `stopReason` of `toolUse`, or of `length` with tool calls present | The text blocks are removed from the shallow presentation copy before layout, so a `toolUse` message carrying only narration occupies zero rows (verified on Pi 0.84.1); a still-streaming `pending` message is never filtered, so narration is briefly visible before the marker flips. | +| `assistant-working-note` | Assistant text in an `AssistantMessageComponent` message the model did not end its response with, identified by its own `stopReason` of `toolUse`, or of `length` with tool calls present | Each settled text block follows the cross-harness preservation contract in [`calm.md`](calm.md); hidden blocks are removed from the shallow presentation copy before layout, a `toolUse` message carrying only short narration occupies zero rows (verified on Pi 0.84.1), and a still-streaming `pending` message is never filtered. | | `assistant-thinking` | Thinking content in `AssistantMessageComponent` | Collapsed reasoning is removed from the shallow presentation copy before layout and occupies zero rows; explicit expansion renders the original reasoning. | | `assistant-tool-call` | `ToolExecutionComponent` | Seven built-ins, `fm_watch_arm_pi`, and `fm_branch_outcomes` hidden; other arbitrary custom tools remain an unsupported boundary. | | `tool-result` | `ToolExecutionComponent` | Text results for the controlled tools hidden; other arbitrary custom results remain an unsupported boundary. | @@ -710,7 +710,7 @@ $ bin/fm-test-run.sh tests/fm-calm-claude-mod.test.sh ok - the Calm mod is one hooks module, linked into the project's auto-load path, with no command, skill, agent, or classic hook path that bypasses its exact opt-in ok - the Pi working ship renders byte-for-byte the shared sprite core's frame painted in standard ANSI, at every width, cadence step, freeze, clamp, and reset ok - the Raster packing lays the shared frame out row-major with the sprite's palette, plain padding, default backgrounds, BMP glyphs, clipping, and a standard base64 encoding -ok - the Calm policy resolves the shared preference exactly as Pi does, reads on, max, and off as Pi does, and classifies working notes by stop reason, tool use, and restored transcript shape +ok - the Calm policy resolves the shared preference exactly as Pi does, reads on, max, and off as Pi does, and shares Pi's 240-character-or-newline preservation behavior while classifying working notes by stop reason, tool use, and restored transcript shape ok - the mod's operational-input classifier agrees with bin/fm-operational-input.sh on all 77 corpus cases: every current kind the owner encodes, every legacy shape, and every near miss $ bin/fm-test-run.sh tests/fm-calm-pi-extension.test.sh diff --git a/docs/calm.md b/docs/calm.md index 743ca290daf..f590027df40 100644 --- a/docs/calm.md +++ b/docs/calm.md @@ -3,6 +3,8 @@ Calm is Firstmate's conversation-only transcript presentation toggle. It is fully supported on Pi, and available on Claude Code behind that harness's default-off early-access function-hooks flag, as the [Claude Code](#claude-code) section below describes. It is off by default, and the last `/calm` choice persists for the effective Firstmate home across session starts and resumes on either harness, through the one shared preference file [`configuration.md`](configuration.md#calm-preference-configcalm) owns. +Across both harnesses, Calm evaluates each settled assistant text block from a model step that stopped to call tools, or exhausted its token limit while carrying tool calls. +It hides a block only when its raw text contains no newline and its trimmed length is below `CALM_PRESERVE_MIN_CHARS` (240); a newline or at least 240 trimmed characters preserves the block as substantive captain-facing content, while streaming text and the genuine reply that ends a response remain visible. ## Pi @@ -17,10 +19,9 @@ Hidden elapsed time does not advance the animation, and a resize while hidden cl A fresh Pi session or new Calm extension lifetime starts at the normal initial position. Very narrow terminals fall back to a smaller deterministic sprite. While Calm is off, Pi's stock working row is left exactly as Pi renders it. -Calm hides collapsed thinking labels, mid-turn assistant working notes, the shells for the Pi built-in tool names Calm owns, the `fm_watch_arm_pi` and `fm_branch_outcomes` tool shells, and canonically classified Firstmate operational user rows. -A mid-turn working note is assistant text in a message the model did not end its response with, identified by that message's own `stopReason` of `toolUse`, or of `length` with tool calls present. -Hiding it removes the narration a model emits alongside its tool calls, while the genuine reply that ends a response stays visible. -Text that is still streaming is never hidden, because suppressing it would also stop a genuine reply from streaming, so a working note is briefly visible before its row collapses. +Calm hides collapsed thinking labels, the mid-turn assistant working-note blocks governed by the shared preservation rule above, the shells for the Pi built-in tool names Calm owns, the `fm_watch_arm_pi` and `fm_branch_outcomes` tool shells, and canonically classified Firstmate operational user rows. +Pi applies that rule independently to each text block, so a short working note can hide beside preserved substantive content in the same message. +A working note is briefly visible while it streams before its settled row collapses. The narration is hidden only from the live transcript presentation, and remains in the message, model context, session storage, and `/export` artifacts. The operational inputs Calm classifies remain ordinary user-role messages, while Pi's transcript layout renders their complete rows at zero height. The session-start nudge remains on its existing non-displayed custom-message path. @@ -51,7 +52,7 @@ If the other extension wins, a session-start console diagnostic names the tool a [`calm-mode-feasibility.md`](calm-mode-feasibility.md) owns the version-scoped renderer taxonomy, built-in override constraints, and empirical evidence. [`configuration.md`](configuration.md#calm-preference-configcalm) owns the persisted preference file and resolution rules. -`.pi/extensions/lib/fm-calm-visibility.ts` owns the visibility policy, `.pi/extensions/lib/fm-calm-operational-user-layout.ts` owns the zero-height operational-user row adapter, and `.pi/extensions/lib/fm-calm-working-ship.ts` owns Pi's animated working presentation over the sprite geometry both harnesses share in `.claude/mods/firstmate-calm/lib/fm-calm-working-ship-sprite.ts`. +`.pi/extensions/lib/fm-calm-visibility.ts` owns the visibility policy, `.claude/mods/firstmate-calm/lib/fm-calm-preservation.ts` owns the shared substantive mid-turn text rule that Pi imports through its tracked symlink, `.pi/extensions/lib/fm-calm-operational-user-layout.ts` owns the zero-height operational-user row adapter, and `.pi/extensions/lib/fm-calm-working-ship.ts` owns Pi's animated working presentation over the sprite geometry both harnesses share in `.claude/mods/firstmate-calm/lib/fm-calm-working-ship-sprite.ts`. Regression entry points: @@ -76,8 +77,7 @@ On Claude Code the boat is painted in Claude Code's own theme colors rather than The family follows the `theme` setting by its prefix, `dark` or `light`, is re-read when the theme changes, and uses the light set as the both-readable fallback for `auto`, custom, missing, or unreadable values; the Pi extension keeps its standard ANSI blue and yellow. Tool rows, tool result blocks, and folded tool groups draw at zero height, so a turn that used tools takes the same space as one that did not. A user row whose text the canonical operational-input parser recognizes, a Firstmate session-start, watcher, turn-end guard, away-supervisor, launch-brief, or branch-outcome envelope, a from-firstmate routed message, or one of the narrow pre-protocol shapes kept for old transcripts, draws at zero height; every other user row, including near misses such as a quoted or ASCII-only marker, stays visible. -A mid-turn working note, the text of a model step that stopped to call tools or ran out of tokens while calling them, draws at zero height once that step settles only when its raw text contains no newline and its trimmed length is below the 240-character preservation threshold. -Mid-turn content whose raw text contains a newline or whose trimmed length is at least 240 characters is preserved and treated as a final reply, including when `claude --continue` restores the transcript. +Assistant text follows the shared per-block preservation rule above, including when `claude --continue` restores the transcript. Toggling Calm redraws every hooked row already on screen, so rows drawn before the toggle hide or restore retroactively, and the preference is read before the first row draws. Nothing is rewritten: hidden rows remain in the message, model context, session storage, and exports, and the mod never touches tool execution, prompts, or the stored transcript. diff --git a/docs/configuration.md b/docs/configuration.md index 5a64feb92f2..729965492f6 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -1100,7 +1100,7 @@ FM_WHEN_OUTPUT_TAIL_BYTES=8192 # bound on the command-output tail insid FM_CODEX_WATCH_CHECKPOINT=180 # seconds per foreground watcher checkpoint in Codex primary supervision FM_CREW_STATE_NM_TIMEOUT=10 # seconds allowed per no-mistakes query inside fm-crew-state.sh FM_TEARDOWN_NM_TIMEOUT=10 # seconds allowed per no-mistakes query or abort inside fm-teardown.sh -FM_CREW_STATE_RUNS_LIMIT=200 # recent no-mistakes run rows scanned when the runs ledger is consulted: axi status cannot be attributed directly, or its answer is terminal and may have a live sibling run +FM_CREW_STATE_RUNS_LIMIT=200 # plain runs-ledger rows scanned for fallback attribution; does not change the CLI's AXI overview window (selection owner: bin/fm-nm-run-lib.sh) FM_TEARDOWN_NM_RUNS_LIMIT=200 # recent no-mistakes run rows scanned to prove an unresolved-head parked run belongs to teardown's task FM_CREW_STATE_BIN=bin/fm-crew-state.sh # test override for the current-state reader used by working/paused watcher triage FM_MAIL_USER= # mail-plane IMAP/SMTP login, from .env or environment (docs/configuration.md "Mail plane") diff --git a/docs/documentation-audiences.json b/docs/documentation-audiences.json index f639220b551..e459e95006a 100644 --- a/docs/documentation-audiences.json +++ b/docs/documentation-audiences.json @@ -511,6 +511,10 @@ { "path": "skills/stow/SKILL.md", "audience": "public-product" + }, + { + "path": "tests/captures/no-mistakes-v1.70.1/README.md", + "audience": "maintainer-verification" } ] } diff --git a/docs/supervision-protocols/claude.md b/docs/supervision-protocols/claude.md index 477476da54d..5de60e63eae 100644 --- a/docs/supervision-protocols/claude.md +++ b/docs/supervision-protocols/claude.md @@ -18,8 +18,8 @@ When this session owns supervision and away mode is not active: No PreToolUse hook denies fleet commands based on watcher status. [`watcher-continuity.md`](../watcher-continuity.md) owns the exact session-lock recovery boundary. 8. The turn-end guard (`bin/fm-turnend-guard.sh --claude`) remains the final backstop. - It requires the PID-strict live-watcher and fresh-beacon predicate at the Stop boundary; [`turnend-guard.md`](../turnend-guard.md#guard-predicates) owns the distinct model-aware mid-turn pull-guard rules. - It allows the stop when a watcher is healthy or an open auto-arm generation claim owns recovery, while fresh failure epochs advance the bounded one-time attended fail-open progression described in [`turnend-guard.md`](../turnend-guard.md). + It requires the PID-strict live-watcher and fresh-beacon predicate at the Stop boundary, except for the Claude-specific foreign-live-owner safe exit owned by [`turnend-guard.md`](../turnend-guard.md#guard-predicates); that document also owns the distinct model-aware mid-turn pull-guard rules. + Otherwise, it allows the stop when a watcher is healthy or an open auto-arm generation claim owns recovery, while fresh failure epochs advance the bounded one-time attended fail-open progression described there. 9. Waiting on the hook-owned cycle is silent: do not send idle progress while the watcher is parked. The watcher itself remains `bin/fm-watch.sh`, and `bin/fm-watch-arm.sh` remains the verified arm wrapper that the Stop hook foregrounds. diff --git a/docs/supervision-protocols/pi.md b/docs/supervision-protocols/pi.md index 20e27bdc13c..51cb1f9be86 100644 --- a/docs/supervision-protocols/pi.md +++ b/docs/supervision-protocols/pi.md @@ -25,7 +25,9 @@ A no-change heartbeat outcome explicitly reported with `task=fleet` and `silent= A captain-facing outcome instead appears as one exact, sequence-keyed visible transcript entry, and then arrives in this conversation as one hidden supervision processing request listing each `[seq N] task: summary` it covers. That request is the one turn in which MAIN processes the outcome: give the captain a visible response where one is due, answer or escalate a decision, act on a blocker or failure, or record that no further action is needed, then call the `fm_branch_processed` tool with the highest sequence the request listed, exactly once. Only that call closes the outcome; an unrelated, empty, or paraphrased answer leaves it open, and the current unprocessed sequence set is presented again at the next run boundary and at session start until it is acknowledged. -The persisted entry is already the captain-visible record, so MAIN must not re-emit it verbatim merely because it appeared. +The persisted entry is already the captain-visible record, so MAIN must not re-emit it verbatim merely because it appeared; this prevents repetition but does not replace any captain-facing outcome response required by `AGENTS.md` section 9. +Regression example - keep verbatim and never condense away: `[seq 41] claude-mod: implementation complete, ready for review` requires relaying a captain-facing outcome response, not just `Captain, shipshape.`. +A merge ask with no URL that leans on the dim anchor violates `AGENTS.md` section 9. Before MAIN steers, controls lifecycle, or cleans up a task, claim its lease with `bin/fm-lease.sh claim ` and release it afterwards; a refused claim means the branch is acting on that task right now. This conversation still receives every other fleet-wide or unresolvable wake, the branch's wakes when it is unavailable or a legacy away daemon flag is active, and every watcher-failure alarm regardless, so the arm and repair contract above is unchanged. Treat the merged fleet event as already handled for fleet operations: MAIN must not re-drain, re-run, or acknowledge it. diff --git a/docs/turnend-guard.md b/docs/turnend-guard.md index 34297b9bf90..dc435bb614e 100644 --- a/docs/turnend-guard.md +++ b/docs/turnend-guard.md @@ -34,6 +34,9 @@ Every mode treats `state/x-watch.check.sh` as supervision need, so Relay polling A custom check registered with `bin/fm-check-register.sh` counts the same way, so an operator's home-level poll keeps running after the last task is torn down. Otherwise it calls `fm_watcher_healthy [grace-seconds] [home]` from `bin/fm-wake-lib.sh`, the same PID-strict identity-matched lock and fresh-beacon check used by `bin/fm-watch-arm.sh`: a stale beacon blocks even when a watcher pid is live, and a fresh leftover beacon blocks when the lock is missing, dead, or identity-mismatched. The turn-end guard needs that strict check because it fires at the turn boundary, where the auto-arm is bringing a fresh watcher up for the upcoming idle period, and it cooperates with that arm rather than trusting a beacon left by the cycle that just ended. +When an active home instead has a live session lock held by a verified harness outside the current session's contiguous ancestry, the Claude guard emits a read-only ownership diagnostic and allows the turn to end safely. +That Claude session cannot arm or repair the home without stealing the live owner's lock, so blocking it would create an unbounded loop; the lock-owning session remains responsible for restoring supervision. +Malformed, absent, dead, or ancestry-uncertain lock records do not satisfy this Claude-specific exception and retain the ordinary guard behavior. `bin/fm-guard.sh`, the pull warning, instead uses the model-aware `fm_watcher_supervision_verdict` from the same library, because it fires mid-turn when the auto-arm model runs no watcher at all. Under the Claude Stop auto-arm model a beacon fresh within grace is healthy even with no live watcher process. A stale beacon is still healthy while `fm_autoarm_midturn_healthy` in `bin/fm-wake-lib.sh` proves a Claude rewake explains the mid-turn gap: the rewake is bound to the current recovery generation and live session-lock owner, and no later watcher beacon or exhausted-failure marker supersedes it, because that session's turn-end will re-arm. @@ -95,6 +98,7 @@ Both payloads carry `stop_hook_active`. In the default Codex mode, a true value lets the second stop finish after one forced continuation. Claude runs the guard with `--claude`, which ignores `stop_hook_active` and cooperates with the Stop-owned auto-arm. +Before the Claude cooperative budget can re-block a Stop, the guard checks for a live foreign session-lock owner and takes the same safe diagnostic exit described under "Guard predicates". Claude Code sets `stop_hook_active=true` on every stop after any stop-hook continuation, including `asyncRewake` rewakes, which re-opened the 2026-07-21 blind window under the default one-shot behavior. The Claude mode waits up to `FM_CLAUDE_AUTOARM_SYNC_WAIT_MS` (default 800 milliseconds) and allows the stop when the watcher is healthy, the auto-arm's generation claim is open, or `state/.claude-autoarm-epoch` contains a fresh actionable rewake owned by this event epoch. The claim is the ledger entry itself: the epoch sequence in `state/.claude-autoarm-epoch` is a monotonic claim generation, line 1 records the claim and terminal outcome, and line 2 records the claiming process's mandatory pid-identity; `fm_autoarm_claim_open` and `fm_autoarm_claim_next` in `bin/fm-wake-lib.sh` own the format contract. @@ -114,8 +118,9 @@ When none of those proofs appears, it re-blocks up to `FM_CLAUDE_TURNEND_BLOCK_B In Claude mode, positive watcher recovery clears the block budget, failure notice, and attended alarm together under the existing budget lock before either hook reports ordinary recovery. The one loud attended fail-open is available only when the auto-arm has recorded an exhausted failure, its one notice is already consumed, the block budget is exhausted, and a final check finds neither a healthy watcher nor an automatic continuation. Each epoch identity is charged at most once per Stop under the budget lock, and a re-block against an epoch the auto-arm did not advance past the previous re-block is charged as well. -That second rule is what bounds an inert auto-arm: a hook kept silent by a session lock held by a live harness outside its ancestry, a hook that never fires, or a hook failing before its generation claim leaves the ledger frozen at its last outcome. -Charging only epoch changes let the count freeze with that ledger, so the guard re-blocked without limit and the attended fail-open was never reachable; `budget_account_current_epoch` in `bin/fm-turnend-guard.sh` owns the rule. +That second rule still bounds an inert auto-arm when a hook never fires or fails before its generation claim and therefore leaves the ledger frozen at its last outcome. +A verified live foreign session-lock owner takes the earlier diagnostic safe exit instead and never reaches this budget path. +Charging only epoch changes let the count freeze with that ledger, so the remaining inert-hook cases could re-block without limit and make the attended fail-open unreachable; `budget_account_current_epoch` in `bin/fm-turnend-guard.sh` owns the rule. Whenever both coordination locks are needed, positive auto-arm recovery and the terminal check acquire the auto-arm owner lock before the budget lock. After that alarm, the Stop auto-arm suppresses further exit-2 continuations until positive watcher recovery, so the final fail-open remains reachable. The alarm cannot repeat during that failure episode, and a later unhealthy stop blocks again. @@ -188,6 +193,7 @@ That warning uses `bin/fm-supervision-instructions.sh --repair-line`, so it alwa ## Regression coverage `tests/fm-turnend-guard.test.sh` covers the predicate, main and secondmate primary scope, child-worktree exclusion, `FM_HOME` and `FM_STATE_OVERRIDE` precedence, the live-lock and fresh-beacon guard predicate, the cooperative `--claude` open-generation claim wait, monotonic failed-epoch progression, bounded attended fail-open, the same bound against a ledger frozen by an inert auto-arm with and without a verified failure episode, post-alarm continuation suppression, positive recovery reset, generation and legacy claim cases that must block or clear instead of allowing a blind stop, away-mode daemon ownership between watcher cycles and over a watcher lock left behind by an exited watcher, plus its dead, pid-reused, absent, stale-beacon, and away-mode-off negatives, the away-mode beacon's poll-derived grace widening for a live daemon still mid-cycle and its bound against a dead daemon, a beacon older than that wider grace, and FM_POLL's inapplicability with away mode off, Pi logical-run latching, missing-`jq` behavior, all five primary registrations, Grok native and legacy selection, typed field precedence, malformed input, and exactly-one-path safety. +`tests/fm-turnend-foreign-owner-arm-fix.test.sh` runs the extracted isolated executable reproduction against real auto-arm and turn-end guard scripts, proving that a live foreign owner still prevents arming while repeated non-owner Stops receive a diagnostic and exit safely. `tests/fm-guard-stale-banner.test.sh` covers the pull-guard predicate, including the persistent-model fresh-leftover-beacon negative control; the auto-arm model's healthy fresh-beacon-without-a-watcher case, session-and-recovery-bound long-turn rewake tolerance, independently broken tolerance signals, open-claim negative control, stale-beacon alarm, and isolation from other models; and the extension model's live-watcher path, ownership-qualified fresh hand-off, held-lock failures, independently broken ownership signals, stale-beacon alarm, queued-wake warning, and Pi and pi-signed harness routing. It also covers true-reason banner wording and reason-keyed episode dedup surviving a beacon mtime change. `tests/fm-cursor-primary.test.sh` covers the Cursor park end to end over real processes with no harness installed: each tracked Claude-shaped entrypoint standing down on a Cursor payload, both follow-up sources, the bounded repair nag and its reset, the nested loop bounds, supersession, away-mode and lock-ownership inertness, Pi-host stand-down without Cursor identity and continued parking when `PI_CODING_AGENT` leaks alongside `CURSOR_AGENT` or `CURSOR_INVOKED_AS`, child-worktree exclusion, and that the adapter never exits 2. diff --git a/docs/watcher-continuity.md b/docs/watcher-continuity.md index fe5f24d53da..86f672c0d98 100644 --- a/docs/watcher-continuity.md +++ b/docs/watcher-continuity.md @@ -15,6 +15,7 @@ Cursor's `.cursor/hooks.json` `stop` hook (`bin/fm-turnend-guard-cursor.sh`) own Claude's `.claude/settings.json` Stop `asyncRewake` hook (`bin/fm-claude-stop-autoarm.sh`) owns routine tokenless re-arm. The hook fires on every Stop, and an eligible primary with supervision need admits one home-scoped owner that foregrounds `bin/fm-watch-arm.sh` inside the hook-owned process tree. A numeric session-lock owner that fails the shared `fm_harness_pid_alive` predicate is reclaimed through `bin/fm-lock.sh` before auto-arm state changes, while a live owner, absent lock, or malformed lock keeps the competing hook inert. +[`turnend-guard.md`](turnend-guard.md#guard-predicates) owns the Claude guard's behavior when that live owner is outside the current session's harness ancestry. The stale-owner claim occurs only after the existing AFK and supervision-need gates pass. After each non-actionable arm close, the hook rechecks the identity-matched watcher lock and fresh beacon before retrying a bounded number of times. A cycle-end failure is benign when that live-watcher predicate is true, and the hook suppresses the arm output and continues silently. diff --git a/tests/captures/no-mistakes-v1.70.1/README.md b/tests/captures/no-mistakes-v1.70.1/README.md new file mode 100644 index 00000000000..75cc474d955 --- /dev/null +++ b/tests/captures/no-mistakes-v1.70.1/README.md @@ -0,0 +1,51 @@ +# AXI run-state input captures + +These files own recorded serialized CLI inputs and a persisted run-inventory projection for the `test_captured_*` cases in `../../fm-crew-state.test.sh`. +They were captured on 2026-09-14 at 22:06 UTC with `no-mistakes version v1.70.1 (9c380d4) 2026-09-07T20:47:58Z`. +They are replay inputs, not evidence that every composed scenario was driven live. + +## Capture provenance + +Each status file is unchanged stdout from `no-mistakes axi status --run ` executed from the test-phase worktree, without entering the recorded run's checkout. +The command exited zero for all five run-specific captures. +`uninitialized.toon` is unchanged stdout from `no-mistakes axi status` in that worktree, which exited one. +Update notices on stderr are not part of the captured stdout contract. + +| File | Recorded run ID | Observed state | +| --- | --- | --- | +| `replacement.toon` | `01M2GAWMSDQK4B5EA9GZW35RXE` | Live CI step after a rerun | +| `superseded.toon` | `01M2FNFPK984YP0EHFTD1XEF8P` | Same-branch predecessor cancelled with `superseded by new push` | +| `parked.toon` | `01M20NDQH0G96AQYH1EHWGKT5F` | Separate branch parked at the test gate with one finding | +| `failed.toon` | `01M289YXN7V0513AKCF53MJ1BC` | Failed push step | +| `completed.toon` | `01M2FG7SEEP1VBZ3B5SX35BJ6Q` | Completed validation | + +`same-branch-inventory.json` preserves all nine rows for the replacement's branch, selected in a read-only transaction from the real `state.sqlite` database. +The projection is `id, repo_id, branch, status, head_sha, created_at`, ordered by `created_at DESC, id DESC`. +The source contained 78 runs for repository `acf4a767348a`; its `runs` schema confirmed the fixture's text identity/status/head fields and integer creation times. +No branch in that repository had two recorded live runs at capture time. + +`overview.toon` is the unchanged `count` and `runs` section emitted by the real `no-mistakes axi` executable against an isolated database copy of those 78 recorded runs. +Only the copy's repository `working_path` was relocated to the permitted worktree; no pipeline was initialized or controlled. +The copy omitted step data and had no daemon, so the unrelated active-run detail from that output is intentionally excluded. +The retained section demonstrates the actual ten-row cap, row order, quoting, and field layout. +Original stdout, source projections, and SHA-256 digests were retained in the test-phase evidence directory under `real-anchors/`. + +## Replay transformations and limits + +`captured_axi_status` substitutes only the run ID, branch, and head fields so the captures bind to disposable Git repositories. +Status, outcome, steps, findings, and gate bytes remain unchanged. +The inventory replay substitutes its disposable repository key and path, preserves every captured same-branch row, and hashes the database before and after the state read to detect writes. +Its ambiguity case explicitly changes one hidden cancelled row to running; this is a counterfactual, not a captured competing-live history. +The original review-gate, rebased-head, unrelated-metadata, and malformed-input assertions remain unchanged. + +| Required shape | Real anchor used | What remains unproven live | +| --- | --- | --- | +| Superseded cancellation yields to a parked replacement | Genuine cancellation/successor history plus separately captured parked-gate output | The captured successor was in CI, and the captured gate was at test on another branch; a same-rerun replacement parked specifically at review on an unfetched rebased head was not captured | +| Competing live identities beyond the cap and beside unrelated metadata | Real capped overview and complete nine-row branch history | No real branch had two live rows; changing a hidden row to running and injecting unrelated unusual metadata are controlled fixtures | +| Newer failure outranks an older live run | Genuine failed status and genuine live status | This relative ordering with both states on one branch was composed, not observed | +| Changing or unverifiable authority | Genuine live and cancelled status formats | The transition between reads, malformed records, wrong identities, and unreadable inventory are injected; no live race or corrupt production inventory was captured | +| Uninitialized repository preserves worker reporting | Actual uninitialized stdout; earlier live lifecycle-event/pane evidence | The portable test replays stdout and uses the existing pane fake | +| Development continues after completed validation | Genuine completed status | Advancing Git and emitting worker events after completion are disposable-repository actions, not an observed recorded worker sequence | +| Optional inventory dependencies are absent | Genuine gate output and capped inventory | Missing Python/SQLite and complete-inventory compositions are simulated; the captured host had both dependencies | + +Passing replay assertions establish behavior for these explicit inputs, not the absent live scenarios in the final column. diff --git a/tests/captures/no-mistakes-v1.70.1/completed.toon b/tests/captures/no-mistakes-v1.70.1/completed.toon new file mode 100644 index 00000000000..b2e0df67240 --- /dev/null +++ b/tests/captures/no-mistakes-v1.70.1/completed.toon @@ -0,0 +1,20 @@ +run: + id: "01M2FG7SEEP1VBZ3B5SX35BJ6Q" + branch: fm/fm-installed-timeout-watcher-will-not-stop + status: completed + head: 1129818e + head_sha: 1129818ef720b6827ae75eb690957c2c7393d82b + pr: "https://github.com/kunchenguid/firstmate/pull/4073" + findings: 1 awaiting + steps[9]{step,status,findings,duration_ms}: + intent,completed,0,20 + rebase,completed,0,1309 + review,completed,0,159353 + test,completed,0,2655669 + document,completed,0,119685 + lint,completed,0,6647 + push,completed,0,5691 + pr,completed,0,62239 + ci,completed,1,1845554 +outcome: passed-with-override +ci_override_reason: "live checks for https://github.com/kunchenguid/firstmate/pull/4073 not all passed: Lint (fail)" diff --git a/tests/captures/no-mistakes-v1.70.1/failed.toon b/tests/captures/no-mistakes-v1.70.1/failed.toon new file mode 100644 index 00000000000..8ef2a2a2520 --- /dev/null +++ b/tests/captures/no-mistakes-v1.70.1/failed.toon @@ -0,0 +1,19 @@ +run: + id: "01M289YXN7V0513AKCF53MJ1BC" + branch: fm/fm-bearings-board-loses-owner-state-and-links + status: failed + head: 9b76c588 + head_sha: 9b76c588dadf8d2e39405526fdbc41bbf8245513 + findings: none + steps[9]{step,status,findings,duration_ms}: + intent,completed,0,207 + rebase,skipped,0,0 + review,completed,0,840023 + test,completed,0,3218433 + document,skipped,0,0 + lint,completed,0,8278 + push,failed,0,6980 + pr,pending,0,0 + ci,pending,0,0 +outcome: failed +error: "step push failed: push to fork: git push https://github.com/mremond/firstmate --force-with-lease=refs/heads/fm/fm-bearings-board-loses-owner-state-and-links:723830dc438374c69661f37fdee6d9606ce744cf 9b76c588dadf8d2e39405526fdbc41bbf8245513:refs/heads/fm/fm-bearings-board-loses-owner-state-and-links: exit status 1: To https://github.com/mremond/firstmate\n ! [remote rejected] 9b76c588dadf8d2e39405526fdbc41bbf8245513 -> fm/fm-bearings-board-loses-owner-state-and-links (refusing to allow an OAuth App to create or update workflow `.github/workflows/ci.yml` without `workflow` scope)\nerror: failed to push some refs to 'https://github.com/mremond/firstmate'" diff --git a/tests/captures/no-mistakes-v1.70.1/overview.toon b/tests/captures/no-mistakes-v1.70.1/overview.toon new file mode 100644 index 00000000000..26839460f7e --- /dev/null +++ b/tests/captures/no-mistakes-v1.70.1/overview.toon @@ -0,0 +1,12 @@ +count: 10 of 78 total +runs[10]{id,branch,status,head,pr}: + "01M2GV0N1CJ7TGNYN3P76KK9TS",fm/fm-superseded-cancelled-run-outranks-live,running,7feb0272,"" + "01M2GV0HN9PMBPA7YNVGGC9FSS",fm/fm-spawn-tests-borrow-the-checkout,running,a0eb3010,"https://github.com/kunchenguid/firstmate/pull/4056" + "01M2GQTYBGKJGSFTQPP3F8NX0G",fm/fm-origin-credentials-printed-in-errors,running,4e5713f5,"https://github.com/kunchenguid/firstmate/pull/4138" + "01M2GB01EQ2MPCJ200TG71Y9SV",fm/fm-ci-refresh-portable-serial-hints,running,e6be9247,"https://github.com/kunchenguid/firstmate/pull/4144" + "01M2GAWMSDQK4B5EA9GZW35RXE",fm/fm-bearings-board-loses-owner-state-and-links,running,146a90ee,"https://github.com/kunchenguid/firstmate/pull/4019" + "01M2FR2HYYDP8NP8HDDX4D1MXS",fm/fm-absorb-fresh-replacement-wait,running,"71042535","https://github.com/kunchenguid/firstmate/pull/3605" + "01M2FNFPK984YP0EHFTD1XEF8P",fm/fm-bearings-board-loses-owner-state-and-links,cancelled,735a1fc5,"https://github.com/kunchenguid/firstmate/pull/4019" + "01M2FG7SEEP1VBZ3B5SX35BJ6Q",fm/fm-installed-timeout-watcher-will-not-stop,completed,1129818e,"https://github.com/kunchenguid/firstmate/pull/4073" + "01M289YXN7V0513AKCF53MJ1BC",fm/fm-bearings-board-loses-owner-state-and-links,failed,9b76c588,"" + "01M289X89E6CCFQ696MFF8G0MR",fm/fm-bearings-board-loses-owner-state-and-links,cancelled,d5825696,"" diff --git a/tests/captures/no-mistakes-v1.70.1/parked.toon b/tests/captures/no-mistakes-v1.70.1/parked.toon new file mode 100644 index 00000000000..4fb3ea76baf --- /dev/null +++ b/tests/captures/no-mistakes-v1.70.1/parked.toon @@ -0,0 +1,26 @@ +run: + id: "01M20NDQH0G96AQYH1EHWGKT5F" + branch: fm/fm-codex-hook-trust-dialog-undocumented + status: running + awaiting_agent: parked 2d14h + head: fb90db9f + head_sha: fb90db9f09db9b0cd1a83539959367c612659638 + pr: "https://github.com/kunchenguid/firstmate/pull/3998" + findings: 1 awaiting + steps[9]{step,status,findings,duration_ms}: + intent,completed,0,200 + rebase,completed,0,6521 + review,completed,0,180772 + test,awaiting_approval,1,90408 + document,pending,0,0 + lint,pending,0,0 + push,pending,0,0 + pr,pending,0,0 + ci,pending,0,0 +gate: + step: test + status: awaiting_approval + summary: Documentation-only change with no live test surface. Previously declined missing-evidence findings were not repeated. + findings[1]{id,severity,file,action,description}: + test-1,warning,"",ask-user,"this change has no live-validatable surface; proceed without live validation? (0 of 4 scenarios were driven live against the product); An operator reads the notes and encounters silent loss of supervision as the first warning.: Only documentation changed, with no executable surface. Demonstrating operator response would require a separate operator-driven evaluation.; An operator dismisses hook review with Escape without selecting review, declining hooks, or treating dismissal as approval.: No executable dialog handling changed. Live corroboration requires an operator-controlled isolated sessio… (truncated, 1236 chars total)" +help[2]: The explicitly selected gate for run 01M20NDQH0G96AQYH1EHWGKT5F is inspection-only; no run-scoped response command exists,Run `no-mistakes axi logs --run 01M20NDQH0G96AQYH1EHWGKT5F --step test --full` to read the full step log diff --git a/tests/captures/no-mistakes-v1.70.1/replacement.toon b/tests/captures/no-mistakes-v1.70.1/replacement.toon new file mode 100644 index 00000000000..99491b31a69 --- /dev/null +++ b/tests/captures/no-mistakes-v1.70.1/replacement.toon @@ -0,0 +1,20 @@ +run: + id: "01M2GAWMSDQK4B5EA9GZW35RXE" + branch: fm/fm-bearings-board-loses-owner-state-and-links + status: running + head: 146a90ee + head_sha: 146a90ee48c675ec7908faf141026efef5158c5a + pr: "https://github.com/kunchenguid/firstmate/pull/4019" + findings: 6 info + steps[9]{step,status,findings,duration_ms}: + intent,completed,0,129 + rebase,skipped,6,2090 + review,completed,0,2396206 + test,completed,0,1245156 + document,completed,0,226600 + lint,completed,0,15165 + push,completed,0,8511 + pr,completed,0,79292 + ci,running,0,0 + active_steps[1]{step,status,active_for,round_active_for,last_activity,agent_pid,round}: + ci,running,4h28m,4h28m,"quiet 2h58m ago: log: all CI checks passed - still monitoring until merged or closed","",starting diff --git a/tests/captures/no-mistakes-v1.70.1/same-branch-inventory.json b/tests/captures/no-mistakes-v1.70.1/same-branch-inventory.json new file mode 100644 index 00000000000..35b2f0c0e90 --- /dev/null +++ b/tests/captures/no-mistakes-v1.70.1/same-branch-inventory.json @@ -0,0 +1,74 @@ +[ + { + "id": "01M2GAWMSDQK4B5EA9GZW35RXE", + "repo_id": "acf4a767348a", + "branch": "fm/fm-bearings-board-loses-owner-state-and-links", + "status": "running", + "head_sha": "146a90ee48c675ec7908faf141026efef5158c5a", + "created_at": 1789402174 + }, + { + "id": "01M2FNFPK984YP0EHFTD1XEF8P", + "repo_id": "acf4a767348a", + "branch": "fm/fm-bearings-board-loses-owner-state-and-links", + "status": "cancelled", + "head_sha": "735a1fc5f8a02dbf2cab3851ad0d28ec85a177e3", + "created_at": 1789379730 + }, + { + "id": "01M289YXN7V0513AKCF53MJ1BC", + "repo_id": "acf4a767348a", + "branch": "fm/fm-bearings-board-loses-owner-state-and-links", + "status": "failed", + "head_sha": "9b76c588dadf8d2e39405526fdbc41bbf8245513", + "created_at": 1789132764 + }, + { + "id": "01M289X89E6CCFQ696MFF8G0MR", + "repo_id": "acf4a767348a", + "branch": "fm/fm-bearings-board-loses-owner-state-and-links", + "status": "cancelled", + "head_sha": "d582569662668edf727b74baefe100be32cf9e9b", + "created_at": 1789132710 + }, + { + "id": "01M21B2PY8J90QVVWWNDZ2WVGJ", + "repo_id": "acf4a767348a", + "branch": "fm/fm-bearings-board-loses-owner-state-and-links", + "status": "completed", + "head_sha": "723830dc438374c69661f37fdee6d9606ce744cf", + "created_at": 1788899056 + }, + { + "id": "01M21AXB75DVRYHRRWBC29GSN9", + "repo_id": "acf4a767348a", + "branch": "fm/fm-bearings-board-loses-owner-state-and-links", + "status": "cancelled", + "head_sha": "4996127f9d095686e4adb0063626753d68d1837f", + "created_at": 1788898880 + }, + { + "id": "01M20RM7ENZX5ZKSTYK5K3EFEG", + "repo_id": "acf4a767348a", + "branch": "fm/fm-bearings-board-loses-owner-state-and-links", + "status": "cancelled", + "head_sha": "d69beeb67488f764128a7e0ef04d583d404fedc2", + "created_at": 1788879707 + }, + { + "id": "01M20MQ02N69VJKXW9N8321SQW", + "repo_id": "acf4a767348a", + "branch": "fm/fm-bearings-board-loses-owner-state-and-links", + "status": "cancelled", + "head_sha": "1ecdbcb34f8045fe74ac3acc15d37677179ccc37", + "created_at": 1788875604 + }, + { + "id": "01M20HPP5XR7K31VHP0DC6S5QA", + "repo_id": "acf4a767348a", + "branch": "fm/fm-bearings-board-loses-owner-state-and-links", + "status": "failed", + "head_sha": "1ecdbcb34f8045fe74ac3acc15d37677179ccc37", + "created_at": 1788872448 + } +] diff --git a/tests/captures/no-mistakes-v1.70.1/superseded.toon b/tests/captures/no-mistakes-v1.70.1/superseded.toon new file mode 100644 index 00000000000..8291fcf07fd --- /dev/null +++ b/tests/captures/no-mistakes-v1.70.1/superseded.toon @@ -0,0 +1,20 @@ +run: + id: "01M2FNFPK984YP0EHFTD1XEF8P" + branch: fm/fm-bearings-board-loses-owner-state-and-links + status: cancelled + head: 735a1fc5 + head_sha: 735a1fc5f8a02dbf2cab3851ad0d28ec85a177e3 + pr: "https://github.com/kunchenguid/firstmate/pull/4019" + findings: "1 awaiting, 2 auto-fix" + steps[9]{step,status,findings,duration_ms}: + intent,completed,0,18 + rebase,completed,0,1205 + review,completed,3,321482 + test,completed,0,906557 + document,completed,0,242319 + lint,completed,0,7346 + push,completed,0,8523 + pr,completed,0,195015 + ci,failed,0,20551480 +outcome: cancelled +error: "cancelled: superseded by new push" diff --git a/tests/captures/no-mistakes-v1.70.1/uninitialized.toon b/tests/captures/no-mistakes-v1.70.1/uninitialized.toon new file mode 100644 index 00000000000..c124abfc66d --- /dev/null +++ b/tests/captures/no-mistakes-v1.70.1/uninitialized.toon @@ -0,0 +1,2 @@ +error: repo not initialized (run 'no-mistakes init' first) +help[1]: Run `no-mistakes init` to set up the gate in this repository diff --git a/tests/fm-calm-claude-mod.test.sh b/tests/fm-calm-claude-mod.test.sh index 7b3898d10ec..c5fa0715d9b 100644 --- a/tests/fm-calm-claude-mod.test.sh +++ b/tests/fm-calm-claude-mod.test.sh @@ -234,6 +234,7 @@ test_presentation_policy() { cat >"$TMP_ROOT/policy.mjs" < { if (!condition) throw new Error(message); }; const plugin = "/repo/.claude/mods/firstmate-calm"; check(policy.calmPreferencePath({}, plugin) === "/repo/config/calm", "plugin-root fallback"); @@ -252,7 +253,18 @@ const shortNote = "Checking briefly."; const multiLineReply = "The result is substantive.\\nHere is the context needed to continue."; const atThresholdReply = "x".repeat(240); const belowThresholdNote = "x".repeat(239); -check(policy.CALM_PRESERVE_MIN_CHARS === 240, "preservation threshold"); +check(policy.CALM_PRESERVE_MIN_CHARS === 240, "Claude preservation threshold"); +check(piPreservation.CALM_PRESERVE_MIN_CHARS === policy.CALM_PRESERVE_MIN_CHARS, "Pi and Claude preservation thresholds"); +for (const [text, expectedPreserved, label] of [ + [belowThresholdNote, false, "239-character single line"], + [atThresholdReply, true, "240-character single line"], + [multiLineReply, true, "multi-line text"], +]) { + const claudePreserved = !policy.stepTextIsWorkingNote({ stopReason: "tool_use", toolUses: [] }, text); + const piPreserved = piPreservation.calmTextIsSubstantive(text); + check(claudePreserved === expectedPreserved, "Claude did not classify " + label + " as expected"); + check(piPreserved === expectedPreserved, "Pi did not classify " + label + " as expected"); +} check(policy.stepTextIsWorkingNote({ stopReason: "tool_use", toolUses: [] }, shortNote) === true, "short single-line tool_use note"); check(policy.stepTextIsWorkingNote({ stopReason: "tool_use", toolUses: [] }, multiLineReply) === false, "multi-line tool_use reply"); check(policy.stepTextIsWorkingNote({ stopReason: "tool_use", toolUses: [] }, atThresholdReply) === false, "threshold-length tool_use reply"); @@ -296,7 +308,7 @@ console.log("policy-ok"); JS out=$(run_node "$TMP_ROOT/policy.mjs" 2>&1) || fail "presentation policy: $out" assert_contains "$out" "policy-ok" "the policy check did not complete" - pass "the Calm policy resolves the shared preference exactly as Pi does, reads on, max, and off as Pi does, and classifies working notes by stop reason, tool use, and restored transcript shape" + pass "the Calm policy resolves the shared preference exactly as Pi does, reads on, max, and off as Pi does, and shares Pi's 240-character-or-newline preservation behavior while classifying working notes by stop reason, tool use, and restored transcript shape" } # The classifier parity corpus: envelopes the shell owner encodes itself, its legacy diff --git a/tests/fm-calm-pi-extension.test.sh b/tests/fm-calm-pi-extension.test.sh index 156136b01fe..02cee20e6e3 100755 --- a/tests/fm-calm-pi-extension.test.sh +++ b/tests/fm-calm-pi-extension.test.sh @@ -8,6 +8,7 @@ set -u TMP_ROOT=$(fm_test_tmproot fm-calm-pi-extension) EXT="$ROOT/.pi/extensions/fm-calm.ts" ASSISTANT_LAYOUT="$ROOT/.pi/extensions/lib/fm-calm-assistant-layout.ts" +PRESERVATION="$ROOT/.pi/extensions/lib/fm-calm-preservation.ts" OPERATIONAL_USER_LAYOUT="$ROOT/.pi/extensions/lib/fm-calm-operational-user-layout.ts" VISIBILITY="$ROOT/.pi/extensions/lib/fm-calm-visibility.ts" WORKING_SHIP="$ROOT/.pi/extensions/lib/fm-calm-working-ship.ts" @@ -169,6 +170,7 @@ test_home_resolution() { "$fixture/launch-cwd" cp "$EXT" "$fixture/project/.pi/extensions/fm-calm.ts" cp "$ASSISTANT_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-assistant-layout.ts" + cp "$PRESERVATION" "$fixture/project/.pi/extensions/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" cp "$VISIBILITY" "$fixture/project/.pi/extensions/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$fixture/project/.pi/extensions/lib/fm-calm-working-ship.ts" @@ -292,6 +294,7 @@ test_pi_compat_degraded_adapter() { "$fixture/project/node_modules/@earendil-works" cp "$EXT" "$fixture/project/.pi/extensions/fm-calm.ts" cp "$ASSISTANT_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-assistant-layout.ts" + cp "$PRESERVATION" "$fixture/project/.pi/extensions/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" cp "$VISIBILITY" "$fixture/project/.pi/extensions/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$fixture/project/.pi/extensions/lib/fm-calm-working-ship.ts" @@ -392,6 +395,7 @@ test_pi_compat_missing_adapter_exports() { "$fixture/project/.pi/extensions/lib" \ "$fixture/project/node_modules/@earendil-works/pi-coding-agent" cp "$ASSISTANT_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-assistant-layout.ts" + cp "$PRESERVATION" "$fixture/project/.pi/extensions/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" cp "$VISIBILITY" "$fixture/project/.pi/extensions/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$fixture/project/.pi/extensions/lib/fm-calm-working-ship.ts" @@ -453,6 +457,7 @@ test_builtin_gate_load_time() { "$fixture/home-on/config" cp "$EXT" "$fixture/project/.pi/extensions/fm-calm.ts" cp "$ASSISTANT_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-assistant-layout.ts" + cp "$PRESERVATION" "$fixture/project/.pi/extensions/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" cp "$VISIBILITY" "$fixture/project/.pi/extensions/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$fixture/project/.pi/extensions/lib/fm-calm-working-ship.ts" @@ -540,6 +545,7 @@ test_calm_activation_collision_and_regression_bound() { "$fixture/home/config" cp "$EXT" "$fixture/project/.pi/extensions/fm-calm.ts" cp "$ASSISTANT_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-assistant-layout.ts" + cp "$PRESERVATION" "$fixture/project/.pi/extensions/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" cp "$VISIBILITY" "$fixture/project/.pi/extensions/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$fixture/project/.pi/extensions/lib/fm-calm-working-ship.ts" @@ -755,6 +761,7 @@ test_rendering_and_session_lifecycle() { mkdir -p "$fixture/home" "$fixture/lib" "$fixture/node_modules/@earendil-works" cp "$EXT" "$fixture/fm-calm.ts" cp "$ASSISTANT_LAYOUT" "$fixture/lib/fm-calm-assistant-layout.ts" + cp "$PRESERVATION" "$fixture/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$fixture/lib/fm-calm-operational-user-layout.ts" cp "$VISIBILITY" "$fixture/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$fixture/lib/fm-calm-working-ship.ts" @@ -1473,6 +1480,7 @@ test_calm_mid_turn_working_notes() { mkdir -p "$fixture/home" "$fixture/lib" "$fixture/node_modules/@earendil-works" cp "$EXT" "$fixture/fm-calm.ts" cp "$ASSISTANT_LAYOUT" "$fixture/lib/fm-calm-assistant-layout.ts" + cp "$PRESERVATION" "$fixture/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$fixture/lib/fm-calm-operational-user-layout.ts" cp "$VISIBILITY" "$fixture/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$fixture/lib/fm-calm-working-ship.ts" @@ -1501,6 +1509,7 @@ setCapabilities({ images: null, trueColor: true, hyperlinks: false }); // the same module URLs, so they share one live visibility policy exactly the way a // single Pi process does. const visibility = await import(pathToFileURL(`${process.cwd()}/lib/fm-calm-visibility.ts`).href); +const preservation = await import(pathToFileURL(`${process.cwd()}/lib/fm-calm-preservation.ts`).href); const calmPreferencePath = `${process.env.FM_HOME}/config/calm`; const components = []; const ui = { @@ -1562,6 +1571,13 @@ const assistantBase = { timestamp: 1, }; const toolCall = { type: "toolCall", id: "calm-mid-turn-tool", name: "read", arguments: { path: "sample.txt" } }; +const substantiveLongText = "SUBSTANTIVE_LONG_MIDTURN_REPORT " + "context ".repeat(35); +const substantiveMultilineText = "SUBSTANTIVE_MIDTURN_REPORT\nAdditional context needed to continue."; +const belowThresholdText = "b".repeat(preservation.CALM_PRESERVE_MIN_CHARS - 1); +const atThresholdText = "t".repeat(preservation.CALM_PRESERVE_MIN_CHARS); +if (preservation.CALM_PRESERVE_MIN_CHARS !== 240) { + throw new Error(`Pi Calm preservation threshold changed to ${preservation.CALM_PRESERVE_MIN_CHARS}`); +} const messages = { // The reported incident: narration emitted in the same assistant message as a tool call. midTurn: { @@ -1569,6 +1585,36 @@ const messages = { stopReason: "toolUse", content: [{ type: "text", text: "MIDTURN_WORKING_NOTE" }, toolCall], }, + // Substantive mid-turn content must remain visible even when the message also calls a tool. + substantiveLong: { + ...assistantBase, + stopReason: "toolUse", + content: [{ type: "text", text: substantiveLongText }, toolCall], + }, + substantiveMultiline: { + ...assistantBase, + stopReason: "toolUse", + content: [{ type: "text", text: substantiveMultilineText }, toolCall], + }, + belowThreshold: { + ...assistantBase, + stopReason: "toolUse", + content: [{ type: "text", text: belowThresholdText }, toolCall], + }, + atThreshold: { + ...assistantBase, + stopReason: "toolUse", + content: [{ type: "text", text: atThresholdText }, toolCall], + }, + mixedBlocks: { + ...assistantBase, + stopReason: "toolUse", + content: [ + { type: "text", text: "MIXED_SHORT_WORKING_NOTE" }, + { type: "text", text: substantiveLongText }, + toolCall, + ], + }, // The genuine reply that ends a response, which Calm never hides. finalReply: { ...assistantBase, @@ -1634,8 +1680,14 @@ if (readFileSync(calmPreferencePath, "utf8") !== "on\n") { throw new Error("plain /calm from off did not persist on"); } if (rendered("midTurn").length !== 0) { - throw new Error(`Calm on left mid-turn working-note rows: ${JSON.stringify(rendered("midTurn"))}`); -} + throw new Error(`Calm on left short mid-turn working-note rows: ${JSON.stringify(rendered("midTurn"))}`); +} +requireVisible("substantiveLong", "SUBSTANTIVE_LONG_MIDTURN_REPORT", "Calm on"); +requireVisible("substantiveMultiline", "SUBSTANTIVE_MIDTURN_REPORT", "Calm on"); +requireHidden("belowThreshold", belowThresholdText.slice(0, 32), "Calm on"); +requireVisible("atThreshold", atThresholdText.slice(0, 32), "Calm on"); +requireHidden("mixedBlocks", "MIXED_SHORT_WORKING_NOTE", "Calm on"); +requireVisible("mixedBlocks", "SUBSTANTIVE_LONG_MIDTURN_REPORT", "Calm on"); requireHidden("truncatedMidTurn", "TRUNCATED_MIDTURN_NOTE", "Calm on"); // Pi owns the wording of its truncation notice; Calm must leave that row's own notice // standing rather than collapsing an incomplete response to nothing. @@ -1734,6 +1786,7 @@ test_operational_followup_turn_e2e() { fm_git_init_commit "$project" cp "$EXT" "$project/.pi/extensions/fm-calm.ts" cp "$ASSISTANT_LAYOUT" "$project/.pi/extensions/lib/fm-calm-assistant-layout.ts" + cp "$PRESERVATION" "$project/.pi/extensions/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" cp "$VISIBILITY" "$project/.pi/extensions/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$project/.pi/extensions/lib/fm-calm-working-ship.ts" @@ -2109,6 +2162,7 @@ test_hidden_block_geometry_e2e() { fm_git_init_commit "$project" cp "$EXT" "$project/.pi/extensions/fm-calm.ts" cp "$ASSISTANT_LAYOUT" "$project/.pi/extensions/lib/fm-calm-assistant-layout.ts" + cp "$PRESERVATION" "$project/.pi/extensions/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" cp "$VISIBILITY" "$project/.pi/extensions/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$project/.pi/extensions/lib/fm-calm-working-ship.ts" @@ -2344,6 +2398,7 @@ test_working_ship_geometry_and_lifecycle() { mkdir -p "$fixture/home" "$fixture/lib" "$fixture/node_modules/@earendil-works" cp "$EXT" "$fixture/fm-calm.ts" cp "$ASSISTANT_LAYOUT" "$fixture/lib/fm-calm-assistant-layout.ts" + cp "$PRESERVATION" "$fixture/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$fixture/lib/fm-calm-operational-user-layout.ts" cp "$VISIBILITY" "$fixture/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$fixture/lib/fm-calm-working-ship.ts" @@ -3374,6 +3429,7 @@ test_interactive_terminal_e2e() { : > "$project/AGENTS.md" cp "$EXT" "$project/.pi/extensions/fm-calm.ts" cp "$ASSISTANT_LAYOUT" "$project/.pi/extensions/lib/fm-calm-assistant-layout.ts" + cp "$PRESERVATION" "$project/.pi/extensions/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" cp "$VISIBILITY" "$project/.pi/extensions/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$project/.pi/extensions/lib/fm-calm-working-ship.ts" diff --git a/tests/fm-crew-state.test.sh b/tests/fm-crew-state.test.sh index 2f3faf3ca3f..a51574f0f14 100755 --- a/tests/fm-crew-state.test.sh +++ b/tests/fm-crew-state.test.sh @@ -21,10 +21,9 @@ # (d2) terminal failed run whose only failure is an orphaned ci monitor # after checks read green -> done # (e) cross-branch attribution: this branch's own run found via list lookup -# (e2) several runs bound to one worktree: the live one outranks the corpse -# (an unclassifiable status word keeps the ledger's newest-first order) -# (e3) the live sibling's head was never fetched into the task copy: it still -# outranks a terminal row sitting at the worktree's exact commit +# (e2) multiple runs: creation order preserves newer failures, replacement +# gates retain their run identity, and competing live runs read unknown +# (e3) an older live sibling with an unfetched head cannot hide a newer failure # (f) no run + semantic busy -> pane # (g) no run + semantic idle falls to the status-log verb -> status-log # (h) dead pane: no run -> unknown/none; with a run -> run-step (not the shell) @@ -68,7 +67,8 @@ make_repo_on_branch() { # # A fakebin with a fake `no-mistakes` (serves the env-driven run output) and a # fake `tmux` (serves a busy or idle pane). The fake no-mistakes mirrors the real -# command surface the helper uses: `axi status`, `axi status --run ` (the +# command surface the helper uses: `axi` (the identity overview), `axi status`, +# and `axi status --run ` (the # `axi` surface - no runs-listing subcommand exists under it, verified against # the real CLI), and the actual top-level run-listing command, `no-mistakes # runs --limit N`, which is plain text - no run id, no quoting - serving @@ -82,11 +82,20 @@ set -u case "${1:-}" in axi) shift + if [ "$#" = 0 ]; then + printf '%s\n' "${FM_FAKE_AXI_HOME:-${FM_FAKE_AXI_STATUS:-}}" + exit "${FM_FAKE_AXI_HOME_ERROR:-0}" + fi case "${1:-}" in status) shift - if [ "${1:-}" = --run ]; then printf '%s\n' "${FM_FAKE_AXI_STATUS_RUN:-}" - else printf '%s\n' "${FM_FAKE_AXI_STATUS:-}"; fi ;; + if [ "${1:-}" = --run ]; then + printf '%s\n' "${FM_FAKE_AXI_STATUS_RUN:-}" + exit "${FM_FAKE_AXI_STATUS_RUN_ERROR:-0}" + else + printf '%s\n' "${FM_FAKE_AXI_STATUS:-}" + exit "${FM_FAKE_AXI_STATUS_ERROR:-0}" + fi ;; logs) printf '%s\n' "${FM_FAKE_CI_LOGS:-}" ;; esac @@ -267,7 +276,13 @@ arm_idle_record() { # # assignments below stay exported into the fakes without an `export VAR=$(...)` # command-substitution assignment (SC2155). reset_fakes() { + NM_HOME="$TMP_ROOT/no-mistakes-unused" + export NM_HOME FM_FAKE_AXI_STATUS="" + FM_FAKE_AXI_STATUS_ERROR=0 + FM_FAKE_AXI_HOME="" + FM_FAKE_AXI_HOME_ERROR=0 + FM_FAKE_AXI_STATUS_RUN_ERROR=0 FM_FAKE_AXI_STATUS_RUN="" FM_FAKE_RUNS_LIST="" FM_FAKE_BUSY=0 @@ -294,7 +309,8 @@ reset_fakes() { unset FM_FAKE_PR_47_STATE FM_FAKE_PR_47_MERGED FM_FAKE_PR_48_STATE FM_FAKE_PR_48_MERGED export FM_FAKE_AXI_STATUS FM_FAKE_AXI_STATUS_RUN FM_FAKE_RUNS_LIST FM_FAKE_BUSY FM_FAKE_BUSY_TEXT FM_FAKE_TMUX_MISSING FM_FAKE_TMUX_UNREADABLE export FM_FAKE_HERDR_BUSY FM_FAKE_HERDR_MISSING FM_FAKE_HERDR_READ_FAIL FM_FAKE_HERDR_HUSK FM_FAKE_HERDR_AGENT_STATUS FM_FAKE_HERDR_PROCESS FM_FAKE_HERDR_SHELL_PID FM_FAKE_CI_LOGS - export FM_FAKE_DAEMON_DOWN + export FM_FAKE_DAEMON_DOWN FM_FAKE_AXI_HOME + export FM_FAKE_AXI_HOME_ERROR FM_FAKE_AXI_STATUS_RUN_ERROR FM_FAKE_AXI_STATUS_ERROR export FM_FAKE_PR_STATE FM_FAKE_PR_MERGED FM_FAKE_PR_READ_FAIL FM_FAKE_PR_READ_LOG FM_FAKE_PR_STATE_AXI export FM_FAKE_GLAB_STATE FM_FAKE_GLAB_READ_FAIL FM_FAKE_GLAB_READ_LOG export FM_FAKE_PR_47_STATE FM_FAKE_PR_47_MERGED FM_FAKE_PR_48_STATE FM_FAKE_PR_48_MERGED @@ -1440,13 +1456,10 @@ EOF pass "cross-branch attribution picks the branch's most recent row" } -# Live-over-terminal selection (bin/fm-nm-run-lib.sh). Reproduces the proven -# 2026-08 case: a crashed validation daemon left a FAILED run at the worktree's -# exact commit, while the live run that replaced it validates a descendant -# commit on the same branch. Both bind - the corpse by the equal-commit rule, -# the live run by the ancestor rule - and bare `axi status` answers with the -# corpse, so every recomputation read a healthy task as failed. -test_terminal_corpse_loses_to_live_run_on_same_branch() { +# The plain ledger is ordered by creation time, not the time a status changed. +# A newer failure must not be hidden by an older live run, even when both heads +# bind to the worktree. These legacy CLI cases lack the AXI identity table. +test_terminal_run_keeps_newer_failure_over_live_sibling() { reset_fakes local d base_head live_head short_base short_live out d=$(new_case live-beats-corpse) @@ -1461,28 +1474,25 @@ test_terminal_corpse_loses_to_live_run_on_same_branch() { [ "$short_base" != "$short_live" ] || fail "live run head did not advance past the worktree" make_fakebin "$d" >/dev/null fm_write_meta "$d/state/corpse.meta" "window=fm:fm-corpse" "worktree=$d/wt" "kind=ship" - # The corpse is the most-recently-touched run, so it is what `axi status` - # reports, at this worktree's own commit. + # The newest run failed at this worktree's own commit. FM_FAKE_RUN_HEAD="$base_head" FM_FAKE_AXI_STATUS="$(run_failed fm/feat-corpse)" - # It is also the newest row in the listing (the crash marked it after the - # live run started), so row order alone still selects the corpse. + # The older live run may have advanced its tip, but it did not replace this run. FM_FAKE_RUNS_LIST="$(cat <