diff --git a/.claude/mods/firstmate-calm/hooks/register.ts b/.claude/mods/firstmate-calm/hooks/register.ts index 3e7960e23cc..558b28f851e 100644 --- a/.claude/mods/firstmate-calm/hooks/register.ts +++ b/.claude/mods/firstmate-calm/hooks/register.ts @@ -15,7 +15,7 @@ // glue under `claude plugin test`. Nothing here rewrites a message: `ui.render` changes // drawings and leaves the stored transcript, model context, and session storage alone. // -// Presentation while Calm is on, matching Pi Calm's policy where the mods API allows: +// Presentation while Calm is on, sharing Pi Calm's goals where the mods API allows: // the stock working row (`Spinner`) becomes the two-row sailboat, repainted through // `$.ui.blit` on the sprite's own tick; `ToolUse`, `ToolResult`, and `ToolGroup` rows // draw as zero-height boxes; a `UserMessage` whose text the canonical operational-input @@ -45,7 +45,7 @@ import { import { calmPreferencePath, parseCalmPreference, - restoredAssistantText, + classifyRestoredTranscript, serializeCalmPreference, stepTextIsWorkingNote, userTextIsOperational, @@ -109,7 +109,7 @@ async function load($: EngineInterface): Promise { calm = parseCalmPreference(await readPreference($, preferencePath)); palette = CALM_SHIP_RASTER_PALETTES[calmShipPaletteFamily(await readTheme($))]; try { - const restored = restoredAssistantText(await $.session.messages()); + const restored = classifyRestoredTranscript(await $.session.messages()); for (const note of restored.workingNotes) workingNotes.add(note); for (const reply of restored.finalReplies) finalReplies.add(reply); } catch { @@ -229,17 +229,14 @@ export const register: Register = (on) => { const result = await stream.result; if (e.agentId === undefined) { let changed = false; - if (stepTextIsWorkingNote(result)) { - for (const text of [...blocks.values(), result.answer]) { - const key = workingNoteKey(text); - if (key === "" || finalReplies.has(key) || workingNotes.has(key)) continue; + for (const text of [...blocks.values(), result.answer]) { + const key = workingNoteKey(text); + if (key === "") continue; + if (stepTextIsWorkingNote(result, text)) { + if (finalReplies.has(key) || workingNotes.has(key)) continue; workingNotes.add(key); changed = true; - } - } else { - for (const text of [...blocks.values(), result.answer]) { - const key = workingNoteKey(text); - if (key === "") continue; + } else { if (!finalReplies.has(key)) { finalReplies.add(key); changed = true; diff --git a/.claude/mods/firstmate-calm/lib/fm-calm-presentation.ts b/.claude/mods/firstmate-calm/lib/fm-calm-presentation.ts index c07b37ba1b7..c8a4e556a79 100644 --- a/.claude/mods/firstmate-calm/lib/fm-calm-presentation.ts +++ b/.claude/mods/firstmate-calm/lib/fm-calm-presentation.ts @@ -2,11 +2,11 @@ // // This module owns the decisions ../hooks/register.ts applies through `$`: where the // shared per-home Calm preference lives and how its value reads, which assistant text is -// a mid-turn working note, and which transcript rows Calm hides. It mirrors the Pi -// policy in .pi/extensions/lib/fm-calm-visibility.ts and .pi/extensions/fm-calm.ts: -// genuine user prompts, genuine agent responses, and working activity stay visible; -// tool rows, tool groups, working notes, and canonically classified operational user -// rows hide. docs/calm.md owns the captain-facing contract and docs/configuration.md +// a mid-turn working note, and which transcript rows Calm hides. It shares Pi Calm's +// broad presentation boundary: genuine user prompts, genuine agent responses, and +// working activity stay visible; tool rows, tool groups, classified working notes, and +// canonically classified operational user rows hide. docs/calm.md owns the exact +// captain-facing contract and docs/configuration.md // the persisted preference schema. Everything here is pure so tests run it under Node. import { classifyFirstmateOperationalText } from "./fm-operational-input.ts"; @@ -68,18 +68,34 @@ export type CalmStepOutcome = { }; /** - * Whether the text of a model step is a mid-turn working note: the model did not end + * Single-line narration in session history topped out around 215 characters, while + * substantive single-line content began around 270; every multi-line message was + * substantive, so this empirical boundary stays deliberately tunable. + */ +export const CALM_PRESERVE_MIN_CHARS = 240; + +/** Whether text is substantive enough to preserve despite ending alongside a tool call. */ +function shouldPreserveMidTurnText(text: string): boolean { + const trimmedText = text.trim(); + return text.includes("\n") || trimmedText.length >= CALM_PRESERVE_MIN_CHARS; +} + +/** + * Whether text from a model step is a mid-turn working note: the model did not end * its response there, because it stopped to call tools, or ran out of tokens while - * calling them. The same rule as Pi Calm's `assistant-working-note` class. + * calling them. Short single-line narration stays a note; substantive text is a final + * reply even when the step also called tools. */ -export function stepTextIsWorkingNote(step: CalmStepOutcome): boolean { - if (step.stopReason === "tool_use") return true; - return step.stopReason === "max_tokens" && step.toolUses.length > 0; +export function stepTextIsWorkingNote(step: CalmStepOutcome, text: string): boolean { + const midTurn = step.stopReason === "tool_use" || (step.stopReason === "max_tokens" && step.toolUses.length > 0); + return midTurn && !shouldPreserveMidTurnText(text); } -/** The key a working note is remembered under: its trimmed text; empty text is no note. */ +/** A trimmed text key that retains whether the raw row contained a newline. */ export function workingNoteKey(text: string): string { - return text.trim(); + const trimmedText = text.trim(); + if (trimmedText === "") return ""; + return text.includes("\n") ? `${trimmedText}\n` : trimmedText; } /** The shape of one `$.session.messages()` row this policy reads. */ @@ -93,9 +109,10 @@ export type CalmSessionRow = { * The structurally identified working notes and final replies in a restored transcript. * The stored transcript keeps each content block as its own row, so assistant text is a * working note when its own row called tools, or when a tool-calling assistant row - * follows it before the next user row. + * follows it before the next user row. Substantive text in either position is preserved + * as a final reply, matching the live classifier. */ -export function restoredAssistantText(rows: readonly CalmSessionRow[]): { +export function classifyRestoredTranscript(rows: readonly CalmSessionRow[]): { workingNotes: string[]; finalReplies: string[]; } { @@ -113,7 +130,8 @@ export function restoredAssistantText(rows: readonly CalmSessionRow[]): { break; } } - if (followedByToolCall) notes.add(key); + if (followedByToolCall && shouldPreserveMidTurnText(row.text)) finalReplies.add(key); + else if (followedByToolCall) notes.add(key); else finalReplies.add(key); } for (const key of finalReplies) notes.delete(key); diff --git a/.claude/mods/firstmate-calm/tests/calm.test.ts b/.claude/mods/firstmate-calm/tests/calm.test.ts index 46892f51013..7babd94d8cc 100644 --- a/.claude/mods/firstmate-calm/tests/calm.test.ts +++ b/.claude/mods/firstmate-calm/tests/calm.test.ts @@ -240,7 +240,7 @@ describe("mid-turn working notes", () => { return { seen, result: step.value as { answer: string; stopReason: string | null } }; } - test("hides the text blocks of a step that stopped to call tools, and forwards the stream untouched", async ($, on) => { + test("hides brief narration but preserves substantive text before tool calls, and forwards the stream untouched", async ($, on) => { const { journal } = world(on, { preference: "on\n" }); const set = stepper(on); set({ @@ -248,7 +248,7 @@ describe("mid-turn working notes", () => { { kind: "text", index: 0, text: "Let me " }, { kind: "text", index: 0, text: "look first." }, { kind: "tool", index: 1, id: "t1", name: "Bash" }, - { kind: "text", index: 2, text: "Then I read it." }, + { kind: "text", index: 2, text: "Then I read it.\n" }, { kind: "stop", stopReason: "tool_use", usage: null }, ], result: { answer: "Let me look first.\nThen I read it.", toolUses: [{ name: "Bash", input: {} }], stopReason: "tool_use" }, @@ -258,10 +258,10 @@ describe("mid-turn working notes", () => { expect(seen).toHaveLength(5); expect(result.answer).toBe("Let me look first.\nThen I read it."); expect(journal.invalidations).toContain("ui.render"); - expect(isHidden(await $.ui.render(assistantMessage("Let me look first.")))).toBe(true); - expect(isHidden(await $.ui.render(assistantMessage("Then I read it.\n")))).toBe(true); - expect(isHidden(await $.ui.render(assistantMessage("Let me look first.\nThen I read it.")))).toBe(true); - expect(isStock(await $.ui.render(assistantMessage("Something else")))).toBe(true); + expect(isHidden(await $.ui.render(assistantMessage("Let me look first."))), "brief narration").toBe(true); + expect(isStock(await $.ui.render(assistantMessage("Then I read it.\n"))), "multi-line block").toBe(true); + expect(isStock(await $.ui.render(assistantMessage("Let me look first.\nThen I read it."))), "complete answer").toBe(true); + expect(isStock(await $.ui.render(assistantMessage("Something else"))), "unrelated text").toBe(true); }); test("keeps a final reply visible when its text matches an earlier working note", async ($, on) => { @@ -379,6 +379,37 @@ describe("mid-turn working notes", () => { expect(isHidden(await $.ui.render(assistantMessage("Checking.")))).toBe(true); }); + test("preserves substantive mid-turn text restored from the transcript", async ($, on) => { + const multiLine = "The result is substantive.\nHere is the context needed to continue."; + const atThreshold = "x".repeat(240); + const belowThreshold = "x".repeat(239); + world(on, { + preference: "on\n", + messages: [ + { role: "user", text: "multi-line", toolUses: [] }, + { role: "assistant", text: multiLine, toolUses: [] }, + { role: "assistant", text: "", toolUses: [{}] }, + { role: "user", text: "at threshold", toolUses: [] }, + { role: "assistant", text: atThreshold, toolUses: [] }, + { role: "assistant", text: "", toolUses: [{}] }, + { role: "user", text: "below threshold", toolUses: [] }, + { role: "assistant", text: belowThreshold, toolUses: [] }, + { role: "assistant", text: "", toolUses: [{}] }, + { role: "user", text: "newline collision", toolUses: [] }, + { role: "assistant", text: "Checking.\n", toolUses: [] }, + { role: "assistant", text: "", toolUses: [{}] }, + { role: "user", text: "single-line collision", toolUses: [] }, + { role: "assistant", text: "Checking.", toolUses: [] }, + { role: "assistant", text: "", toolUses: [{}] }, + ], + }); + expect(isStock(await $.ui.render(assistantMessage(multiLine)))).toBe(true); + expect(isStock(await $.ui.render(assistantMessage(atThreshold)))).toBe(true); + expect(isHidden(await $.ui.render(assistantMessage(belowThreshold)))).toBe(true); + expect(isStock(await $.ui.render(assistantMessage("Checking.\n")))).toBe(true); + expect(isHidden(await $.ui.render(assistantMessage("Checking.")))).toBe(true); + }); + test("seeds notes from a restored transcript without hiding a colliding final reply", async ($, on) => { world(on, { preference: "on\n", diff --git a/docs/calm-mode-feasibility.md b/docs/calm-mode-feasibility.md index 787c906d822..f67f8c12dc5 100644 --- a/docs/calm-mode-feasibility.md +++ b/docs/calm-mode-feasibility.md @@ -290,7 +290,7 @@ It asserts one persisted and rendered captain answer, exact user-role operationa Quoted current markers, ASCII-only labels, ordinary text before a marker, unrelated U+2063 placement, and image-bearing input remain visible in component and native transcript checks. `tests/fm-pi-primary-live-e2e.test.sh` also proves the working ship replaces the built-in `Working...` row while Calm is active on the credentialed provider path, and that it clears when the run settles, before continuing its ordinary watcher lifecycle. `tests/fm-pi-primary-types.test.sh` performs strict no-emit TypeScript checking against whichever Pi declarations are installed, without pinning a version of its own. -`tests/fm-calm-claude-mod.test.sh` needs no Claude Code binary: it proves the mod is one hooks module with no command, skill, agent, or classic hook path around its opt-in, that Pi's working ship renders byte-for-byte the shared sprite core painted in ANSI at every width and step, that the Raster packing lays that frame out exactly, that the mod's home resolution and working-note policy match Pi's, and that its operational-input classifier agrees with `bin/fm-operational-input.sh` on a corpus the shell owner itself encodes plus legacy shapes and near misses. +`tests/fm-calm-claude-mod.test.sh` needs no Claude Code binary: it proves the mod is one hooks module with no command, skill, agent, or classic hook path around its opt-in, that Pi's working ship renders byte-for-byte the shared sprite core painted in ANSI at every width and step, that the Raster packing lays that frame out exactly, that the mod resolves its home like Pi, that its live and restored working-note classifiers enforce the visibility boundaries [`calm.md`](calm.md#claude-code) owns, and that its operational-input classifier agrees with `bin/fm-operational-input.sh` on a corpus the shell owner itself encodes plus legacy shapes and near misses. `tests/fm-calm-claude-mod-plugin.test.sh` runs wherever `claude` is installed without spending a model turn: strict `claude plugin validate` on the folder and on the `.claude/skills` auto-load path, then the mod's own `claude plugin test` suites, which drive the hooks module in the engine's host against a mocked clock, environment, file system, and drawing surface. `tests/fm-calm-claude-mod-live-e2e.test.sh` is the opt-in credentialed guard in a real Claude Code TUI under tmux: flag off is a complete no-op with the preference already on, flag on shows the moving boat, hides tool and operational rows, toggles and persists through `/calm`, and `claude --continue` restores the hidden rows. @@ -691,7 +691,7 @@ Three further observations, recorded so they are not read as failures: the `ctrl `.claude/mods/firstmate-calm` holds the plugin: its manifest, `hooks/hooks.json` naming the one module, `hooks/register.ts` (the only file that touches `$`), and pure libraries the tests drive under Node: the sprite core both harnesses share, the Raster packing, the presentation policy, and a port of `bin/fm-operational-input.sh`'s `classify` guarded by a corpus parity test. `.agents/skills/firstmate-calm` is a symlink to it, so the project's `.claude/skills` scan adopts it, and it carries no `SKILL.md` so other harnesses' skill loaders see nothing. The mod declares no command file, skill, agent, or classic hook; its function-hooks handlers independently require the exact environment opt-in before `/calm` registration or any other side effect, including when Claude Code loads the module through its rollout flag. -Working notes are recorded from `turn.step` per text block (a step that stopped for `tool_use`, or `max_tokens` with tool calls) and seeded from `$.session.messages()` for a restored transcript, the same rule as Pi's `assistant-working-note` class. +Working-note and preserved-reply keys are recorded from `turn.step` per text block and seeded from `$.session.messages()` for a restored transcript, with [`calm.md`](calm.md#claude-code) owning the exact Claude Code visibility contract. ```text $ CLAUDE_CODE_ENABLE_FUNCTION_HOOKS=1 claude plugin validate --strict .claude/mods/firstmate-calm diff --git a/docs/calm.md b/docs/calm.md index cf547c44ce9..743ca290daf 100644 --- a/docs/calm.md +++ b/docs/calm.md @@ -76,8 +76,9 @@ On Claude Code the boat is painted in Claude Code's own theme colors rather than The family follows the `theme` setting by its prefix, `dark` or `light`, is re-read when the theme changes, and uses the light set as the both-readable fallback for `auto`, custom, missing, or unreadable values; the Pi extension keeps its standard ANSI blue and yellow. Tool rows, tool result blocks, and folded tool groups draw at zero height, so a turn that used tools takes the same space as one that did not. A user row whose text the canonical operational-input parser recognizes, a Firstmate session-start, watcher, turn-end guard, away-supervisor, launch-brief, or branch-outcome envelope, a from-firstmate routed message, or one of the narrow pre-protocol shapes kept for old transcripts, draws at zero height; every other user row, including near misses such as a quoted or ASCII-only marker, stays visible. -A mid-turn working note, the text of a model step that stopped to call tools or ran out of tokens while calling them, draws at zero height once that step settles, so narration is briefly visible while it streams and then collapses; the reply that ends a response stays visible. -Toggling Calm redraws every hooked row already on screen, so rows drawn before the toggle hide or restore retroactively, and `claude --continue` restores a transcript with Calm's rows still hidden because the preference is read before the first row draws. +A mid-turn working note, the text of a model step that stopped to call tools or ran out of tokens while calling them, draws at zero height once that step settles only when its raw text contains no newline and its trimmed length is below the 240-character preservation threshold. +Mid-turn content whose raw text contains a newline or whose trimmed length is at least 240 characters is preserved and treated as a final reply, including when `claude --continue` restores the transcript. +Toggling Calm redraws every hooked row already on screen, so rows drawn before the toggle hide or restore retroactively, and the preference is read before the first row draws. Nothing is rewritten: hidden rows remain in the message, model context, session storage, and exports, and the mod never touches tool execution, prompts, or the stored transcript. Bounds of the Claude Code support, each recorded with evidence in [`calm-mode-feasibility.md`](calm-mode-feasibility.md#2026-09-15-claude-code-21272-mods-feasibility-and-the-shipped-mod): diff --git a/tests/fm-calm-claude-mod.test.sh b/tests/fm-calm-claude-mod.test.sh index a278ee7d050..7b3898d10ec 100644 --- a/tests/fm-calm-claude-mod.test.sh +++ b/tests/fm-calm-claude-mod.test.sh @@ -248,13 +248,21 @@ for (const [stored, expected] of [["on\\n", true], ["on", true], [" on \\n", tru check(policy.parseCalmPreference(stored) === expected, \`preference \${JSON.stringify(stored)}\`); } check(policy.serializeCalmPreference(true) === "on\\n" && policy.serializeCalmPreference(false) === "off\\n", "serialized values"); -check(policy.stepTextIsWorkingNote({ stopReason: "tool_use", toolUses: [] }) === true, "tool_use"); -check(policy.stepTextIsWorkingNote({ stopReason: "max_tokens", toolUses: [{}] }) === true, "max_tokens with tools"); -check(policy.stepTextIsWorkingNote({ stopReason: "max_tokens", toolUses: [] }) === false, "max_tokens without tools"); -check(policy.stepTextIsWorkingNote({ stopReason: "end_turn", toolUses: [{}] }) === false, "end_turn"); -check(policy.stepTextIsWorkingNote({ stopReason: null, toolUses: [] }) === false, "no response"); -check(policy.workingNoteKey(" note \\n") === "note" && policy.workingNoteKey(" ") === "", "note key"); -const restored = policy.restoredAssistantText([ +const shortNote = "Checking briefly."; +const multiLineReply = "The result is substantive.\\nHere is the context needed to continue."; +const atThresholdReply = "x".repeat(240); +const belowThresholdNote = "x".repeat(239); +check(policy.CALM_PRESERVE_MIN_CHARS === 240, "preservation threshold"); +check(policy.stepTextIsWorkingNote({ stopReason: "tool_use", toolUses: [] }, shortNote) === true, "short single-line tool_use note"); +check(policy.stepTextIsWorkingNote({ stopReason: "tool_use", toolUses: [] }, multiLineReply) === false, "multi-line tool_use reply"); +check(policy.stepTextIsWorkingNote({ stopReason: "tool_use", toolUses: [] }, atThresholdReply) === false, "threshold-length tool_use reply"); +check(policy.stepTextIsWorkingNote({ stopReason: "tool_use", toolUses: [] }, belowThresholdNote) === true, "just-under-threshold tool_use note"); +check(policy.stepTextIsWorkingNote({ stopReason: "max_tokens", toolUses: [{}] }, shortNote) === true, "max_tokens with tools"); +check(policy.stepTextIsWorkingNote({ stopReason: "max_tokens", toolUses: [] }, shortNote) === false, "max_tokens without tools"); +check(policy.stepTextIsWorkingNote({ stopReason: "end_turn", toolUses: [{}] }, shortNote) === false, "end_turn"); +check(policy.stepTextIsWorkingNote({ stopReason: null, toolUses: [] }, shortNote) === false, "no response"); +check(policy.workingNoteKey(" note \\n") === "note\\n" && policy.workingNoteKey(" note ") === "note" && policy.workingNoteKey(" ") === "", "note key"); +const restored = policy.classifyRestoredTranscript([ { role: "user", text: "go", toolUses: [] }, { role: "assistant", text: " own call ", toolUses: [{}] }, { role: "assistant", text: "before a tool row", toolUses: [] }, @@ -265,9 +273,24 @@ const restored = policy.restoredAssistantText([ { role: "assistant", text: "collision", toolUses: [] }, { role: "user", text: "last", toolUses: [] }, { role: "assistant", text: "plain reply", toolUses: [] }, + { role: "user", text: "multi-line case", toolUses: [] }, + { role: "assistant", text: multiLineReply, toolUses: [] }, + { role: "assistant", text: "", toolUses: [{}] }, + { role: "user", text: "threshold case", toolUses: [] }, + { role: "assistant", text: atThresholdReply, toolUses: [] }, + { role: "assistant", text: "", toolUses: [{}] }, + { role: "user", text: "below-threshold case", toolUses: [] }, + { role: "assistant", text: belowThresholdNote, toolUses: [] }, + { role: "assistant", text: "", toolUses: [{}] }, + { role: "user", text: "newline collision", toolUses: [] }, + { role: "assistant", text: "Checking.\\n", toolUses: [] }, + { role: "assistant", text: "", toolUses: [{}] }, + { role: "user", text: "single-line collision", toolUses: [] }, + { role: "assistant", text: "Checking.", toolUses: [] }, + { role: "assistant", text: "", toolUses: [{}] }, ]); -check(JSON.stringify(restored.workingNotes) === JSON.stringify(["own call", "before a tool row"]), \`restored notes \${JSON.stringify(restored.workingNotes)}\`); -check(JSON.stringify(restored.finalReplies) === JSON.stringify(["final", "collision", "plain reply"]), \`restored final replies \${JSON.stringify(restored.finalReplies)}\`); +check(JSON.stringify(restored.workingNotes) === JSON.stringify(["own call", "before a tool row", belowThresholdNote, "Checking."]), \`restored notes \${JSON.stringify(restored.workingNotes)}\`); +check(JSON.stringify(restored.finalReplies) === JSON.stringify(["final", "collision", "plain reply", multiLineReply + "\\n", atThresholdReply, "Checking.\\n"]), \`restored final replies \${JSON.stringify(restored.finalReplies)}\`); check(policy.userTextIsOperational("\\u2063FIRSTMATE_OP: v1 watcher: x") && !policy.userTextIsOperational("hello"), "operational recognition"); console.log("policy-ok"); JS diff --git a/tests/fm-contributions.test.sh b/tests/fm-contributions.test.sh index 017c49f2e1c..b1a2e335ef7 100755 --- a/tests/fm-contributions.test.sh +++ b/tests/fm-contributions.test.sh @@ -455,7 +455,7 @@ test_watcher_keeps_diagnostics_separate_from_contribution_wakes() { out="$home/watcher-diagnostics.out" rc=0 with_home "$home" env FM_POLL=1 FM_SIGNAL_GRACE=0 FM_CHECK_INTERVAL=0 FM_HEARTBEAT=999999 \ - "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 5 > "$out" 2> "$home/watcher-diagnostics.err" || rc=$? + "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 15 > "$out" 2> "$home/watcher-diagnostics.err" || rc=$? [ "$rc" -eq 0 ] || fail "watcher did not surface contribution diagnostics: $(cat "$home/watcher-diagnostics.err")" diagnostic=$(awk -F '\t' -v key="$home/state/contributions.check.sh" '$3 == "check" && $4 == key { print $5 }' "$home/state/.wake-queue") [ "$diagnostic" = "check: $home/state/contributions.check.sh: contributions: 1 unreadable durable record(s)" ] \