From 244358a4e5fd9fa2a48a62fb938643609973e974 Mon Sep 17 00:00:00 2001 From: juniorlovestmh Date: Thu, 30 Jul 2026 07:35:17 -0300 Subject: [PATCH 01/52] feat: pin and restore fleet toolchain offline (#4) * feat: standardize fleet secrets on Doppler (#3) * test: make scheduler refill proof deterministic * feat: standardize fleet secrets on Doppler * no-mistakes(review): Captain: Hardened Doppler runner gates and migration review windows * no-mistakes(review): Captain: Enforced workflow runner validation and sealed clock bypass * no-mistakes(review): Captain: Made Doppler workflow parsing refuse uncertain job structures * no-mistakes(review): Captain: Sealed public and fork-exposed Doppler override paths * no-mistakes(review): Captain: Unified Doppler permission across all trust axes * no-mistakes(review): Captain: Made tracked workflow validation complete and fail-closed * no-mistakes: apply CI fixes --------- Co-authored-by: juniorlovestmh <272474227+juniorlovestmh@users.noreply.github.com> * feat: add offline fleet toolchain mirror * no-mistakes(review): Captain: restore now uses validated captured npm bin mappings --------- Co-authored-by: juniorlovestmh <272474227+juniorlovestmh@users.noreply.github.com> --- .agents/skills/project-management/SKILL.md | 8 +- .agents/skills/secrets-management/SKILL.md | 237 +++++ .github/workflows/ci.yml | 2 + AGENTS.md | 1 + README.md | 1 + bin/fm-secrets-check.mjs | 1063 +++++++++++++++++++ bin/fm-secrets-check.sh | 53 + bin/fm-test-run.sh | 12 +- bin/fm-toolchain-mirror.sh | 457 ++++++++ docs/configuration.md | 1 + docs/examples/doppler-oidc-job.yml | 33 + docs/examples/doppler-service-token-job.yml | 26 + docs/examples/project-secrets-policy.json | 27 + docs/promotion-ladder.md | 7 +- docs/scripts.md | 2 + docs/secrets-policy.schema.json | 313 ++++++ docs/secrets-rollout.json | 393 +++++++ docs/secrets-rollout.schema.json | 164 +++ docs/toolchain-versions.md | 104 ++ tests/fm-secrets-check.test.sh | 528 +++++++++ tests/fm-test-run.test.sh | 30 +- tests/fm-toolchain-mirror.test.sh | 128 +++ 22 files changed, 3579 insertions(+), 11 deletions(-) create mode 100644 .agents/skills/secrets-management/SKILL.md create mode 100755 bin/fm-secrets-check.mjs create mode 100755 bin/fm-secrets-check.sh create mode 100755 bin/fm-toolchain-mirror.sh create mode 100644 docs/examples/doppler-oidc-job.yml create mode 100644 docs/examples/doppler-service-token-job.yml create mode 100644 docs/examples/project-secrets-policy.json create mode 100644 docs/secrets-policy.schema.json create mode 100644 docs/secrets-rollout.json create mode 100644 docs/secrets-rollout.schema.json create mode 100644 docs/toolchain-versions.md create mode 100755 tests/fm-secrets-check.test.sh create mode 100755 tests/fm-toolchain-mirror.test.sh diff --git a/.agents/skills/project-management/SKILL.md b/.agents/skills/project-management/SKILL.md index af35d469ee2..f2cc0b11a17 100644 --- a/.agents/skills/project-management/SKILL.md +++ b/.agents/skills/project-management/SKILL.md @@ -3,7 +3,7 @@ name: project-management description: >- Agent-only procedure for Firstmate project management. Use before adding, creating, removing, or initializing a project. - Owns project add, create, clone, remove, initialization, registry, delivery-mode, autonomy, and outward-consent decisions. + Owns project add, create, clone, remove, initialization, registry, delivery-mode, autonomy, outward-consent decisions, and the secrets-intake handoff. user-invocable: false metadata: internal: true @@ -68,6 +68,12 @@ Initialization configures the local gate and does not vendor a no-mistakes skill Do not create a commit merely because initialization ran. If doctor reports an environment, authentication, or daemon problem, resolve that blocker before dispatching work and never restart the shared daemon from a project operation. +Load `secrets-management` during every project intake or initialization. +From the Firstmate root, run `bin/fm-secrets-check.sh inventory projects/` after the gate check. +The inventory is read-only and value-safe, and it is not a substitute for the declared project classification. +If the project lacks `docs/secrets-policy.json`, assign its first ship task to copy and complete `docs/examples/project-secrets-policy.json` before any secret-bearing workflow or deployment change. +Firstmate never hand-writes that project file. + ## Remove Project removal is destructive and is not one of Firstmate's current direct-write exceptions under `projects/`. diff --git a/.agents/skills/secrets-management/SKILL.md b/.agents/skills/secrets-management/SKILL.md new file mode 100644 index 00000000000..c6d8dfa1e0e --- /dev/null +++ b/.agents/skills/secrets-management/SKILL.md @@ -0,0 +1,237 @@ +--- +name: secrets-management +description: >- + Agent-only policy for Firstmate-managed secrets. + Use before project intake or initialization and before work that handles credentials or adds secret access to CI or deployment. + Owns Doppler defaults, secretless identity preference, naming, least privilege, owned-runner injection, local use, redaction, rotation, offboarding, break-glass, exceptions, and migration. +user-invocable: false +metadata: + internal: true +--- + +# secrets-management + +Use this procedure before project intake or initialization and before work that handles credentials or adds secret access to CI or deployment. +This skill is the single owner of Firstmate's secrets-management policy. +The tracked schemas and rollout data are validated by `bin/fm-secrets-check.sh`. + +## Authority and precedence + +Prefer no stored secret when a workload can prove its identity directly to the provider. +Google Cloud Workload Identity Federation, GitHub's job-scoped token, and trusted publishing are preferred over creating a vault entry that must later be protected and rotated. + +When a real secret is required, Doppler is the default source of truth for project and CI secrets. +Do not add a second general-purpose vault or use a platform store as an undocumented alternative source of truth. +A platform-mandated secret store may receive the value needed by that platform at deploy time, but the project manifest must document why that copy exists and which system is authoritative. + +1Password is only the sealed break-glass store for platform-root and recovery credentials. +Agents receive no ambient 1Password access and never use 1Password as the normal project or CI path. +Changing that boundary requires the captain's explicit decision. + +Anything that rotates, transfers, moves, or exposes a real credential requires escalation before the value is handled. +Anything that commits the captain to a paid Doppler tier also requires escalation. + +## Project and environment names + +Use the repository or stable product slug as the Doppler project name. +Record any legacy name and its migration reason in the project manifest instead of creating an undocumented alias. + +Use `dev`, `stg`, and `prd` as the canonical configs. +Local development uses `dev`, persistent pre-production uses `stg`, and production uses `prd`. +A pull-request preview may inherit `stg` only when it is isolated from production data and the manifest documents that choice. +An ephemeral config uses `pr-`, has no production values, and is removed when the pull request closes. +Do not point a production GitHub Environment at `dev` or `stg`. + +The committed project manifest is `docs/secrets-policy.json`. +It contains names, scopes, identity modes, exceptions, and review metadata only. +It never contains a secret value, token fingerprint, encoded credential, or reversible derivative. +Validate it with `bin/fm-secrets-check.sh manifest `. + +## Least-privilege access + +Grant humans individual Doppler access only to the projects and environments required by their role. +Normal local development access is read-only for `dev`. +Staging or production access is granted only for a named task and is removed when that task or role ends. + +Agents use the human-approved local Doppler session only for the named project, config, and command in scope. +Agents do not create tokens, broaden project membership, download an environment, enumerate values, or fall back to 1Password. +If the required project or config is unavailable, stop and escalate rather than borrowing access from another environment. + +CI uses one identity per project and environment. +Config-scoped read-only service tokens are the shipped CI default. +Each token is scoped to exactly one Doppler project config. +Store the service token as the matching GitHub Environment secret named `DOPPLER_TOKEN`. +Never use one service token across projects or across `dev`, `stg`, and `prd`. +Never grant a CI token write access. + +## Owned-runner injection + +The runner host remains credential-free. +`/usr/local/sbin/gh-runner-preflight` must find no Doppler token, Doppler project selection, 1Password token, GitHub token, cloud key, credential file, or secret-bearing shell profile in the runner service environment or home before a job is admitted. +A standing token on the runner is prohibited even if file permissions are restrictive. + +GitHub sends the credential only after the job is admitted. +The workflow attaches the matching GitHub Environment to the secret-consuming job. +The `DOPPLER_TOKEN` secret is passed only to the pinned Doppler Secrets Fetch action step, never at workflow, job, runner-service, or shell-profile scope. +The action fetches the exact config, registers fetched values for GitHub log masking, and exposes them only to later steps in that job. +Every secret in Doppler must use masked visibility. +The job must not persist the action output or a Doppler CLI configuration under the runner home. + +The secret lifetime on the machine is the job lifetime. +The runner process discards the job environment and GitHub removes runner-managed job temporary files when the job ends. +The next job passes the same credential-free preflight before receiving work. +If a workflow creates an additional secret-bearing temporary file, it must use a `0700` directory under `RUNNER_TEMP`, install an `if: always()` cleanup step, and prove absence without printing contents. + +Copy the shipped job shape from `docs/examples/doppler-service-token-job.yml`. +The OIDC file at `docs/examples/doppler-oidc-job.yml` is an optional reference only after the captain approves a paid-plan change. +Both templates pin Doppler's v1.3.0 Secrets Fetch action to commit `cd2efbf9a404504316435873eff298b82f7e0562`. +Both also pin actions/checkout v6.0.2 to commit `de0fac2e4500dabe0009e67214ff5f5447ce83dd` before secret injection. + +Install actions and tools before injecting secrets whenever possible. +Check out only trusted code before injection. +Never expose Doppler or provider credentials to pull requests from forks, Dependabot, or another untrusted event. + +The sanctioned `waku-agent` hosted-runner exception remains documented because it is a public fork whose untrusted pull requests must not run on captain-owned hardware. +That runner exception does not authorize secrets on untrusted events. + +Doppler injection is permitted only when the runner target is provably owned AND the triggering event is provably not fork-originated AND the repository exposure cannot admit an untrusted trigger; unknown on ANY axis refuses; no override unlocks the fork or public axes. +Workflow validation is refuse-unless-provable: parser uncertainty, quoted or ambiguous job structure, missing or computed `runs-on`, unknown labels, missing or ambiguous triggers, and indirection all fail closed. The `repositoryVisibility` and `forkExposure` policy fields are explicit walls, and a `hosted-runner` or `non-owned-runner-doppler` exception never changes a false permission result. +Workflow intake is two-phase: enumerate the complete authoritative tracked workflow set first, then record exactly one `permit-with-proof` or `refuse` verdict for every enumerated workflow; unreadable or unavailable content refuses, and any missing, duplicate, or extra verdict fails validation. + +## Captain decision record for optional Doppler OIDC + +The 2026-07-27 owned-runner pilot proves that GitHub job OIDC works from the fleet runner and that the runner can keep its host environment credential-free. +Doppler Service Account Identities can exchange that GitHub OIDC assertion for a short-lived Doppler token without storing `DOPPLER_TOKEN`. +Doppler currently limits that feature to Team and Enterprise plans. +As checked on 2026-07-27, Doppler lists Team at $21 per user per month and Enterprise at custom pricing. + +The recommendation for the captain is to consider Doppler OIDC for secret-bearing GitHub Actions environments because it removes the stored Doppler CI token and its rotation burden. +Whether that benefit justifies the paid plan is the captain's decision. +This recommendation does not authorize a purchase, trial, plan change, or OIDC rollout. + +If the captain approves a paid plan and a separate implementation, each identity should be read-only, scoped to one Doppler project and environment, and bound with exact audience and subject claims plus immutable repository identity, GitHub Environment, workflow identity, and `runner_environment=self-hosted` where supported. +Avoid wildcard claim rules. +Store the Doppler identity ID as a non-secret GitHub Environment variable. +Grant `id-token: write` only to the job that fetches secrets. + +Do not upgrade the plan as part of routine implementation. +The shipped standard does not require OIDC, a service account, or a paid plan. +Project/config-scoped read-only service tokens with per-job injection remain the official CI implementation unless the captain separately approves the plan and migration. +That service-token default never permits a token to persist on the runner. + +Official references: + +- [Doppler pricing](https://www.doppler.com/pricing) +- [Doppler Service Tokens](https://docs.doppler.com/docs/service-tokens) +- [Doppler Service Account Identities](https://docs.doppler.com/docs/service-account-identities) +- [Doppler GitHub OIDC examples](https://docs.doppler.com/docs/github-oidc-examples) +- [GitHub OIDC claims](https://docs.github.com/en/actions/reference/security/oidc) +- [GitHub self-hosted runner network requirements](https://docs.github.com/en/actions/reference/runners/self-hosted-runners) + +## Local use + +Humans authenticate the Doppler CLI with their own identity. +Scope project selection to the repository directory and select `dev` by default. +Do not paste a token into a command, prompt, script, dotfile, shell profile, or committed environment file. + +Run the consuming process through `doppler run --project --config -- `. +Use environment variables or standard input when a downstream tool needs a value. +Do not put a secret in a command argument because process listings, audit events, and shell history may retain it. +Production local access is exceptional and requires the named task to authorize it. + +## Redaction and evidence + +Never print a secret value to prove that it exists. +Presence checks report only the secret name, boolean presence, source project/config, and command exit status. +Do not run `printenv`, `env`, `set -x`, an unredacted Compose render, `doppler secrets download`, or `doppler secrets get --plain` in captured output. +Do not paste a value into a prompt, issue, report, pull request, commit, test fixture, or status message. + +Logs and reports may contain secret names, project names, config names, rotation dates, and opaque decision identifiers. +They must not contain values, partial values, fingerprints derived from values, or screenshots that reveal them. +When a command fails, sanitize its output before retaining evidence. + +Run `bin/fm-secrets-check.sh leak-scan ` against every tracked artifact produced by secret-management work. +The scanner reports only the file, line, and matched rule. +It never echoes the matching text. +A clean scan is supporting evidence, not proof that an untracked or external system contains no secret. + +## Rotation and incident response + +Set Doppler service-token expiry to no more than 90 days when the current plan supports the required expiry. +Review every project manifest at least every 90 days. +Rotate sooner when a provider requires it, access changes, the token scope changes, or exposure is suspected. + +Perform a normal rotation in this order: + +1. Create a replacement with the same or narrower scope through an approved interactive path. +2. Update only the matching GitHub Environment secret. +3. Run a presence-only canary on the intended environment and owned runner. +4. Revoke the old token after the canary passes. +5. Record the identity name, scope, timestamps, evidence URL, and operator without recording either value. + +On suspected exposure, revoke first, stop affected jobs, rotate downstream credentials the token could read, and preserve only sanitized evidence. +Do not wait for a normal canary before revoking a credential believed exposed. + +OIDC removes Doppler service-token rotation but not rotation of provider credentials stored in Doppler. +Keep provider rotation independent and follow the provider's shorter limit when one exists. + +## Offboarding + +Remove the person's Doppler membership and project roles. +Revoke personal CLI sessions and any service token created for that person's task. +Remove GitHub Environment access and any local project authorization. +Rotate each shared provider credential the person could retrieve. +Verify with name-only inventory and audit events that access is gone. +Do not add agents as Doppler users, so an agent shutdown normally requires no vault offboarding. + +## Break-glass + +Break-glass covers Doppler workplace recovery, cloud or forge root recovery, and another platform-root credential whose normal identity path cannot recover itself. +The sealed 1Password item remains captain-controlled and unavailable to agents. +Use requires a named incident, explicit captain authority, the minimum credential, and a bounded recovery window. +After use, rotate the recovered root credential, re-seal the replacement, revoke temporary access, and record value-free evidence. +Do not copy a platform-root credential into a normal Doppler project. + +## Exceptions + +Every exception lives in `docs/secrets-policy.json` with a narrow scope, owner, review cadence, and non-empty reason. +An exception without a reason is invalid. + +Allowed exception kinds are: + +- `secretless` for a project or job that has no secret and therefore needs no Doppler project. +- `provider-oidc` for direct short-lived identity such as GCP Workload Identity Federation. +- `platform-mandated` for a provider store or job token the platform requires. +- `break-glass` for the sealed root recovery boundary. +- `hosted-runner` for the named public-fork safety exception. +- `shared-nonproduction-config` for an isolated preview that intentionally shares `stg`, never production data. +- `temporary-migration` for a time-bounded incompatibility with an explicit removal step. +- `non-owned-runner-doppler` only for a named workflow that has a documented reason to inject Doppler secrets on a non-owned runner. + +Every `temporary-migration` exception records a `reviewBy` date no more than 90 days away and a non-empty `removalCondition`. +Cost, convenience, an existing unscoped token, or a token already present on a machine is not a valid exception. + +## Project intake and migration + +During project intake, run `bin/fm-secrets-check.sh inventory `. +The inventory reads tracked filenames and reference classes, and rejects Doppler +workflow injection unless the complete three-axis predicate is proven. +It never reads untracked environment files or prints matched lines. + +If `docs/secrets-policy.json` is absent, the first project ship copies `docs/examples/project-secrets-policy.json` from Firstmate, replaces the example fields, and validates it. +Firstmate does not hand-write that file in a project clone. +A secretless project still records the reason it needs no Doppler project. + +Migrate without exposing or silently moving a real value: + +1. Inventory names, consumers, environments, owners, and current stores without reading values into logs. +2. Classify every credential as unnecessary identity, Doppler-managed, platform-mandated, bootstrap root, or break-glass. +3. Create the canonical Doppler project and `dev`, `stg`, and `prd` configs without values. +4. Ask the captain or approved operator to transfer real values through a secure interactive path. +5. Configure the per-environment CI identity and per-job injection. +6. Run presence-only and real consumer canaries without value output. +7. Remove the old read path only after the new path passes. +8. Revoke or rotate the old credential and update the manifest. + +Never use a general export, bulk environment dump, prompt, report, or shell-history command as a migration transport. +Do not delete the old path until rollback evidence exists. diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c04c0965c42..8ae5f669bcd 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -23,6 +23,8 @@ jobs: # Single owner of the lint definition (file set + config + version). Do not # re-spell the shellcheck command here; keep CI and the pre-push gate on it. - run: bin/fm-lint.sh + - name: Validate tracked secrets standard + run: bin/fm-secrets-check.sh # Deterministic proof that portable parallel shards + portable serial + Herdr # equal the complete tests/*.test.sh inventory with no missing or duplicates. diff --git a/AGENTS.md b/AGENTS.md index 979cfe6ff9e..fbc3d1ed4b2 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -469,6 +469,7 @@ These skills are not captain-invocable; load them only at their precise triggers - `harness-adapters` - load before spawning or recovering a crewmate or secondmate, handling a trust dialog, sending a harness-specific skill invocation, interrupting or exiting an agent, resuming an exited agent, or verifying a new harness adapter. - `firstmate-orca` - load before switching to Orca, spawning or supervising Orca-backed work, smoke-testing Orca backend behavior, debugging Orca task state, or reconciling Orca-backed task metadata. - `project-management` - load before adding, creating, removing, or initializing a project. +- `secrets-management` - load before project intake or initialization and before work that handles credentials or adds secret access to CI or deployment. - `stuck-crewmate-recovery` - load when the session-start digest reports an ordinary direct report's endpoint dead or its metadata has no window, or after a stale wake, looping pane, repeated confusion, an answered-by-brief question, an unresponsive crewmate, or a failed steer. - `secondmate-provisioning` - load before creating, seeding, validating, launching, handing backlog to, recovering, pushing inherited local material into, or retiring a secondmate home, and before editing `data/secondmates.md`. - `decision-hold-lifecycle` - load before treating an investigation or visual review as complete, before ending a visual review that exposed a decision, and when recording or routing the captain's answer. diff --git a/README.md b/README.md index f7e92ae4eb9..7eb82be38c3 100644 --- a/README.md +++ b/README.md @@ -50,6 +50,7 @@ Launching a supported harness inside it instantiates your first mate - and makes - **Event-driven, zero-token supervision** - a bash watcher sleeps on the fleet and wakes the first mate only when something needs you; verified primary harnesses also get a turn-end backstop that blocks or follows up on a blind stop when work is under way and supervision is not live. - **Optional X mode** - opt in with one local `.env` token so firstmate can answer your public `@myfirstmate` mentions, act on normal reversible mention requests through the same lifecycle as chat requests, acknowledge spawned work, and post up to three public-safe completion follow-ups within seven days for genuine milestones and the final outcome without changing non-X behavior; dry-run preview records would-be replies and dismissals locally before go-live. - **Guarded by construction** - the first mate is read-only over your projects except for the guarded paths authorized by [hard rule 1](AGENTS.md#1-identity-and-prime-directives), with fleet sync's safe branch pruning remaining part of the fleet-sync exception; crewmates make every project change behind the configured merge authority. +- **Doppler by default** - the conditional [secrets-management policy](.agents/skills/secrets-management/SKILL.md) prefers secretless provider identity, otherwise scopes Doppler by project and environment, and validates declarations and rollout data through `bin/fm-secrets-check.sh`. - **Restart-proof** - all state lives on disk and in the active session backend (tmux by hard default, herdr or cmux when selected or auto-detected, zellij/orca when explicitly selected); kill the session anytime and the next one reconciles, including confirmed-dead secondmate agents, and carries on. Full detail on every feature lives in [docs/architecture.md](docs/architecture.md). diff --git a/bin/fm-secrets-check.mjs b/bin/fm-secrets-check.mjs new file mode 100755 index 00000000000..0d20f8b59ed --- /dev/null +++ b/bin/fm-secrets-check.mjs @@ -0,0 +1,1063 @@ +#!/usr/bin/env node + +import { execFileSync } from "node:child_process"; +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +const scriptDir = path.dirname(fileURLToPath(import.meta.url)); +const root = path.resolve(scriptDir, ".."); +const policySkill = path.join(root, ".agents/skills/secrets-management/SKILL.md"); +const agentsFile = path.join(root, "AGENTS.md"); +const policySchema = path.join(root, "docs/secrets-policy.schema.json"); +const rolloutSchema = path.join(root, "docs/secrets-rollout.schema.json"); +const rolloutFile = path.join(root, "docs/secrets-rollout.json"); +const exampleFile = path.join(root, "docs/examples/project-secrets-policy.json"); +const serviceTokenTemplate = path.join(root, "docs/examples/doppler-service-token-job.yml"); +const oidcTemplate = path.join(root, "docs/examples/doppler-oidc-job.yml"); +const standardArtifacts = [ + policySkill, + path.join(root, ".agents/skills/project-management/SKILL.md"), + path.join(root, ".github/workflows/ci.yml"), + agentsFile, + path.join(root, "README.md"), + path.join(root, "bin/fm-secrets-check.mjs"), + path.join(root, "bin/fm-secrets-check.sh"), + path.join(root, "bin/fm-test-run.sh"), + path.join(root, "docs/promotion-ladder.md"), + path.join(root, "docs/scripts.md"), + policySchema, + rolloutSchema, + rolloutFile, + exampleFile, + serviceTokenTemplate, + oidcTemplate, + path.join(root, "tests/fm-secrets-check.test.sh"), +]; + +const allowedClassifications = new Set(["doppler", "secretless", "platform-native"]); +const allowedRunners = new Set([ + "owned-linux", + "owned-macos", + "github-hosted-public-fork", + "none", +]); +const allowedIdentities = new Set([ + "none", + "doppler-service-token", + "doppler-oidc", + "provider-oidc", + "github-job-token", + "mixed", +]); +const allowedInjections = new Set([ + "none", + "per-job-doppler-fetch", + "job-oidc", + "github-job-context", + "mixed", +]); +const allowedExceptions = new Set([ + "secretless", + "provider-oidc", + "platform-mandated", + "break-glass", + "hosted-runner", + "shared-nonproduction-config", + "temporary-migration", + "non-owned-runner-doppler", +]); +const canonicalConfigs = new Set(["dev", "stg", "prd"]); +const expectedBacklogIds = [ + "ci-selfhost-appheat-site-h1", + "ci-selfhost-bible-agents-h2", + "ci-selfhost-boostin-h3", + "ci-selfhost-resume-matcher-h4", + "ci-selfhost-factory-h5", + "ci-selfhost-organicops-h6", + "ci-selfhost-tab-agent-h7", + "ci-selfhost-macops-registry-h8", +]; + +function fail(message) { + process.stderr.write(`fm-secrets-check: ${message}\n`); + process.exitCode = 1; +} + +function readJson(file) { + try { + return JSON.parse(fs.readFileSync(file, "utf8")); + } catch (error) { + fail(`${file}: invalid JSON: ${error.message}`); + return null; + } +} + +function nonEmptyString(value) { + return typeof value === "string" && value.trim().length > 0; +} + +function parseIsoDate(value) { + const match = /^(\d{4})-(\d{2})-(\d{2})$/.exec(value || ""); + if (!match) { + return null; + } + const timestamp = Date.UTC(Number(match[1]), Number(match[2]) - 1, Number(match[3])); + const date = new Date(timestamp); + return date.toISOString().slice(0, 10) === value ? date : null; +} + +function policyToday() { + return new Date(`${new Date().toISOString().slice(0, 10)}T00:00:00Z`); +} + +function validateSchemaDocument(file, expectedTitle) { + const doc = readJson(file); + if (!doc) { + return; + } + if (doc.$schema !== "https://json-schema.org/draft/2020-12/schema") { + fail(`${file}: $schema must be JSON Schema draft 2020-12`); + } + if (doc.title !== expectedTitle) { + fail(`${file}: title must be ${JSON.stringify(expectedTitle)}`); + } + if (doc.type !== "object") { + fail(`${file}: root type must be object`); + } + if (!Array.isArray(doc.required) || doc.required.length === 0) { + fail(`${file}: required must be a non-empty array`); + } +} + +function sameJsonValue(left, right) { + return JSON.stringify(left) === JSON.stringify(right); +} + +function schemaTypeMatches(value, type) { + switch (type) { + case "null": + return value === null; + case "array": + return Array.isArray(value); + case "object": + return value !== null && typeof value === "object" && !Array.isArray(value); + case "integer": + return Number.isInteger(value); + case "number": + return typeof value === "number" && Number.isFinite(value); + default: + return typeof value === type; + } +} + +function schemaPath(parent, key) { + if (typeof key === "number") { + return `${parent}[${key}]`; + } + return /^[A-Za-z_][A-Za-z0-9_]*$/.test(key) + ? `${parent}.${key}` + : `${parent}[${JSON.stringify(key)}]`; +} + +function resolveJsonPointer(document, fragment, ref) { + if (!fragment) { + return document; + } + if (!fragment.startsWith("/")) { + throw new Error(`unsupported schema fragment in ${ref}`); + } + return fragment + .slice(1) + .split("/") + .map((part) => part.replace(/~1/g, "/").replace(/~0/g, "~")) + .reduce((value, part) => { + if (value === undefined || value === null || !(part in value)) { + throw new Error(`unresolved schema pointer ${ref}`); + } + return value[part]; + }, document); +} + +const schemaCache = new Map(); + +function loadSchema(file) { + if (!schemaCache.has(file)) { + const document = readJson(file); + if (!document) { + return null; + } + schemaCache.set(file, document); + } + return schemaCache.get(file); +} + +function validateSchemaValue(value, schema, sourceFile, instancePath = "$") { + const errors = []; + const add = (message) => errors.push(`${instancePath} ${message}`); + + if (typeof schema === "boolean") { + if (!schema) { + add("is forbidden by the schema"); + } + return errors; + } + if (!schema || typeof schema !== "object" || Array.isArray(schema)) { + add("has an invalid schema"); + return errors; + } + + if (schema.$ref) { + try { + const [relativeFile, fragment = ""] = schema.$ref.split("#", 2); + const refFile = relativeFile + ? path.resolve(path.dirname(sourceFile), decodeURIComponent(relativeFile)) + : sourceFile; + const refDocument = loadSchema(refFile); + if (refDocument) { + const target = resolveJsonPointer(refDocument, fragment, schema.$ref); + errors.push(...validateSchemaValue(value, target, refFile, instancePath)); + } + } catch (error) { + add(`cannot resolve schema reference ${JSON.stringify(schema.$ref)}: ${error.message}`); + } + } + + if ("const" in schema && !sameJsonValue(value, schema.const)) { + add(`must equal ${JSON.stringify(schema.const)}`); + } + if (Array.isArray(schema.enum) && !schema.enum.some((item) => sameJsonValue(value, item))) { + add(`must be one of ${schema.enum.map((item) => JSON.stringify(item)).join(", ")}`); + } + + if (schema.type) { + const types = Array.isArray(schema.type) ? schema.type : [schema.type]; + if (!types.some((type) => schemaTypeMatches(value, type))) { + add(`must have type ${types.join(" or ")}`); + return errors; + } + } + + if (Array.isArray(schema.allOf)) { + schema.allOf.forEach((part) => { + errors.push(...validateSchemaValue(value, part, sourceFile, instancePath)); + }); + } + if (schema.if) { + const conditionErrors = validateSchemaValue(value, schema.if, sourceFile, instancePath); + if (conditionErrors.length === 0 && schema.then) { + errors.push(...validateSchemaValue(value, schema.then, sourceFile, instancePath)); + } else if (conditionErrors.length > 0 && schema.else) { + errors.push(...validateSchemaValue(value, schema.else, sourceFile, instancePath)); + } + } + + if (typeof value === "string") { + if (Number.isInteger(schema.minLength) && value.length < schema.minLength) { + add(`must have length at least ${schema.minLength}`); + } + if (schema.pattern && !new RegExp(schema.pattern).test(value)) { + add(`must match ${JSON.stringify(schema.pattern)}`); + } + if (schema.format === "date") { + const match = /^(\d{4})-(\d{2})-(\d{2})$/.exec(value); + const timestamp = match + ? Date.UTC(Number(match[1]), Number(match[2]) - 1, Number(match[3])) + : Number.NaN; + const canonical = Number.isNaN(timestamp) + ? "" + : new Date(timestamp).toISOString().slice(0, 10); + if (canonical !== value) { + add("must be a valid YYYY-MM-DD date"); + } + } + } + + if (Array.isArray(value)) { + if (Number.isInteger(schema.minItems) && value.length < schema.minItems) { + add(`must contain at least ${schema.minItems} items`); + } + if (Number.isInteger(schema.maxItems) && value.length > schema.maxItems) { + add(`must contain at most ${schema.maxItems} items`); + } + if (schema.uniqueItems) { + const serialized = value.map((item) => JSON.stringify(item)); + if (new Set(serialized).size !== serialized.length) { + add("must contain unique items"); + } + } + if (schema.items) { + value.forEach((item, index) => { + errors.push( + ...validateSchemaValue(item, schema.items, sourceFile, schemaPath(instancePath, index)), + ); + }); + } + if (schema.contains) { + const matching = value.filter( + (item, index) => + validateSchemaValue( + item, + schema.contains, + sourceFile, + schemaPath(instancePath, index), + ).length === 0, + ).length; + const minimum = Number.isInteger(schema.minContains) ? schema.minContains : 1; + if (matching < minimum) { + add(`must contain at least ${minimum} matching item`); + } + } + } + + if (value !== null && typeof value === "object" && !Array.isArray(value)) { + const properties = schema.properties || {}; + if (Array.isArray(schema.required)) { + schema.required.forEach((key) => { + if (!(key in value)) { + errors.push(`${schemaPath(instancePath, key)} is required`); + } + }); + } + Object.entries(properties).forEach(([key, propertySchema]) => { + if (key in value) { + errors.push( + ...validateSchemaValue( + value[key], + propertySchema, + sourceFile, + schemaPath(instancePath, key), + ), + ); + } + }); + if (schema.additionalProperties === false) { + Object.keys(value) + .filter((key) => !(key in properties)) + .forEach((key) => { + errors.push(`${schemaPath(instancePath, key)} is not allowed`); + }); + } + } + + if (typeof value === "number") { + if (typeof schema.minimum === "number" && value < schema.minimum) { + add(`must be at least ${schema.minimum}`); + } + if (typeof schema.maximum === "number" && value > schema.maximum) { + add(`must be at most ${schema.maximum}`); + } + } + + return errors; +} + +function validateAgainstSchema(doc, schemaFile, label) { + const schema = loadSchema(schemaFile); + if (!schema) { + return []; + } + return validateSchemaValue(doc, schema, schemaFile).map((error) => `${label}: ${error}`); +} + +function validateManifestSemantics(doc, label, today = policyToday()) { + const errors = []; + const add = (field, message) => errors.push(`${label}: ${field} ${message}`); + + if (!doc || typeof doc !== "object" || Array.isArray(doc)) { + add("$", "must be an object"); + return errors; + } + if (doc.schemaVersion !== 1) { + add("schemaVersion", "must equal 1"); + } + if (!nonEmptyString(doc.project)) { + add("project", "must be a non-empty string"); + } + if (!nonEmptyString(doc.repository) || !doc.repository.includes("/")) { + add("repository", "must be an owner/name string"); + } + if (doc.repositoryVisibility !== "private" && doc.repositoryVisibility !== "public") { + add("repositoryVisibility", "must be private or public"); + } + if (!["none", "pull-request", "pull-request-target"].includes(doc.forkExposure)) { + add("forkExposure", "must be none, pull-request, or pull-request-target"); + } + if (!allowedClassifications.has(doc.classification)) { + add("classification", `must be one of ${[...allowedClassifications].join(", ")}`); + } + + if (!doc.doppler || typeof doc.doppler !== "object" || Array.isArray(doc.doppler)) { + add("doppler", "must be an object"); + } else { + if (doc.doppler.project !== null && !nonEmptyString(doc.doppler.project)) { + add("doppler.project", "must be null or a non-empty string"); + } + if (!Array.isArray(doc.doppler.configs)) { + add("doppler.configs", "must be an array"); + } else { + const seen = new Set(); + doc.doppler.configs.forEach((config, index) => { + if (!canonicalConfigs.has(config)) { + add(`doppler.configs[${index}]`, "must be dev, stg, or prd"); + } + if (seen.has(config)) { + add(`doppler.configs[${index}]`, "must be unique"); + } + seen.add(config); + }); + if (doc.classification === "doppler" && doc.doppler.configs.length === 0) { + add("doppler.configs", "must not be empty for a Doppler project"); + } + if (doc.classification === "secretless" && doc.doppler.configs.length !== 0) { + add("doppler.configs", "must be empty for a secretless project"); + } + } + if (doc.classification === "doppler" && !nonEmptyString(doc.doppler.project)) { + add("doppler.project", "must name the project for classification doppler"); + } + if (doc.classification === "secretless" && doc.doppler.project !== null) { + add("doppler.project", "must be null for a secretless project"); + } + } + + if (!doc.ci || typeof doc.ci !== "object" || Array.isArray(doc.ci)) { + add("ci", "must be an object"); + } else { + if (!allowedRunners.has(doc.ci.runner)) { + add("ci.runner", `must be one of ${[...allowedRunners].join(", ")}`); + } + if (!allowedIdentities.has(doc.ci.identity)) { + add("ci.identity", `must be one of ${[...allowedIdentities].join(", ")}`); + } + if (!allowedInjections.has(doc.ci.injection)) { + add("ci.injection", `must be one of ${[...allowedInjections].join(", ")}`); + } + } + + if (!Array.isArray(doc.exceptions)) { + add("exceptions", "must be an array"); + } else { + doc.exceptions.forEach((exception, index) => { + if (!exception || typeof exception !== "object" || Array.isArray(exception)) { + add(`exceptions[${index}]`, "must be an object"); + return; + } + if (!allowedExceptions.has(exception.kind)) { + add(`exceptions[${index}].kind`, `must be one of ${[...allowedExceptions].join(", ")}`); + } + if (!nonEmptyString(exception.scope)) { + add(`exceptions[${index}].scope`, "must be a non-empty string"); + } + if (!nonEmptyString(exception.reason)) { + add(`exceptions[${index}].reason`, "must be a non-empty string"); + } + if (exception.kind === "temporary-migration") { + const reviewDate = parseIsoDate(exception.reviewBy); + if (!reviewDate) { + add(`exceptions[${index}].reviewBy`, "must be a YYYY-MM-DD date"); + } else { + const latestReview = new Date(today); + latestReview.setUTCDate(latestReview.getUTCDate() + 90); + if (reviewDate < today || reviewDate > latestReview) { + add( + `exceptions[${index}].reviewBy`, + "must be today or within the next 90 days", + ); + } + } + if (!nonEmptyString(exception.removalCondition)) { + add(`exceptions[${index}].removalCondition`, "must be a non-empty string"); + } + } + if (exception.kind === "non-owned-runner-doppler" && !nonEmptyString(exception.workflow)) { + add(`exceptions[${index}].workflow`, "must identify the exceptional workflow"); + } + }); + if ( + doc.classification === "secretless" && + !doc.exceptions.some((exception) => exception?.kind === "secretless") + ) { + add("exceptions", "must document the secretless reason"); + } + if ( + doc.classification === "platform-native" && + !doc.exceptions.some((exception) => exception?.kind === "platform-mandated") + ) { + add("exceptions", "must document the platform-mandated reason"); + } + } + + if (!doc.review || typeof doc.review !== "object" || Array.isArray(doc.review)) { + add("review", "must be an object"); + } else { + if (!nonEmptyString(doc.review.owner)) { + add("review.owner", "must be a non-empty string"); + } + if ( + !Number.isInteger(doc.review.cadenceDays) || + doc.review.cadenceDays < 1 || + doc.review.cadenceDays > 90 + ) { + add("review.cadenceDays", "must be an integer from 1 through 90"); + } + } + + if (doc.ci && typeof doc.ci === "object" && !Array.isArray(doc.ci)) { + const expectedInjection = new Map([ + ["none", "none"], + ["doppler-service-token", "per-job-doppler-fetch"], + ["doppler-oidc", "job-oidc"], + ["provider-oidc", "job-oidc"], + ["github-job-token", "github-job-context"], + ["mixed", "mixed"], + ]).get(doc.ci.identity); + if (expectedInjection && doc.ci.injection !== expectedInjection) { + add("ci.injection", `must be ${expectedInjection} for identity ${doc.ci.identity}`); + } + const secretBearingIdentity = + doc.ci.identity === "doppler-service-token" || doc.ci.identity === "doppler-oidc"; + const ownedRunner = doc.ci.runner === "owned-linux" || doc.ci.runner === "owned-macos"; + const hostedException = Array.isArray(doc.exceptions) + ? doc.exceptions.some((exception) => exception?.kind === "hosted-runner") + : false; + const overrideUnavailable = + doc.repositoryVisibility !== "private" || doc.forkExposure !== "none" || hostedException; + const declaredOverride = Array.isArray(doc.exceptions) + ? doc.exceptions.some((exception) => exception?.kind === "non-owned-runner-doppler") + : false; + if (declaredOverride && overrideUnavailable) { + add( + "exceptions", + "non-owned-runner-doppler is unavailable for public or fork-exposed repositories and hosted-runner exceptions", + ); + } + if (secretBearingIdentity && !ownedRunner) { + add("ci.runner", "Doppler secret injection requires an owned runner"); + } + } + + return errors; +} + +function validateManifest(doc, label) { + return [ + ...validateAgainstSchema(doc, policySchema, label), + ...validateManifestSemantics(doc, label), + ]; +} + +const ownedWorkflowTargets = new Set([ + "[self-hosted, Linux, X64, fleet-ci]", + "[self-hosted, macOS, ARM64, fleet-ci]", +]); + +function dopplerWorkflowPermitted({ runner, triggerTrust, repositoryVisibility, forkExposure }) { + return ( + ownedWorkflowTargets.has(runner) && + triggerTrust === "trusted-only" && + repositoryVisibility === "private" && + forkExposure === "none" + ); +} + +function workflowTriggerTrust(content) { + const lines = content.split(/\r?\n/); + const onIndex = lines.findIndex((line) => /^on:\s*/.test(line)); + if (onIndex === -1) { + return "unknown"; + } + const inline = lines[onIndex].replace(/^on:\s*/, "").trim(); + const allowed = new Set(["push", "workflow_dispatch"]); + const forkEvents = new Set(["pull_request", "pull_request_target"]); + const classify = (events) => { + if (events.some((event) => forkEvents.has(event))) { + return "fork-originated"; + } + return events.length > 0 && events.every((event) => allowed.has(event)) + ? "trusted-only" + : "unknown"; + }; + if (inline) { + if (inline.startsWith("[") && inline.endsWith("]")) { + return classify( + inline + .slice(1, -1) + .split(",") + .map((event) => event.trim().replace(/^['\"]|['\"]$/g, "")) + .filter(Boolean), + ); + } + return classify([inline.replace(/^['\"]|['\"]$/g, "")]); + } + const events = []; + for (let index = onIndex + 1; index < lines.length; index += 1) { + const line = lines[index]; + if (/^\S/.test(line) && line.trim() !== "") { + break; + } + if (line.trim() === "" || /^\s*#/.test(line)) { + continue; + } + const event = /^ ([A-Za-z0-9_-]+):\s*(?:#.*)?$/.exec(line); + if (!event) { + return "unknown"; + } + events.push(event[1]); + } + return classify(events); +} + +function workflowJobs(content) { + const lines = content.split(/\r?\n/); + const jobs = []; + const structuralErrors = []; + let inJobs = false; + let current = null; + lines.forEach((line) => { + if (/^jobs:\s*$/.test(line)) { + inJobs = true; + return; + } + if (!inJobs) { + return; + } + if (line.trim() === "" || /^\s*#/.test(line)) { + return; + } + const jobMatch = /^ ([A-Za-z0-9_-]+):\s*$/.exec(line); + if (jobMatch) { + current = { name: jobMatch[1], lines: [] }; + jobs.push(current); + return; + } + if (/^\S/.test(line)) { + inJobs = false; + current = null; + return; + } + if (/^ \S/.test(line)) { + structuralErrors.push(line.trim()); + current = null; + return; + } + if (current) { + current.lines.push(line); + } + }); + return { jobs, structuralErrors }; +} + +function validateTrackedWorkflows(workflowFiles, readTracked, manifest, projectLabel) { + const verdicts = []; + const verdictNames = new Set(); + const recordVerdict = (workflowFile, verdict) => { + if (verdict !== "permit-with-proof" && verdict !== "refuse") { + fail(`${projectLabel}/${workflowFile}: invalid workflow validation verdict`); + } + if (verdictNames.has(workflowFile)) { + fail(`${projectLabel}/${workflowFile}: duplicate workflow validation verdict`); + } + verdictNames.add(workflowFile); + verdicts.push({ workflowFile, verdict }); + }; + + workflowFiles.forEach((workflowFile) => { + const content = readTracked(workflowFile); + if (content === null) { + fail(`${projectLabel}/${workflowFile}: tracked workflow is unavailable for validation`); + recordVerdict(workflowFile, "refuse"); + return; + } + const parsed = workflowJobs(content); + const hasDopplerReference = + /dopplerhq\/secrets-fetch-action@|doppler-token:\s*|auth-method:\s*oidc\b/.test(content); + if (parsed.jobs.length === 0 || parsed.structuralErrors.length > 0) { + fail( + `${projectLabel}/${workflowFile}: workflow requires an unambiguous parsed jobs structure`, + ); + recordVerdict(workflowFile, "refuse"); + return; + } + let injectingJobs = 0; + let permitted = true; + const triggerTrust = workflowTriggerTrust(content); + parsed.jobs.forEach((job) => { + const jobText = job.lines.join("\n"); + const usesDoppler = /uses:\s*dopplerhq\/secrets-fetch-action@/.test(jobText); + const usesServiceToken = /doppler-token:\s*\$\{\{\s*secrets\.[^}]+\}\}/.test(jobText); + const usesOidc = /auth-method:\s*oidc\b/.test(jobText); + if (!usesDoppler || (!usesServiceToken && !usesOidc)) { + return; + } + injectingJobs += 1; + + const runnerMatches = job.lines.filter((line) => /^ runs-on:\s*/.test(line)); + const runner = runnerMatches.length === 1 ? runnerMatches[0].replace(/^ runs-on:\s*/, "").trim() : null; + if ( + !dopplerWorkflowPermitted({ + runner, + triggerTrust, + repositoryVisibility: manifest?.repositoryVisibility, + forkExposure: manifest?.forkExposure, + }) + ) { + permitted = false; + fail( + `${projectLabel}/${workflowFile} job ${job.name}: Doppler injection requires an owned runner, trusted-only trigger, private repository, and no fork exposure (runner=${runner ?? "unknown"} trigger=${triggerTrust} visibility=${manifest?.repositoryVisibility ?? "unknown"} forkExposure=${manifest?.forkExposure ?? "unknown"})`, + ); + } + }); + if (hasDopplerReference && injectingJobs === 0) { + permitted = false; + fail(`${projectLabel}/${workflowFile}: Doppler references were not confined to a parsed injecting job`); + } + recordVerdict(workflowFile, permitted ? "permit-with-proof" : "refuse"); + }); + + const expectedNames = new Set(workflowFiles); + if (expectedNames.size !== workflowFiles.length) { + fail(`${projectLabel}: authoritative workflow enumeration contains duplicates`); + } + if (verdicts.length !== workflowFiles.length || verdictNames.size !== expectedNames.size) { + fail(`${projectLabel}: workflow validation verdict set is incomplete`); + } + expectedNames.forEach((workflowFile) => { + if (!verdictNames.has(workflowFile)) { + fail(`${projectLabel}/${workflowFile}: authoritative workflow has no validation verdict`); + } + }); + return verdicts; +} + +function validateRollout() { + const rollout = readJson(rolloutFile); + if (!rollout) { + return 0; + } + validateAgainstSchema(rollout, rolloutSchema, rolloutFile).forEach(fail); + if (rollout.schemaVersion !== 1) { + fail(`${rolloutFile}: schemaVersion must equal 1`); + } + if (rollout.standard !== "doppler-default-v1") { + fail(`${rolloutFile}: standard must equal doppler-default-v1`); + } + const assessment = rollout.oidcAssessment; + if ( + !assessment || + assessment.decisionOwner !== "captain" || + assessment.status !== "recommendation-only" + ) { + fail(`${rolloutFile}: paid OIDC must remain a captain-owned recommendation only`); + } + if ( + assessment.recommendation !== "consider-oidc-to-remove-stored-doppler-ci-tokens" || + assessment.pricingCheckedOn !== "2026-07-27" || + assessment.teamPriceUsdPerUserMonth !== 21 || + assessment.enterprisePrice !== "custom" || + assessment.pricingUrl !== "https://www.doppler.com/pricing" + ) { + fail(`${rolloutFile}: OIDC assessment must record the dated public paid-plan cost`); + } + if (!nonEmptyString(assessment.reasoning) || !nonEmptyString(assessment.shippedDefault)) { + fail(`${rolloutFile}: OIDC assessment must include reasoning and the shipped default`); + } + if ( + !assessment.shippedDefault.includes("read-only service token") || + !assessment.shippedDefault.includes("no Team-plan") + ) { + fail(`${rolloutFile}: shipped default must remain service-token based and paid-plan independent`); + } + if (!Array.isArray(rollout.projects)) { + fail(`${rolloutFile}: projects must be an array`); + return 0; + } + + const ids = []; + const projectNames = new Set(); + rollout.projects.forEach((project, index) => { + validateManifestSemantics(project, `${rolloutFile}: projects[${index}]`).forEach(fail); + if (!nonEmptyString(project.backlogId)) { + fail(`${rolloutFile}: projects[${index}].backlogId must be a non-empty string`); + } else { + ids.push(project.backlogId); + } + if (projectNames.has(project.project)) { + fail(`${rolloutFile}: duplicate project ${project.project}`); + } + projectNames.add(project.project); + for (const field of ["currentState", "targetState"]) { + if (!nonEmptyString(project[field])) { + fail(`${rolloutFile}: projects[${index}].${field} must be a non-empty string`); + } + } + for (const field of ["rollout", "validation"]) { + if ( + !Array.isArray(project[field]) || + project[field].length === 0 || + project[field].some((item) => !nonEmptyString(item)) + ) { + fail(`${rolloutFile}: projects[${index}].${field} must contain non-empty strings`); + } + } + }); + + const actualIds = [...ids].sort(); + const expectedIds = [...expectedBacklogIds].sort(); + if (JSON.stringify(actualIds) !== JSON.stringify(expectedIds)) { + fail(`${rolloutFile}: backlog IDs must match the eight ci-selfhost migrations exactly`); + } + return rollout.projects.length; +} + +const leakRules = [ + ["private-key", /-----BEGIN (?:RSA |EC |OPENSSH |DSA )?PRIVATE KEY-----/], + ["doppler-service-token", /\bdp\.(?:st|sa|ct)\.[A-Za-z0-9._-]{12,}\b/], + ["github-token", /\b(?:gh[pousr]_[A-Za-z0-9]{20,}|github_pat_[A-Za-z0-9_]{20,})\b/], + ["aws-access-key", /\b(?:AKIA|ASIA)[A-Z0-9]{16}\b/], + ["stripe-live-key", /\b(?:sk|rk)_live_[A-Za-z0-9]{16,}\b/], + ["slack-token", /\bxox[baprs]-[A-Za-z0-9-]{20,}\b/], + [ + "credential-assignment", + /\b(?:DOPPLER_TOKEN|OP_SERVICE_ACCOUNT_TOKEN|GH_TOKEN|AWS_SECRET_ACCESS_KEY|GOOGLE_APPLICATION_CREDENTIALS)\s*=\s*(?!(?:\$\{|["']?\$|<|example|synthetic|redacted|REDACTED))[^"' \t\r\n]{12,}/, + ], +]; + +function scanFile(file) { + let content; + try { + content = fs.readFileSync(file, "utf8"); + } catch (error) { + fail(`${file}: cannot read: ${error.message}`); + return 0; + } + let findings = 0; + content.split(/\r?\n/).forEach((line, index) => { + leakRules.forEach(([name, pattern]) => { + if (pattern.test(line)) { + process.stderr.write(`${file}:${index + 1}: ${name}\n`); + findings += 1; + } + }); + }); + return findings; +} + +function leakScan(files) { + if (files.length === 0) { + fail("leak-scan requires at least one path"); + return; + } + let findings = 0; + files.forEach((file) => { + const resolved = path.resolve(file); + findings += scanFile(resolved); + }); + if (findings > 0) { + process.exitCode = 1; + } else { + process.stdout.write(`leak-scan: ok files=${files.length}\n`); + } +} + +function validatePolicyPointers() { + const skill = fs.readFileSync(policySkill, "utf8"); + const agents = fs.readFileSync(agentsFile, "utf8"); + const requiredPhrases = [ + "single owner of Firstmate's secrets-management policy", + "Doppler is the default source of truth", + "The runner host remains credential-free.", + "Config-scoped read-only service tokens are the shipped CI default.", + "Whether that benefit justifies the paid plan is the captain's decision.", + "Team at $21 per user per month", + "1Password is only the sealed break-glass store", + ]; + requiredPhrases.forEach((phrase) => { + if (!skill.includes(phrase)) { + fail(`${policySkill}: missing required contract phrase ${JSON.stringify(phrase)}`); + } + }); + const trigger = + "- `secrets-management` - load before project intake or initialization and before work that handles credentials or adds secret access to CI or deployment."; + if ((agents.match(new RegExp(trigger.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"), "g")) || []).length !== 1) { + fail(`${agentsFile}: secrets-management trigger must appear exactly once`); + } +} + +function validateJobTemplates() { + const actionPin = + "dopplerhq/secrets-fetch-action@cd2efbf9a404504316435873eff298b82f7e0562"; + const checkoutPin = + "actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd"; + const service = fs.readFileSync(serviceTokenTemplate, "utf8"); + const oidc = fs.readFileSync(oidcTemplate, "utf8"); + for (const [file, content] of [ + [serviceTokenTemplate, service], + [oidcTemplate, oidc], + ]) { + if (!content.includes("runs-on: [self-hosted, Linux, X64, fleet-ci]")) { + fail(`${file}: must use the complete owned Linux runner label set`); + } + if (!content.includes(actionPin)) { + fail(`${file}: must pin the audited Doppler fetch action commit`); + } + if (!content.includes("inject-env-vars: true")) { + fail(`${file}: must inject fetched values only into the job environment`); + } + if ( + !content.includes( + "if: github.event_name == 'push' || github.event_name == 'workflow_dispatch'", + ) + ) { + fail(`${file}: secret-bearing job must allow only explicit trusted event classes`); + } + const checkoutIndex = content.indexOf(`uses: ${checkoutPin}`); + const dopplerIndex = content.indexOf(actionPin); + if (checkoutIndex === -1 || checkoutIndex > dopplerIndex) { + fail(`${file}: trusted code checkout must precede Doppler secret injection`); + } + } + if (!service.includes("doppler-token: ${{ secrets.DOPPLER_TOKEN }}")) { + fail(`${serviceTokenTemplate}: must read the environment-scoped service token`); + } + if (/^env:\s*[\r\n]+\s+DOPPLER_TOKEN:/m.test(service)) { + fail(`${serviceTokenTemplate}: must not place DOPPLER_TOKEN in workflow or job env`); + } + for (const phrase of [ + "Optional paid-plan reference only", + "not the shipped default", + "id-token: write", + "auth-method: oidc", + "doppler-identity-id: ${{ vars.DOPPLER_SERVICE_IDENTITY_ID }}", + ]) { + if (!oidc.includes(phrase)) { + fail(`${oidcTemplate}: missing ${JSON.stringify(phrase)}`); + } + } +} + +function validateStandard() { + validateSchemaDocument(policySchema, "Firstmate Project Secrets Policy"); + validateSchemaDocument(rolloutSchema, "Firstmate Secrets Rollout"); + const example = readJson(exampleFile); + if (example) { + validateManifest(example, exampleFile).forEach(fail); + } + validatePolicyPointers(); + validateJobTemplates(); + const projects = validateRollout(); + leakScan(standardArtifacts); + if (!process.exitCode) { + process.stdout.write(`secrets-standard: ok projects=${projects}\n`); + } +} + +function validateOneManifest(file, today = policyToday()) { + const resolved = path.resolve(file); + const doc = readJson(resolved); + if (doc) { + validateManifestSemantics(doc, resolved, today).forEach(fail); + validateAgainstSchema(doc, policySchema, resolved).forEach(fail); + scanFile(resolved); + } + if (!process.exitCode) { + process.stdout.write(`manifest: ok ${resolved}\n`); + } +} + +function inventoryProject(directory) { + const resolved = path.resolve(directory); + let files; + try { + files = execFileSync("git", ["-C", resolved, "ls-files", "-z"], { + encoding: "utf8", + stdio: ["ignore", "pipe", "pipe"], + }) + .split("\0") + .filter(Boolean); + } catch { + fail(`${resolved}: inventory requires a git worktree`); + return; + } + const workflowFiles = files.filter((file) => /^\.github\/workflows\/.*\.ya?ml$/.test(file)); + let doppler = false; + let secretContext = false; + let oidc = false; + const readTracked = (file) => { + try { + return execFileSync("git", ["-C", resolved, "show", `HEAD:${file}`], { + encoding: "utf8", + stdio: ["ignore", "pipe", "ignore"], + maxBuffer: 4 * 1024 * 1024, + }); + } catch { + return null; + } + }; + files.forEach((file) => { + if (doppler && secretContext && oidc) { + return; + } + const content = readTracked(file); + if (content === null) return; + doppler ||= /doppler/i.test(content); + secretContext ||= /\bsecrets\.[A-Za-z_][A-Za-z0-9_]*/.test(content); + oidc ||= /id-token|workload_identity|oidc/i.test(content); + }); + let manifestDoc = null; + if (files.includes("docs/secrets-policy.json")) { + try { + manifestDoc = JSON.parse(readTracked("docs/secrets-policy.json")); + validateManifest(manifestDoc, `${resolved}/docs/secrets-policy.json`).forEach(fail); + } catch (error) { + fail(`${resolved}/docs/secrets-policy.json: invalid JSON: ${error.message}`); + } + } + validateTrackedWorkflows(workflowFiles, readTracked, manifestDoc, resolved); + const manifest = manifestDoc ? "present" : "absent"; + process.stdout.write( + `secrets-inventory: project=${path.basename(resolved)} workflows=${workflowFiles.length} doppler_refs=${doppler} secret_refs=${secretContext} oidc_refs=${oidc} manifest=${manifest}\n`, + ); +} + +const [command = "standard", ...args] = process.argv.slice(2); +switch (command) { + case "standard": + if (args.length !== 0) { + fail("standard accepts no arguments"); + } else { + validateStandard(); + } + break; + case "manifest": + if (args.length !== 1) { + fail("manifest requires exactly one path"); + } else { + validateOneManifest(args[0]); + } + break; + case "test-manifest-fixture": + if (args.length !== 2) { + fail("test-manifest-fixture requires a manifest path and fixed YYYY-MM-DD date"); + } else { + const today = parseIsoDate(args[1]); + if (!today) { + fail("test-manifest-fixture requires a valid fixed YYYY-MM-DD date"); + } else { + validateOneManifest(args[0], today); + } + } + break; + case "inventory": + if (args.length !== 1) { + fail("inventory requires exactly one project directory"); + } else { + inventoryProject(args[0]); + } + break; + case "leak-scan": + leakScan(args); + break; + default: + fail(`unknown command ${JSON.stringify(command)}`); +} diff --git a/bin/fm-secrets-check.sh b/bin/fm-secrets-check.sh new file mode 100755 index 00000000000..b43d1cfb090 --- /dev/null +++ b/bin/fm-secrets-check.sh @@ -0,0 +1,53 @@ +#!/usr/bin/env bash +# Validate Firstmate's tracked secrets standard and project manifests without +# reading or printing secret values. +# +# Usage: +# fm-secrets-check.sh +# fm-secrets-check.sh standard +# fm-secrets-check.sh manifest +# fm-secrets-check.sh test-manifest-fixture +# fm-secrets-check.sh inventory +# fm-secrets-check.sh leak-scan [path...] +# fm-secrets-check.sh --help +# +# The default "standard" command validates both JSON schemas, the project +# example, the exact eight-project rollout, policy ownership pointers, and a +# value-leak scan over every tracked standard artifact. +# +# "manifest" validates one project declaration against the deterministic +# contract represented by docs/secrets-policy.schema.json. +# +# "inventory" is a read-only project-intake helper. It reads git-tracked +# filenames, classifies Doppler, GitHub secret-context, and OIDC references, +# and rejects unsafe Doppler workflow runner targets. It never reads untracked +# files and never prints matched source lines. +# +# "leak-scan" reports only ":: "; matching text is never +# echoed. It is intentionally high-confidence and complements, rather than +# replaces, provider-side secret scanning. +set -eu + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" + +usage() { + awk ' + NR == 1 { next } + /^#/ { sub(/^# ?/, ""); print; next } + { exit } + ' "$0" >&2 +} + +case "${1:-}" in + -h|--help) + usage + exit 0 + ;; +esac + +command -v node >/dev/null 2>&1 || { + echo "fm-secrets-check: node is required" >&2 + exit 2 +} + +exec node "$SCRIPT_DIR/fm-secrets-check.mjs" "$@" diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index eb51827b8c2..f1c6b77a065 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -122,13 +122,13 @@ family_for_basename() { fm-composer-ghost.test.sh|fm-composer-lib.test.sh|\ fm-continuity-pretool-check.test.sh|fm-crew-state.test.sh|fm-decision-hold-lifecycle.test.sh|\ fm-dispatch-select.test.sh|fm-ensure-agents-md.test.sh|fm-grok-harness.test.sh|\ - fm-herdr-lab.test.sh|fm-instruction-owners.test.sh|fm-lint.test.sh|\ + fm-herdr-lab.test.sh|fm-instruction-owners.test.sh|fm-lint.test.sh|fm-secrets-check.test.sh|\ fm-install-herdr.test.sh|fm-nm-test-contract.test.sh|fm-no-mistakes-ownership.test.sh|\ fm-operational-input.test.sh|fm-pi-primary-types.test.sh|\ fm-send-popup-settle.test.sh|fm-send-settle.test.sh|fm-stow-contract.test.sh|\ fm-subagent-pretool-check.test.sh|\ fm-supervision-instructions.test.sh|fm-tmux-submit-busy.test.sh|fm-transition-lib.test.sh|\ - fm-test-run.test.sh|fm-test-isolation-proof.test.sh) + fm-test-run.test.sh|fm-test-isolation-proof.test.sh|fm-toolchain-mirror.test.sh) printf '%s\n' pure-contract-unit ;; fm-daemon.test.sh|fm-guard-stale-banner.test.sh|fm-pi-watch-extension.test.sh|\ @@ -678,7 +678,10 @@ families_for_changed_path() { # lane's contract coverage re-runs. printf '%s\n' real-herdr-gated ;; - bin/fm-lint.sh|bin/fm-install-shellcheck.sh|\ + bin/fm-toolchain-mirror.sh) + printf '%s\n' pure-contract-unit + ;; + bin/fm-lint.sh|bin/fm-install-shellcheck.sh|bin/fm-secrets-check.sh|bin/fm-secrets-check.mjs|\ bin/fm-brief.sh|bin/fm-ensure-agents-md.sh|bin/fm-crew-state.sh|\ bin/fm-decision-hold.sh|bin/fm-supervision*|bin/fm-transition-lib.sh|\ bin/fm-tmux-lib.sh|bin/fm-marker-lib.sh|bin/fm-operational-input.sh|bin/fm-tasks-axi-lib.sh|\ @@ -691,7 +694,8 @@ families_for_changed_path() { printf '%s\n' real-herdr-gated ;; docs/fm-test-portable-shards.md|docs/fm-test-isolation-proof.md|\ - docs/fm-test-isolation-proof.json) + docs/fm-test-isolation-proof.json|docs/secrets-*|docs/examples/project-secrets-policy.json|\ + docs/examples/doppler-*-job.yml|.agents/skills/secrets-management/SKILL.md) printf '%s\n' pure-contract-unit ;; .github/*|.tasks.toml|AGENTS.md|CLAUDE.md|CONTRIBUTING.md|\ diff --git a/bin/fm-toolchain-mirror.sh b/bin/fm-toolchain-mirror.sh new file mode 100755 index 00000000000..4515d80d657 --- /dev/null +++ b/bin/fm-toolchain-mirror.sh @@ -0,0 +1,457 @@ +#!/usr/bin/env bash +# fm-toolchain-mirror.sh - snapshot and restore the fleet-critical CLI toolchain. +# +# The mirror is platform-specific and contains complete installed npm package +# trees plus exact binary bytes for no-mistakes, Herdr, and Treehouse. +# Snapshot never updates a tool. +# Restore writes only to a new operator-selected prefix and never overwrites the +# ambient installation. +# +# Default mirror: +# $FM_HOME/data/toolchain-mirror +# +# Usage: +# fm-toolchain-mirror.sh snapshot [--mirror ] +# fm-toolchain-mirror.sh verify [--mirror ] [--snapshot ] +# fm-toolchain-mirror.sh restore [--mirror ] [--snapshot ] \ +# --prefix [--tool ] +# fm-toolchain-mirror.sh --help +# +# FM_TOOLCHAIN_MIRROR overrides the default mirror. +# FM_TOOLCHAIN_SNAPSHOT_ID provides a deterministic snapshot id for tests. +# FM_TOOLCHAIN_NPM_ROOT overrides `npm root -g` for isolated tests. +set -euo pipefail + +SELF_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +FM_ROOT="$(cd "$SELF_DIR/.." && pwd)" +FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" +MIRROR="${FM_TOOLCHAIN_MIRROR:-$FM_HOME/data/toolchain-mirror}" +SNAPSHOT= +PREFIX= +ONLY_TOOL= +TMP_PATH= + +die() { + printf 'fm-toolchain-mirror.sh: %s\n' "$*" >&2 + exit 1 +} + +usage() { + sed -n '2,20{s/^# \{0,1\}//;p;}' "$0" +} + +cleanup() { + if [ -n "$TMP_PATH" ] && [ -e "$TMP_PATH" ]; then + rm -rf -- "$TMP_PATH" + fi +} +trap cleanup EXIT + +sha256_file() { + if command -v sha256sum >/dev/null 2>&1; then + sha256sum "$1" | awk '{print $1}' + elif command -v shasum >/dev/null 2>&1; then + shasum -a 256 "$1" | awk '{print $1}' + else + die "need sha256sum or shasum to verify mirror artifacts" + fi +} + +validate_name() { + case "$2" in + ''|*[!A-Za-z0-9._-]*) die "$1 contains unsupported characters: $2" ;; + esac +} + +validate_tool() { + case "$1" in + tasks-axi|gh-axi|lavish-axi|quota-axi|chrome-devtools-axi|no-mistakes|herdr|treehouse) ;; + *) die "unsupported tool: $1" ;; + esac +} + +validate_npm_bin_metadata() { + metadata_path=$1 + tool=$2 + tab=$(printf '\t') + while IFS="$tab" read -r bin_name bin_relative || [ -n "${bin_name:-}" ]; do + validate_name "npm bin name" "$bin_name" + case "$bin_relative" in + ''|/*|*' '*) die "invalid npm bin target for $tool: $bin_relative" ;; + *"$tab"*) die "invalid npm bin target for $tool: $bin_relative" ;; + *'..'*) die "invalid npm bin target for $tool: $bin_relative" ;; + *[^A-Za-z0-9._/-]*) die "invalid npm bin target for $tool: $bin_relative" ;; + esac + case "$bin_relative" in + ../*|*/../*|*/..|.. ) die "invalid npm bin target for $tool: $bin_relative" ;; + esac + done < "$metadata_path" +} + +update_mechanism() { + case "$1" in + tasks-axi|gh-axi|lavish-axi|quota-axi|chrome-devtools-axi) + printf 'npm update -g %s\n' "$1" + ;; + no-mistakes) printf '%s\n' 'no-mistakes update' ;; + herdr) printf '%s\n' 'brew upgrade herdr (or herdr update)' ;; + treehouse) printf '%s\n' 'treehouse update' ;; + esac +} + +version_from_output() { + tool=$1 + output_file=$2 + case "$tool" in + tasks-axi|gh-axi|lavish-axi|quota-axi|chrome-devtools-axi) + tr -d '[:space:]' < "$output_file" + ;; + no-mistakes) + awk 'NR == 1 { for (i = 1; i <= NF; i++) if ($i ~ /^v[0-9]/) { print $i; exit } }' \ + "$output_file" + ;; + herdr) + awk 'NR == 1 && $1 == "herdr" { print $2; exit }' "$output_file" + ;; + treehouse) + tr -d '[:space:]' < "$output_file" + ;; + esac +} + +resolve_executable() { + perl -MCwd=abs_path -e \ + 'my $path = abs_path($ARGV[0]); defined $path or exit 1; print $path' "$1" +} + +parse_options() { + while [ "$#" -gt 0 ]; do + case "$1" in + --mirror) + [ "$#" -ge 2 ] || die "--mirror requires a directory" + MIRROR=$2 + shift 2 + ;; + --snapshot) + [ "$#" -ge 2 ] || die "--snapshot requires an id" + SNAPSHOT=$2 + shift 2 + ;; + --prefix) + [ "$#" -ge 2 ] || die "--prefix requires a new directory" + PREFIX=$2 + shift 2 + ;; + --tool) + [ "$#" -ge 2 ] || die "--tool requires a tool name" + ONLY_TOOL=$2 + shift 2 + ;; + --help|-h) + usage + exit 0 + ;; + *) + die "unknown option: $1" + ;; + esac + done +} + +mirror_guard() { + case "$MIRROR" in + ''|/) die "refusing unsafe mirror path: ${MIRROR:-}" ;; + esac +} + +resolve_snapshot() { + if [ -z "$SNAPSHOT" ]; then + [ -f "$MIRROR/current" ] || die "mirror has no current snapshot: $MIRROR" + IFS= read -r SNAPSHOT < "$MIRROR/current" || true + fi + validate_name "snapshot id" "$SNAPSHOT" + SNAPSHOT_DIR="$MIRROR/snapshots/$SNAPSHOT" + [ -d "$SNAPSHOT_DIR" ] || die "snapshot does not exist: $SNAPSHOT_DIR" + [ -f "$SNAPSHOT_DIR/manifest.tsv" ] || die "snapshot manifest is missing: $SNAPSHOT_DIR/manifest.tsv" +} + +snapshot_tool() { + tool=$1 + kind=$2 + command_path=$(command -v "$tool" 2>/dev/null) || die "$tool is not installed" + version_file="versions/$tool.txt" + if ! "$command_path" --version > "$TMP_PATH/$version_file" 2>&1; then + die "$tool --version failed" + fi + version=$(version_from_output "$tool" "$TMP_PATH/$version_file") + [ -n "$version" ] || die "could not parse $tool --version" + + if [ "$kind" = npm ]; then + package_dir="$NPM_ROOT/$tool" + [ -f "$package_dir/package.json" ] || die "installed npm package is missing: $package_dir" + package_version=$(node -p 'require(process.argv[1]).version' "$package_dir/package.json") + [ "$version" = "$package_version" ] \ + || die "$tool --version reported $version but package.json records $package_version" + artifact="artifacts/npm/$tool.tar.gz" + bin_metadata="metadata/npm/$tool.bin.tsv" + mkdir -p "$TMP_PATH/$(dirname "$bin_metadata")" + node - "$package_dir/package.json" > "$TMP_PATH/$bin_metadata" <<'NODE' +const fs = require('fs'); +const packagePath = process.argv[2]; +const packageJson = JSON.parse(fs.readFileSync(packagePath, 'utf8')); +const bin = typeof packageJson.bin === 'string' + ? { [packageJson.name]: packageJson.bin } + : packageJson.bin; +if (!bin || typeof bin !== 'object' || Array.isArray(bin)) { + process.exit(1); +} +for (const name of Object.keys(bin).sort()) { + if (typeof bin[name] !== 'string' || bin[name].length === 0) { + process.exit(1); + } + process.stdout.write(`${name}\t${bin[name]}\n`); +} +NODE + [ -s "$TMP_PATH/$bin_metadata" ] || die "installed npm package has no usable bin mapping: $tool" + validate_npm_bin_metadata "$TMP_PATH/$bin_metadata" "$tool" + tar -czf "$TMP_PATH/$artifact" -C "$NPM_ROOT" "$tool" + source="npm-global:$package_dir" + else + resolved=$(resolve_executable "$command_path") || die "could not resolve $command_path" + artifact="artifacts/bin/$tool" + bin_metadata=- + install -m 0755 "$resolved" "$TMP_PATH/$artifact" + source="binary:$resolved" + fi + + checksum=$(sha256_file "$TMP_PATH/$artifact") + mechanism=$(update_mechanism "$tool") + printf '%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\n' \ + "$tool" "$kind" "$version" "$version_file" "$artifact" "$checksum" "$source" "$mechanism" "$bin_metadata" \ + >> "$TMP_PATH/manifest.tsv" +} + +snapshot_action() { + [ -z "$SNAPSHOT" ] || die "--snapshot is not valid with snapshot" + [ -z "$PREFIX" ] || die "--prefix is not valid with snapshot" + [ -z "$ONLY_TOOL" ] || die "--tool is not valid with snapshot" + mirror_guard + command -v node >/dev/null 2>&1 || die "node is required to snapshot npm packages" + command -v npm >/dev/null 2>&1 || die "npm is required to locate global packages" + command -v perl >/dev/null 2>&1 || die "perl is required to resolve installed binaries" + + NPM_ROOT="${FM_TOOLCHAIN_NPM_ROOT:-$(npm root -g)}" + [ -d "$NPM_ROOT" ] || die "npm global root does not exist: $NPM_ROOT" + + snapshot_id="${FM_TOOLCHAIN_SNAPSHOT_ID:-$(date -u +%Y%m%dT%H%M%SZ)-$(uname -s)-$(uname -m)}" + validate_name "snapshot id" "$snapshot_id" + mkdir -p "$MIRROR/snapshots" + final="$MIRROR/snapshots/$snapshot_id" + [ ! -e "$final" ] || die "snapshot already exists: $final" + + TMP_PATH=$(mktemp -d "$MIRROR/.snapshot.XXXXXX") || die "could not create mirror staging directory" + mkdir -p "$TMP_PATH/artifacts/npm" "$TMP_PATH/artifacts/bin" "$TMP_PATH/versions" + printf 'tool\tkind\tversion\tversion_file\tartifact\tsha256\tinstall_source\tupdate_mechanism\tbin_metadata\n' \ + > "$TMP_PATH/manifest.tsv" + printf 'schema\tfm-toolchain-mirror.v1\nos\t%s\narch\t%s\n' "$(uname -s)" "$(uname -m)" \ + > "$TMP_PATH/platform.tsv" + + snapshot_tool tasks-axi npm + snapshot_tool gh-axi npm + snapshot_tool lavish-axi npm + snapshot_tool quota-axi npm + snapshot_tool chrome-devtools-axi npm + snapshot_tool no-mistakes binary + snapshot_tool herdr binary + snapshot_tool treehouse binary + + mv "$TMP_PATH" "$final" + TMP_PATH= + current_tmp=$(mktemp "$MIRROR/.current.XXXXXX") || die "could not create current marker" + TMP_PATH=$current_tmp + printf '%s\n' "$snapshot_id" > "$current_tmp" + mv "$current_tmp" "$MIRROR/current" + TMP_PATH= + + printf 'snapshot: %s\n' "$snapshot_id" + printf 'mirror: %s\n' "$MIRROR" + printf 'manifest: %s\n' "$final/manifest.tsv" + printf 'artifacts: 8\n' +} + +verify_archive_shape() { + tool=$1 + artifact_path=$2 + if ! tar -tzf "$artifact_path" | awk -v prefix="$tool/" ' + index($0, prefix) != 1 || $0 ~ /(^|\/)\.\.(\/|$)/ { bad = 1 } + END { exit bad } + '; then + die "npm archive has an unsafe or unexpected path: $artifact_path" + fi +} + +verify_manifest() { + manifest=$1 + expected_header='tool kind version version_file artifact sha256 install_source update_mechanism bin_metadata' + IFS= read -r actual_header < "$manifest" || true + [ "$actual_header" = "$expected_header" ] || die "snapshot manifest header is invalid" + expected_schema=$(awk -F '\t' '$1 == "schema" { print $2; exit }' "$SNAPSHOT_DIR/platform.tsv") + [ "$expected_schema" = "fm-toolchain-mirror.v1" ] \ + || die "unsupported snapshot schema: ${expected_schema:-}" + expected_os=$(awk -F '\t' '$1 == "os" { print $2; exit }' "$SNAPSHOT_DIR/platform.tsv") + expected_arch=$(awk -F '\t' '$1 == "arch" { print $2; exit }' "$SNAPSHOT_DIR/platform.tsv") + [ "$expected_os" = "$(uname -s)" ] \ + || die "snapshot OS $expected_os does not match $(uname -s)" + [ "$expected_arch" = "$(uname -m)" ] \ + || die "snapshot architecture $expected_arch does not match $(uname -m)" + + found=0 + seen=' ' + tab=$(printf '\t') + while IFS="$tab" read -r tool kind version version_file artifact checksum source mechanism bin_metadata \ + || [ -n "${tool:-}" ]; do + [ "$tool" != tool ] || continue + [ -n "$tool" ] || continue + validate_tool "$tool" + case "$seen" in + *" $tool "*) die "snapshot manifest contains duplicate tool: $tool" ;; + esac + seen="$seen$tool " + if [ -n "$ONLY_TOOL" ] && [ "$tool" != "$ONLY_TOOL" ]; then + continue + fi + [ "$version_file" = "versions/$tool.txt" ] \ + || die "unexpected version output path for $tool: $version_file" + case "$kind" in + npm) + expected_artifact="artifacts/npm/$tool.tar.gz" + expected_bin_metadata="metadata/npm/$tool.bin.tsv" + [ "$bin_metadata" = "$expected_bin_metadata" ] \ + || die "unexpected npm bin metadata path for $tool: $bin_metadata" + [ -f "$SNAPSHOT_DIR/$bin_metadata" ] \ + || die "npm bin metadata is missing for $tool" + ;; + binary) + expected_artifact="artifacts/bin/$tool" + [ "$bin_metadata" = "-" ] || die "binary tool has unexpected bin metadata: $tool" + ;; + *) die "unsupported artifact kind for $tool: $kind" ;; + esac + [ "$artifact" = "$expected_artifact" ] \ + || die "unexpected artifact path for $tool: $artifact" + artifact_path="$SNAPSHOT_DIR/$artifact" + [ -f "$artifact_path" ] || die "artifact is missing: $artifact_path" + actual=$(sha256_file "$artifact_path") + [ "$actual" = "$checksum" ] \ + || die "checksum mismatch for $tool (expected $checksum, got $actual)" + output_path="$SNAPSHOT_DIR/$version_file" + [ -f "$output_path" ] || die "version output is missing for $tool" + recorded_version=$(version_from_output "$tool" "$output_path") + [ "$recorded_version" = "$version" ] \ + || die "version output for $tool reports $recorded_version, expected $version" + case "$kind" in + npm) + verify_archive_shape "$tool" "$artifact_path" + validate_npm_bin_metadata "$SNAPSHOT_DIR/$bin_metadata" "$tool" + ;; + binary) ;; + esac + [ -n "$version" ] && [ -n "$source" ] && [ -n "$mechanism" ] \ + || die "manifest metadata is incomplete for $tool" + found=$((found + 1)) + done < "$manifest" + if [ -n "$ONLY_TOOL" ]; then + [ "$found" -eq 1 ] || die "snapshot does not contain requested tool: $ONLY_TOOL" + else + [ "$found" -eq 8 ] || die "snapshot contains $found tools; expected all 8" + fi + VERIFIED_COUNT=$found +} + +verify_action() { + [ -z "$PREFIX" ] || die "--prefix is not valid with verify" + if [ -n "$ONLY_TOOL" ]; then + validate_tool "$ONLY_TOOL" + fi + mirror_guard + resolve_snapshot + [ -f "$SNAPSHOT_DIR/platform.tsv" ] || die "snapshot platform metadata is missing" + verify_manifest "$SNAPSHOT_DIR/manifest.tsv" + printf 'verified: %s (%s artifacts)\n' "$SNAPSHOT_DIR" "$VERIFIED_COUNT" +} + +restore_action() { + [ -n "$PREFIX" ] || die "restore requires --prefix " + case "$PREFIX" in + ''|/) die "refusing unsafe restore prefix: ${PREFIX:-}" ;; + esac + [ ! -e "$PREFIX" ] || die "restore prefix already exists; choose a new directory: $PREFIX" + if [ -n "$ONLY_TOOL" ]; then + validate_tool "$ONLY_TOOL" + fi + mirror_guard + resolve_snapshot + [ -f "$SNAPSHOT_DIR/platform.tsv" ] || die "snapshot platform metadata is missing" + verify_manifest "$SNAPSHOT_DIR/manifest.tsv" + + prefix_parent=$(dirname "$PREFIX") + mkdir -p "$prefix_parent" + TMP_PATH=$(mktemp -d "$prefix_parent/.fm-toolchain-restore.XXXXXX") \ + || die "could not create restore staging directory" + mkdir -p "$TMP_PATH/bin" "$TMP_PATH/lib/node_modules" + + tab=$(printf '\t') + restored=0 + while IFS="$tab" read -r tool kind version version_file artifact checksum source mechanism bin_metadata \ + || [ -n "${tool:-}" ]; do + [ "$tool" != tool ] || continue + [ -n "$tool" ] || continue + if [ -n "$ONLY_TOOL" ] && [ "$tool" != "$ONLY_TOOL" ]; then + continue + fi + artifact_path="$SNAPSHOT_DIR/$artifact" + if [ "$kind" = npm ]; then + tar -xzf "$artifact_path" -C "$TMP_PATH/lib/node_modules" + bin_count=0 + while IFS="$tab" read -r bin_name bin_relative || [ -n "${bin_name:-}" ]; do + [ -f "$TMP_PATH/lib/node_modules/$tool/$bin_relative" ] \ + || die "restored npm package lacks $bin_relative: $tool" + [ ! -e "$TMP_PATH/bin/$bin_name" ] || die "duplicate restored npm bin: $bin_name" + ln -s "../lib/node_modules/$tool/$bin_relative" "$TMP_PATH/bin/$bin_name" + [ "$bin_name" = "$tool" ] && bin_count=$((bin_count + 1)) + done < "$SNAPSHOT_DIR/$bin_metadata" + [ "$bin_count" -eq 1 ] || die "npm package has no unique $tool bin mapping" + else + install -m 0755 "$artifact_path" "$TMP_PATH/bin/$tool" + fi + if ! "$TMP_PATH/bin/$tool" --version > "$TMP_PATH/$tool.version" 2>&1; then + die "restored $tool --version failed" + fi + restored_version=$(version_from_output "$tool" "$TMP_PATH/$tool.version") + [ "$restored_version" = "$version" ] \ + || die "restored $tool reported $restored_version, expected $version" + rm "$TMP_PATH/$tool.version" + restored=$((restored + 1)) + done < "$SNAPSHOT_DIR/manifest.tsv" + + [ "$restored" -eq "$VERIFIED_COUNT" ] || die "restored $restored of $VERIFIED_COUNT verified artifacts" + mv "$TMP_PATH" "$PREFIX" + TMP_PATH= + printf 'restored: %s (%s tools)\n' "$PREFIX" "$restored" + printf 'activate: prepend %s/bin to PATH\n' "$PREFIX" +} + +ACTION=${1:-} +case "$ACTION" in + snapshot|verify|restore) + shift + parse_options "$@" + "${ACTION}_action" + ;; + --help|-h|'') + usage + ;; + *) + die "unknown action: $ACTION" + ;; +esac diff --git a/docs/configuration.md b/docs/configuration.md index 168054665c6..d669972b0cb 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -247,6 +247,7 @@ On session start the first mate detects what its required toolchain is missing o It installs automatically supported tools only after you say go; manual-only tools remain for you to install from the printed instructions. Required tools come in two parts: a universal toolchain every home needs regardless of backend, and a per-backend delta that follows the runtime backend actually resolved for this home. The universal toolchain is node, git, gh with GitHub auth via `gh auth login`, no-mistakes v1.31.2 or newer, gh-axi, chrome-devtools-axi, lavish-axi, compatible tasks-axi per "Backlog backend" above, and quota-axi. +The observed fleet version manifest and checksummed offline mirror recovery procedure are in [`docs/toolchain-versions.md`](toolchain-versions.md), while `bin/fm-toolchain-mirror.sh` owns the snapshot and restore mechanics. This section is the single owner of that universal toolchain list; backend guides' prerequisites point here and add only their backend-specific tools. In that list, no-mistakes runs the validation pipeline, gh-axi, chrome-devtools-axi, and lavish-axi cover GitHub, browser, and rich-review operations, and tasks-axi plus quota-axi back backlog mutations and quota-aware array dispatch. The per-backend delta is required only for the backend resolved from `FM_BACKEND`, then `config/backend`, then runtime auto-detection, then default `tmux`, so a home is never told to install a tool an inactive backend or feature would need. diff --git a/docs/examples/doppler-oidc-job.yml b/docs/examples/doppler-oidc-job.yml new file mode 100644 index 00000000000..2fa2267fb4f --- /dev/null +++ b/docs/examples/doppler-oidc-job.yml @@ -0,0 +1,33 @@ +# Optional paid-plan reference only; this is not the shipped default. +# DOPPLER_SERVICE_IDENTITY_ID is a non-secret GitHub Environment variable. +# Use only after the captain approves the plan and a separate OIDC migration. +# Bind the identity to this repository, environment, workflow, and self-hosted +# runner claim with exact values before enabling the job. +permissions: + contents: read + +jobs: + deploy: + if: github.event_name == 'push' || github.event_name == 'workflow_dispatch' + runs-on: [self-hosted, Linux, X64, fleet-ci] + environment: staging + permissions: + contents: read + id-token: write + steps: + - name: Check out trusted repository code before secret injection + uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + + - name: Exchange the GitHub job identity and fetch masked Doppler secrets + id: doppler + uses: dopplerhq/secrets-fetch-action@cd2efbf9a404504316435873eff298b82f7e0562 # v1.3.0 + with: + auth-method: oidc + doppler-identity-id: ${{ vars.DOPPLER_SERVICE_IDENTITY_ID }} + doppler-project: example-service + doppler-config: stg + inject-env-vars: true + + - name: Run the only command that consumes the injected values + shell: bash + run: ./scripts/deploy.sh diff --git a/docs/examples/doppler-service-token-job.yml b/docs/examples/doppler-service-token-job.yml new file mode 100644 index 00000000000..9d0ef5e9d21 --- /dev/null +++ b/docs/examples/doppler-service-token-job.yml @@ -0,0 +1,26 @@ +# Current-plan job template. +# DOPPLER_TOKEN is a GitHub Environment secret containing one read-only Doppler +# service token scoped to the exact project config used by this job. +# Do not move it to workflow env, job env, runner service env, or runner home. +permissions: + contents: read + +jobs: + deploy: + if: github.event_name == 'push' || github.event_name == 'workflow_dispatch' + runs-on: [self-hosted, Linux, X64, fleet-ci] + environment: staging + steps: + - name: Check out trusted repository code before secret injection + uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + + - name: Fetch and mask Doppler secrets for this job + id: doppler + uses: dopplerhq/secrets-fetch-action@cd2efbf9a404504316435873eff298b82f7e0562 # v1.3.0 + with: + doppler-token: ${{ secrets.DOPPLER_TOKEN }} + inject-env-vars: true + + - name: Run the only command that consumes the injected values + shell: bash + run: ./scripts/deploy.sh diff --git a/docs/examples/project-secrets-policy.json b/docs/examples/project-secrets-policy.json new file mode 100644 index 00000000000..0dd6b50bc52 --- /dev/null +++ b/docs/examples/project-secrets-policy.json @@ -0,0 +1,27 @@ +{ + "$schema": "../secrets-policy.schema.json", + "schemaVersion": 1, + "project": "example-service", + "repository": "example/example-service", + "repositoryVisibility": "private", + "forkExposure": "none", + "classification": "doppler", + "doppler": { + "project": "example-service", + "configs": [ + "dev", + "stg", + "prd" + ] + }, + "ci": { + "runner": "owned-linux", + "identity": "doppler-service-token", + "injection": "per-job-doppler-fetch" + }, + "exceptions": [], + "review": { + "owner": "project-maintainer", + "cadenceDays": 90 + } +} diff --git a/docs/promotion-ladder.md b/docs/promotion-ladder.md index ddf71a5298e..8fd225c0aee 100644 --- a/docs/promotion-ladder.md +++ b/docs/promotion-ladder.md @@ -37,8 +37,9 @@ The exception ends as soon as the project gains a deployable app or persistent c ## Secrets separation -Each environment's secrets and config stay separate: dev/preview, staging, and production never share a secret set or a data store. -Use Doppler where the project has adopted it; use the platform's native secret store otherwise (GitHub Environments secrets, `wrangler secret`, OpenTofu's provider-native secret backend, and so on). +Production secrets and data stay isolated from every non-production environment. +Preview may share a staging secret set only when it is isolated from production data and the project manifest documents that choice. +The [Firstmate secrets-management policy](../.agents/skills/secrets-management/SKILL.md) owns the secretless-identity preference, Doppler default, platform-mandated exceptions, and per-job injection contract. No secret value ever lands in the repo, in a committed `tfvars`/`wrangler.toml`, or in a workflow file. ## Smoke gate @@ -68,7 +69,7 @@ Auto-apply staging's plan on merge to `staging`; promote the reviewed configurat Never apply a staging plan artifact to production or apply a fresh unreviewed plan. Inject variables and secrets via Doppler or the cloud provider's native secret store, never as committed `tfvars`. -**GitHub Actions deploy jobs** - use GitHub Environments (`staging`, `production`) with environment-scoped secrets. +**GitHub Actions deploy jobs** - use GitHub Environments (`staging`, `production`) for approvals and the environment-scoped identity declared by the secrets-management policy. Give the `production` environment required reviewers where the reviewed `staging`-to-`main` PR is not itself the platform's approval gate. Run the `staging` deploy job automatically on push to `staging`, and run production only after explicit promotion to `main`. diff --git a/docs/scripts.md b/docs/scripts.md index 5b76d808600..0748e5d6aca 100644 --- a/docs/scripts.md +++ b/docs/scripts.md @@ -22,10 +22,12 @@ The shared no-mistakes gate refusal for fleet lifecycle entrypoints is summarize | `fm-herdr-lab.sh` | Provision and guardedly operate an isolated, never-default Herdr lab session | | `fm-install-herdr.sh` | Install CI's exact-version Herdr pin with official asset URL, SHA-256, and protocol checks | | `fm-install-treehouse.sh`| Install CI's exact-version Treehouse pin for real-Herdr E2E that needs spawn worktrees | +| `fm-toolchain-mirror.sh` | Snapshot and restore the installed fleet-critical CLI toolchain from a checksummed local offline mirror | | `fm-herdr-ci-cleanup.sh` | Snapshot and tear down only job-owned `fm-lab-*` sessions in the Herdr CI lane | | `fm-test-run.sh` | Behavior-test runner: selection, portable lanes, proven-isolated `--jobs`, coverage guard, timing/JSON | | `fm-test-isolation-proof.sh` | Phase 2 concurrent isolation proof and proven-isolated candidate set owner | | `fm-ensure-agents-md.sh` | Ensure a project's real `AGENTS.md`, its `CLAUDE.md` symlink, and the canonical self-governance section | +| `fm-secrets-check.sh` | Validate the Doppler standard, project manifests, value-safe inventories, and high-confidence leak rules | | `fm-guard.sh` | Warn on primary-checkout tangles, pending queued wakes, and stale watcher liveness | | `fm-primary-scope-lib.sh` | Shared marker-or-plain-checkout primary-home predicate for tracked hooks | | `fm-turnend-guard.sh` | Shared primary turn-end guard predicate so no turn ends blind (docs/turnend-guard.md) | diff --git a/docs/secrets-policy.schema.json b/docs/secrets-policy.schema.json new file mode 100644 index 00000000000..3f3f9bace83 --- /dev/null +++ b/docs/secrets-policy.schema.json @@ -0,0 +1,313 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/juniorlovestmh/firstmate/blob/main/docs/secrets-policy.schema.json", + "title": "Firstmate Project Secrets Policy", + "description": "Value-free declaration of one project's secrets posture under the Firstmate Doppler standard.", + "type": "object", + "required": [ + "schemaVersion", + "project", + "repository", + "repositoryVisibility", + "forkExposure", + "classification", + "doppler", + "ci", + "exceptions", + "review" + ], + "properties": { + "$schema": { + "type": "string", + "minLength": 1 + }, + "schemaVersion": { + "const": 1 + }, + "project": { + "type": "string", + "minLength": 1, + "pattern": "^[a-z0-9][a-z0-9-]*$" + }, + "repository": { + "type": "string", + "minLength": 3, + "pattern": "^[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+$" + }, + "repositoryVisibility": { + "type": "string", + "enum": [ + "private", + "public" + ], + "description": "Public repositories cannot use non-owned-runner-doppler overrides." + }, + "forkExposure": { + "type": "string", + "enum": [ + "none", + "pull-request", + "pull-request-target" + ], + "description": "Any fork-exposed trigger cannot use non-owned-runner-doppler overrides." + }, + "classification": { + "enum": [ + "doppler", + "secretless", + "platform-native" + ] + }, + "doppler": { + "type": "object", + "required": [ + "project", + "configs" + ], + "properties": { + "project": { + "type": [ + "string", + "null" + ], + "minLength": 1 + }, + "configs": { + "type": "array", + "uniqueItems": true, + "items": { + "enum": [ + "dev", + "stg", + "prd" + ] + } + } + }, + "additionalProperties": false + }, + "ci": { + "type": "object", + "required": [ + "runner", + "identity", + "injection" + ], + "properties": { + "runner": { + "enum": [ + "owned-linux", + "owned-macos", + "github-hosted-public-fork", + "none" + ] + }, + "identity": { + "enum": [ + "none", + "doppler-service-token", + "doppler-oidc", + "provider-oidc", + "github-job-token", + "mixed" + ] + }, + "injection": { + "enum": [ + "none", + "per-job-doppler-fetch", + "job-oidc", + "github-job-context", + "mixed" + ] + } + }, + "additionalProperties": false + }, + "exceptions": { + "type": "array", + "items": { + "type": "object", + "required": [ + "kind", + "scope", + "reason" + ], + "properties": { + "kind": { + "enum": [ + "secretless", + "provider-oidc", + "platform-mandated", + "break-glass", + "hosted-runner", + "shared-nonproduction-config", + "temporary-migration", + "non-owned-runner-doppler" + ] + }, + "scope": { + "type": "string", + "minLength": 1 + }, + "reason": { + "type": "string", + "minLength": 1 + }, + "reviewBy": { + "type": "string", + "format": "date" + }, + "removalCondition": { + "type": "string", + "minLength": 1 + }, + "workflow": { + "type": "string", + "minLength": 1 + } + }, + "allOf": [ + { + "if": { + "properties": { + "kind": { + "const": "temporary-migration" + } + } + }, + "then": { + "required": [ + "reviewBy", + "removalCondition" + ] + } + }, + { + "if": { + "properties": { + "kind": { + "const": "non-owned-runner-doppler" + } + } + }, + "then": { + "required": [ + "workflow" + ] + } + } + ], + "additionalProperties": false + } + }, + "review": { + "type": "object", + "required": [ + "owner", + "cadenceDays" + ], + "properties": { + "owner": { + "type": "string", + "minLength": 1 + }, + "cadenceDays": { + "type": "integer", + "minimum": 1, + "maximum": 90 + } + }, + "additionalProperties": false + } + }, + "allOf": [ + { + "if": { + "required": [ + "classification" + ], + "properties": { + "classification": { + "const": "secretless" + } + } + }, + "then": { + "properties": { + "doppler": { + "properties": { + "project": { + "const": null + }, + "configs": { + "maxItems": 0 + } + } + } + } + } + }, + { + "if": { + "required": [ + "classification" + ], + "properties": { + "classification": { + "const": "platform-native" + } + } + }, + "then": { + "properties": { + "exceptions": { + "contains": { + "type": "object", + "required": [ + "kind" + ], + "properties": { + "kind": { + "const": "platform-mandated" + } + } + } + } + } + } + }, + { + "if": { + "required": [ + "ci" + ], + "properties": { + "ci": { + "type": "object", + "required": [ + "identity" + ], + "properties": { + "identity": { + "const": "doppler-service-token" + } + } + } + } + }, + "then": { + "properties": { + "ci": { + "properties": { + "injection": { + "const": "per-job-doppler-fetch" + } + } + } + } + } + } + ], + "additionalProperties": false +} diff --git a/docs/secrets-rollout.json b/docs/secrets-rollout.json new file mode 100644 index 00000000000..e7de84983c2 --- /dev/null +++ b/docs/secrets-rollout.json @@ -0,0 +1,393 @@ +{ + "$schema": "./secrets-rollout.schema.json", + "schemaVersion": 1, + "standard": "doppler-default-v1", + "evaluatedOn": "2026-07-27", + "oidcAssessment": { + "decisionOwner": "captain", + "status": "recommendation-only", + "recommendation": "consider-oidc-to-remove-stored-doppler-ci-tokens", + "pricingCheckedOn": "2026-07-27", + "teamPriceUsdPerUserMonth": 21, + "enterprisePrice": "custom", + "pricingUrl": "https://www.doppler.com/pricing", + "reasoning": "The owned-runner pilot proved GitHub job OIDC and a credential-free runner can coexist. Short-lived Doppler identity tokens would remove the stored Doppler CI token and its rotation burden, but Doppler restricts service-account identities to paid plans, so the recommendation and cost are recorded for the captain rather than treated as implementation authority.", + "shippedDefault": "Keep one read-only service token per Doppler project config in the matching GitHub Environment and inject it only into the Doppler fetch step. The standard has no Team-plan, Enterprise-plan, service-account, or OIDC dependency." + }, + "projects": [ + { + "schemaVersion": 1, + "backlogId": "ci-selfhost-appheat-site-h1", + "project": "appheat-site", + "repository": "appheat/appheat-site", + "repositoryVisibility": "private", + "forkExposure": "none", + "classification": "doppler", + "doppler": { + "project": "appheat", + "configs": [ + "dev", + "prd" + ] + }, + "ci": { + "runner": "owned-linux", + "identity": "doppler-service-token", + "injection": "per-job-doppler-fetch" + }, + "exceptions": [ + { + "kind": "temporary-migration", + "scope": "production deploy currently selects appheat/dev", + "reason": "The tracked production workflow still selects the development config even though the project already documents prd, so the runner migration must switch the job to prd before retiring the old path.", + "reviewBy": "2026-10-25", + "removalCondition": "Remove this exception when the production workflow selects appheat/prd and a deploy plus rollback check pass." + } + ], + "review": { + "owner": "appheat-site maintainers", + "cadenceDays": 90 + }, + "currentState": "Verify is partly routed by expression and deploy runs on a hosted runner with a config-scoped Doppler token.", + "targetState": "Every job runs on owned Linux, production reads appheat/prd, and the token reaches only the pinned Doppler fetch step.", + "rollout": [ + "Register or authorize the repository for the fleet-ci runner and verify the workflow toolchain before changing labels.", + "Create the production GitHub Environment service-token binding through an approved value-handling path and leave the runner host empty.", + "Replace the CLI token pattern with the pinned masking fetch action and select appheat/prd for production.", + "Remove the development-config production path after a real deploy and rollback check pass." + ], + "validation": [ + "Run the project manifest checker and tracked leak scan.", + "Prove preflight passes before the job and a production deploy consumes only appheat/prd without value output.", + "Prove the next admitted job has no ambient Doppler credential." + ] + }, + { + "schemaVersion": 1, + "backlogId": "ci-selfhost-bible-agents-h2", + "project": "bible-agents", + "repository": "appheat/bible-agents", + "repositoryVisibility": "private", + "forkExposure": "none", + "classification": "doppler", + "doppler": { + "project": "agentbib", + "configs": [ + "dev", + "stg", + "prd" + ] + }, + "ci": { + "runner": "owned-linux", + "identity": "doppler-service-token", + "injection": "per-job-doppler-fetch" + }, + "exceptions": [ + { + "kind": "platform-mandated", + "scope": "Cloudflare Worker runtime secret store", + "reason": "Cloudflare must hold runtime bindings after deployment, while Doppler remains the source of truth and the workflow performs the environment-specific sync." + }, + { + "kind": "shared-nonproduction-config", + "scope": "preview jobs use agentbib/stg", + "reason": "Preview and persistent staging deliberately share sandbox credentials and no production data, so this remains allowed until per-preview configs have a demonstrated need." + } + ], + "review": { + "owner": "bible-agents maintainers", + "cadenceDays": 90 + }, + "currentState": "Most deploy jobs already use fleet-ci with stg or prd service tokens, while one job still requests a hosted runner.", + "targetState": "All jobs use owned Linux and each secret-bearing GitHub Environment injects only its matching agentbib config token.", + "rollout": [ + "Move the remaining hosted job to fleet-ci without changing its functional steps.", + "Replace repeated CLI token handling with the pinned masking fetch action where the job consumes secret values.", + "Keep preview and staging on agentbib/stg and production on agentbib/prd with separate GitHub Environment tokens.", + "Add always-run cleanup for every generated Worker secret file under RUNNER_TEMP." + ], + "validation": [ + "Run the project manifest checker and tracked leak scan.", + "Prove staging, preview, and production select only their declared config without printing fetched values.", + "Prove generated secret files are absent after success and failure." + ] + }, + { + "schemaVersion": 1, + "backlogId": "ci-selfhost-boostin-h3", + "project": "boostin", + "repository": "juniorlovestmh/boostin", + "repositoryVisibility": "private", + "forkExposure": "none", + "classification": "secretless", + "doppler": { + "project": null, + "configs": [] + }, + "ci": { + "runner": "owned-linux", + "identity": "none", + "injection": "none" + }, + "exceptions": [ + { + "kind": "secretless", + "scope": "current verify workflow", + "reason": "The local-first import-only verification path uses deterministic fixtures and requires no external credential, so creating a Doppler project would add a liability without a consumer." + } + ], + "review": { + "owner": "boostin maintainers", + "cadenceDays": 90 + }, + "currentState": "The verify workflow runs on a hosted Linux runner and references no GitHub or Doppler secret.", + "targetState": "The same secretless verification runs on owned Linux with no credential injection.", + "rollout": [ + "Confirm every workflow path remains synthetic and secretless.", + "Move the verify job to the approved owned-runner labels without adding a vault or token.", + "Reclassify before any future integration introduces network credentials." + ], + "validation": [ + "Run the project manifest checker and tracked leak scan.", + "Prove the real verify workflow passes on owned Linux with no secret context reference." + ] + }, + { + "schemaVersion": 1, + "backlogId": "ci-selfhost-resume-matcher-h4", + "project": "resume-matcher", + "repository": "juniorlovestmh/resume-matcher", + "repositoryVisibility": "private", + "forkExposure": "none", + "classification": "doppler", + "doppler": { + "project": "resume-matcher", + "configs": [ + "dev", + "prd" + ] + }, + "ci": { + "runner": "owned-linux", + "identity": "mixed", + "injection": "mixed" + }, + "exceptions": [ + { + "kind": "platform-mandated", + "scope": "GitHub Container Registry authentication", + "reason": "The job-scoped GitHub token is short-lived, automatically issued by GitHub, and removes the need to create or store a separate registry credential." + } + ], + "review": { + "owner": "resume-matcher maintainers", + "cadenceDays": 90 + }, + "currentState": "Container publishing uses hosted Linux, a GitHub job token for GHCR, and repository secrets for Docker Hub.", + "targetState": "Publishing uses owned Linux, keeps the short-lived GitHub job token for GHCR, and fetches Docker Hub credentials from resume-matcher/prd through one read-only Doppler service token.", + "rollout": [ + "Confirm Docker and Buildx are available on the owned runner before changing labels.", + "Move Docker Hub credentials into resume-matcher/prd through an approved value-handling path.", + "Inject the prd service token only into the pinned masking fetch action and leave the GitHub job token unchanged.", + "Remove the superseded Docker Hub repository secrets only after both registry pushes pass." + ], + "validation": [ + "Run the project manifest checker and tracked leak scan.", + "Prove both registry logins and a real image publish succeed without credential output.", + "Prove a pull request from an untrusted source receives neither publishing path." + ] + }, + { + "schemaVersion": 1, + "backlogId": "ci-selfhost-factory-h5", + "project": "factory", + "repository": "appheat/factory-ai", + "repositoryVisibility": "private", + "forkExposure": "none", + "classification": "secretless", + "doppler": { + "project": null, + "configs": [] + }, + "ci": { + "runner": "owned-linux", + "identity": "github-job-token", + "injection": "github-job-context" + }, + "exceptions": [ + { + "kind": "secretless", + "scope": "build, test, and current package-release inputs", + "reason": "The tracked workflows consume no external long-lived secret, so a Doppler project is unnecessary until a real secret-bearing integration appears." + }, + { + "kind": "platform-mandated", + "scope": "GitHub release creation", + "reason": "The release job uses GitHub's short-lived built-in token with contents-only permissions instead of storing a replacement credential." + } + ], + "review": { + "owner": "factory maintainers", + "cadenceDays": 90 + }, + "currentState": "CI and package workflows run on hosted Linux and the release job uses only GitHub's built-in token.", + "targetState": "All jobs use owned Linux and retain the secretless or short-lived platform identity path.", + "rollout": [ + "Verify Node, pnpm, and release tooling on the owned runner before changing labels.", + "Move all jobs to owned Linux without creating a Doppler project.", + "Prefer trusted publishing if a future external package registry supports OIDC." + ], + "validation": [ + "Run the project manifest checker and tracked leak scan.", + "Prove CI, artifact handoff, and release creation pass with the declared GitHub permissions only." + ] + }, + { + "schemaVersion": 1, + "backlogId": "ci-selfhost-organicops-h6", + "project": "organicops", + "repository": "appheat/organicops", + "repositoryVisibility": "private", + "forkExposure": "none", + "classification": "doppler", + "doppler": { + "project": "organicops", + "configs": [ + "dev", + "stg", + "prd" + ] + }, + "ci": { + "runner": "owned-linux", + "identity": "provider-oidc", + "injection": "job-oidc" + }, + "exceptions": [ + { + "kind": "provider-oidc", + "scope": "Google Cloud deployment identity", + "reason": "Workload Identity Federation gives the job a short-lived Google identity and is safer than storing a cloud key in Doppler or GitHub." + }, + { + "kind": "platform-mandated", + "scope": "Google Cloud runtime secret attachment", + "reason": "Cloud Run consumes named Google Secret Manager entries at runtime, while Doppler remains the project source for application and provider secrets that require managed values." + } + ], + "review": { + "owner": "organicops maintainers", + "cadenceDays": 90 + }, + "currentState": "CI and deploy jobs use hosted Linux, deployment already uses GCP Workload Identity Federation, and runtime credentials are documented in organicops dev, stg, and prd.", + "targetState": "All jobs use owned Linux, GCP deployment continues without a stored cloud key, and Doppler is injected only into jobs that actually consume application secrets.", + "rollout": [ + "Verify the self-hosted OIDC request path and GCP claim binding before changing runner labels.", + "Move jobs to owned Linux while preserving id-token scope on deploy jobs only.", + "Move non-secret WIF provider and service-account identifiers from GitHub secrets to environment variables.", + "Do not add a Doppler token to deploy jobs that need only GCP federation." + ], + "validation": [ + "Run the project manifest checker and tracked leak scan.", + "Prove staging and production WIF exchanges succeed from runner_environment self-hosted.", + "Prove no long-lived Google credential or Doppler token is present before, during, or after a WIF-only job." + ] + }, + { + "schemaVersion": 1, + "backlogId": "ci-selfhost-tab-agent-h7", + "project": "tab-agent-workspace", + "repository": "appheat/tab-agent-workspace", + "repositoryVisibility": "private", + "forkExposure": "none", + "classification": "doppler", + "doppler": { + "project": "tab-agent-workspace", + "configs": [ + "dev", + "stg", + "prd" + ] + }, + "ci": { + "runner": "owned-linux", + "identity": "doppler-service-token", + "injection": "per-job-doppler-fetch" + }, + "exceptions": [ + { + "kind": "temporary-migration", + "scope": "private Factory dependency token", + "reason": "The current workflow reads a repository-scoped fine-grained token directly from GitHub, so migration stores that value in tab-agent-workspace/stg until a short-lived GitHub App identity replaces it.", + "reviewBy": "2026-10-25", + "removalCondition": "Remove this exception after the dependency job uses a short-lived GitHub App or equivalent provider identity and the stored fine-grained token is revoked." + }, + { + "kind": "temporary-migration", + "scope": "legacy test config name", + "reason": "The current local command selects a test config outside the dev, stg, and prd convention, so the rollout must move deterministic tests to dev or make them secretless.", + "reviewBy": "2026-10-25", + "removalCondition": "Remove this exception when deterministic tests use the canonical dev config or a fully synthetic secretless path." + } + ], + "review": { + "owner": "tab-agent-workspace maintainers", + "cadenceDays": 90 + }, + "currentState": "All jobs request hosted Linux, cloud runtime values use Doppler locally, and CI reads one cross-repository token directly from a GitHub secret.", + "targetState": "All jobs use owned Linux, the cross-repository credential is fetched from tab-agent-workspace/stg, and deterministic tests use canonical or no secret configuration.", + "rollout": [ + "Verify Node, Go, OpenTofu, and container tooling on the owned runner before changing labels.", + "Transfer the private dependency credential into tab-agent-workspace/stg through an approved value-handling path.", + "Use the pinned masking fetch action only in the private dependency job and remove the direct repository secret after proof.", + "Replace the legacy test config with dev or a fully synthetic secretless test path." + ], + "validation": [ + "Run the project manifest checker and tracked leak scan.", + "Prove the private dependency resolves with read-only scope and no credential in git configuration after the job.", + "Prove every cloud and test job passes on owned Linux with the declared config." + ] + }, + { + "schemaVersion": 1, + "backlogId": "ci-selfhost-macops-registry-h8", + "project": "macops-registry", + "repository": "appheat/macops-registry", + "repositoryVisibility": "private", + "forkExposure": "none", + "classification": "secretless", + "doppler": { + "project": null, + "configs": [] + }, + "ci": { + "runner": "owned-macos", + "identity": "none", + "injection": "none" + }, + "exceptions": [ + { + "kind": "secretless", + "scope": "current repository with no workflows", + "reason": "The repository has no active workflow or secret consumer, so the queued migration is a policy placeholder and no vault entry should be created." + } + ], + "review": { + "owner": "macops-registry maintainers", + "cadenceDays": 90 + }, + "currentState": "The repository has no active GitHub Actions workflow.", + "targetState": "The first workflow declares its secrets posture before launch and uses the owned macOS JIT runner only when the job genuinely requires macOS.", + "rollout": [ + "Keep the migration queued without registering a credential or runner solely for the placeholder.", + "Before the first workflow lands, classify it with the standard and add any required Doppler project/config declaration.", + "Use the owned Linux runner unless the workflow proves a macOS requirement." + ], + "validation": [ + "Run the project manifest checker and tracked leak scan before the first workflow merges.", + "Prove any future workflow uses the declared owned runner and receives no undeclared credential." + ] + } + ] +} diff --git a/docs/secrets-rollout.schema.json b/docs/secrets-rollout.schema.json new file mode 100644 index 00000000000..8e95c9d4b5c --- /dev/null +++ b/docs/secrets-rollout.schema.json @@ -0,0 +1,164 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/juniorlovestmh/firstmate/blob/main/docs/secrets-rollout.schema.json", + "title": "Firstmate Secrets Rollout", + "description": "Value-free rollout plan coupling the queued self-hosted runner migrations to the Firstmate Doppler standard.", + "type": "object", + "required": [ + "schemaVersion", + "standard", + "evaluatedOn", + "oidcAssessment", + "projects" + ], + "properties": { + "$schema": { + "type": "string", + "minLength": 1 + }, + "schemaVersion": { + "const": 1 + }, + "standard": { + "const": "doppler-default-v1" + }, + "evaluatedOn": { + "type": "string", + "format": "date" + }, + "oidcAssessment": { + "type": "object", + "required": [ + "decisionOwner", + "status", + "recommendation", + "pricingCheckedOn", + "teamPriceUsdPerUserMonth", + "enterprisePrice", + "pricingUrl", + "reasoning", + "shippedDefault" + ], + "properties": { + "decisionOwner": { + "const": "captain" + }, + "status": { + "const": "recommendation-only" + }, + "recommendation": { + "const": "consider-oidc-to-remove-stored-doppler-ci-tokens" + }, + "pricingCheckedOn": { + "type": "string", + "format": "date" + }, + "teamPriceUsdPerUserMonth": { + "const": 21 + }, + "enterprisePrice": { + "const": "custom" + }, + "pricingUrl": { + "const": "https://www.doppler.com/pricing" + }, + "reasoning": { + "type": "string", + "minLength": 1 + }, + "shippedDefault": { + "type": "string", + "minLength": 1 + } + }, + "additionalProperties": false + }, + "projects": { + "type": "array", + "minItems": 8, + "maxItems": 8, + "items": { + "type": "object", + "required": [ + "schemaVersion", + "backlogId", + "project", + "repository", + "repositoryVisibility", + "forkExposure", + "classification", + "doppler", + "ci", + "exceptions", + "review", + "currentState", + "targetState", + "rollout", + "validation" + ], + "properties": { + "schemaVersion": { + "$ref": "./secrets-policy.schema.json#/properties/schemaVersion" + }, + "backlogId": { + "type": "string", + "pattern": "^ci-selfhost-[a-z0-9-]+-[a-z][0-9]+$" + }, + "project": { + "$ref": "./secrets-policy.schema.json#/properties/project" + }, + "repository": { + "$ref": "./secrets-policy.schema.json#/properties/repository" + }, + "repositoryVisibility": { + "$ref": "./secrets-policy.schema.json#/properties/repositoryVisibility" + }, + "forkExposure": { + "$ref": "./secrets-policy.schema.json#/properties/forkExposure" + }, + "classification": { + "$ref": "./secrets-policy.schema.json#/properties/classification" + }, + "doppler": { + "$ref": "./secrets-policy.schema.json#/properties/doppler" + }, + "ci": { + "$ref": "./secrets-policy.schema.json#/properties/ci" + }, + "exceptions": { + "$ref": "./secrets-policy.schema.json#/properties/exceptions" + }, + "review": { + "$ref": "./secrets-policy.schema.json#/properties/review" + }, + "currentState": { + "type": "string", + "minLength": 1 + }, + "targetState": { + "type": "string", + "minLength": 1 + }, + "rollout": { + "type": "array", + "minItems": 1, + "items": { + "type": "string", + "minLength": 1 + } + }, + "validation": { + "type": "array", + "minItems": 1, + "items": { + "type": "string", + "minLength": 1 + } + } + }, + "additionalProperties": false + } + } + }, + "additionalProperties": false +} diff --git a/docs/toolchain-versions.md b/docs/toolchain-versions.md new file mode 100644 index 00000000000..f3a618434ec --- /dev/null +++ b/docs/toolchain-versions.md @@ -0,0 +1,104 @@ +# Fleet toolchain version manifest and recovery + +This manifest records the fleet-critical CLI installation observed on 2026-07-29 on Darwin arm64. +It is an evidence snapshot, not an update request. +No tool version was changed while producing it. + +## Installed versions + +| Tool | Live version | Installed source | Current update mechanism | +| --- | --- | --- | --- | +| `tasks-axi` | `0.2.3` | Global npm package under `/opt/homebrew/lib/node_modules/tasks-axi` | `npm update -g tasks-axi` | +| `gh-axi` | `0.1.27` | Global npm package under `/opt/homebrew/lib/node_modules/gh-axi` | `npm update -g gh-axi` | +| `lavish-axi` | `0.1.42` | Global npm package under `/opt/homebrew/lib/node_modules/lavish-axi` | `npm update -g lavish-axi` | +| `quota-axi` | `0.1.6` | Global npm package under `/opt/homebrew/lib/node_modules/quota-axi` | `npm update -g quota-axi` | +| `chrome-devtools-axi` | `0.1.26` | Global npm package under `/opt/homebrew/lib/node_modules/chrome-devtools-axi` | `npm update -g chrome-devtools-axi` | +| `no-mistakes` | `v1.40.3` (`d873960`, built `2026-07-22T01:41:55Z`) | Self-contained binary at `~/.no-mistakes/bin/no-mistakes`, reached through `~/.local/bin/no-mistakes` | `no-mistakes update`, which also resets the shared daemon | +| `herdr` | `0.7.5` | Homebrew core bottle at `/opt/homebrew/Cellar/herdr/0.7.5/bin/herdr` | `brew upgrade herdr`; the binary also exposes `herdr update` | +| `treehouse` | `v2.1.0` | Direct Mach-O arm64 binary at `/opt/homebrew/bin/treehouse`; no installed Homebrew keg was present | `treehouse update` | + +The five npm package versions were also matched against their installed `package.json` files. +Their package metadata points to the corresponding `kunchenguid/*-axi` repositories. + +## Live evidence commands + +The following commands were run from the Firstmate task worktree. + +```sh +for tool in tasks-axi gh-axi lavish-axi quota-axi chrome-devtools-axi no-mistakes herdr treehouse; do + command -v "$tool" + "$tool" --version +done +npm prefix -g +npm root -g +npm list -g --depth=0 +brew list --versions herdr treehouse +brew info --json=v2 herdr treehouse +``` + +The exact `--version` outputs were: + +```text +tasks-axi: 0.2.3 +gh-axi: 0.1.27 +lavish-axi: 0.1.42 +quota-axi: 0.1.6 +chrome-devtools-axi: 0.1.26 +no-mistakes: no-mistakes version v1.40.3 (d873960) 2026-07-22T01:41:55Z +herdr: herdr 0.7.5 +treehouse: v2.1.0 +``` + +The live mirror acceptance run completed on 2026-07-29 local time, or 2026-07-30 UTC. +`bin/fm-toolchain-mirror.sh snapshot` created snapshot `20260730T030202Z-Darwin-arm64` with eight artifacts. +`bin/fm-toolchain-mirror.sh verify` checked all eight artifacts successfully. +An isolated-prefix restore then reproduced every version output above. +The local mirror occupied 67 MiB, and its `manifest.tsv` SHA-256 was `34985504ab51d5eee5b36ba4582dbc9204d816c3eebee5a2976ff5e60e1d422d`. + +## Local offline mirror + +[`bin/fm-toolchain-mirror.sh`](../bin/fm-toolchain-mirror.sh) snapshots the eight installed tools without contacting an upstream or changing their versions. +Its default destination is `$FM_HOME/data/toolchain-mirror`. +The repository already ignores `data/`, so the large platform-specific package trees and binaries remain local. +Set `FM_TOOLCHAIN_MIRROR` or pass `--mirror` to use another operator-controlled volume. + +Each snapshot contains: + +- The complete installed directory for each npm package, including its installed dependencies. +- Exact binary bytes for `no-mistakes`, `herdr`, and `treehouse`. +- Raw `--version` output for every tool. +- A TSV manifest with versions, install sources, update mechanisms, and SHA-256 checksums. +- The operating system and architecture required by the captured binaries. + +Create and verify a snapshot: + +```sh +bin/fm-toolchain-mirror.sh snapshot +bin/fm-toolchain-mirror.sh verify +``` + +Restore the current snapshot offline into a new, isolated prefix: + +```sh +recovery_prefix="$FM_HOME/data/toolchain-recovery/$(date -u +%Y%m%dT%H%M%SZ)" +bin/fm-toolchain-mirror.sh restore --prefix "$recovery_prefix" +export PATH="$recovery_prefix/bin:$PATH" +``` + +Restore one tool by adding `--tool `. +The restore refuses an existing prefix, verifies every selected checksum before writing, and verifies the restored `--version` before publishing the new prefix. + +## Upstream break or disappearance recovery + +1. Stop issuing update commands and preserve the broken installation for diagnosis. +2. Run `bin/fm-toolchain-mirror.sh verify` against the last known-good local snapshot. +3. Restore that snapshot to a new prefix and place its `bin/` first on `PATH`. +4. Re-run the live evidence commands above and the affected Firstmate bootstrap or task path. +5. Resume fleet work only after the pinned commands match the mirror manifest. + +Do not restart or update the shared `no-mistakes` daemon while any lane has an active pipeline. +If the restored CLI and running daemon are incompatible, drain active lanes and let the Firstmate operator own the daemon recovery. + +The mirror does not disable any updater. +That preserves normal update notices and operator choice, but it also means a later manual update can still introduce breaking bytes. +After every deliberate, verified tool upgrade, create a new snapshot and retain the previous known-good snapshot until fleet validation passes. diff --git a/tests/fm-secrets-check.test.sh b/tests/fm-secrets-check.test.sh new file mode 100755 index 00000000000..0b132e6ab29 --- /dev/null +++ b/tests/fm-secrets-check.test.sh @@ -0,0 +1,528 @@ +#!/usr/bin/env bash +# Contract tests for Firstmate's shared Doppler policy, rollout manifest, and +# value-safe deterministic checker. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +CHECK="$ROOT/bin/fm-secrets-check.sh" +SKILL="$ROOT/.agents/skills/secrets-management/SKILL.md" +SCHEMA="$ROOT/docs/secrets-policy.schema.json" +ROLLOUT_SCHEMA="$ROOT/docs/secrets-rollout.schema.json" +ROLLOUT="$ROOT/docs/secrets-rollout.json" +EXAMPLE="$ROOT/docs/examples/project-secrets-policy.json" +PROJECT_MANAGEMENT="$ROOT/.agents/skills/project-management/SKILL.md" +TMP_ROOT=$(mktemp -d "${TMPDIR:-/tmp}/fm-secrets-check.XXXXXX") +trap 'rm -rf "$TMP_ROOT"' EXIT + +test_public_entrypoint_validates_tracked_standard() { + local out + assert_present "$CHECK" "public secrets-policy checker is missing" + [ -x "$CHECK" ] || fail "public secrets-policy checker is not executable" + out=$("$CHECK" 2>&1) || fail "tracked secrets standard did not validate"$'\n'"$out" + assert_contains "$out" "secrets-standard: ok" "checker did not report standard validation" + assert_contains "$out" "projects=8" "checker did not validate the exact eight-project rollout" + assert_contains "$out" "leak-scan: ok files=17" \ + "default dogfood did not scan every tracked standard and integration artifact" + pass "secrets checker validates the tracked standard and eight-project rollout" +} + +test_policy_owner_has_one_precise_trigger() { + local count + assert_present "$SKILL" "secrets-management skill is missing" + assert_grep "name: secrets-management" "$SKILL" "secrets-management skill metadata has the wrong name" + assert_grep "user-invocable: false" "$SKILL" "secrets-management skill must not be user-invocable" + assert_grep "single owner of Firstmate's secrets-management policy" "$SKILL" \ + "secrets-management skill does not declare policy ownership" + count=$(grep -Fc -- "- \`secrets-management\` -" "$ROOT/AGENTS.md") + [ "$count" -eq 1 ] || fail "secrets-management must have exactly one AGENTS.md trigger entry, found $count" + assert_grep "load before project intake or initialization and before work that handles credentials or adds secret access to CI or deployment" "$ROOT/AGENTS.md" \ + "AGENTS.md lost the precise secrets-management trigger" + assert_grep "run \`bin/fm-secrets-check.sh inventory projects/\`" "$PROJECT_MANAGEMENT" \ + "project initialization does not run the value-safe inventory entrypoint" + assert_grep "copy and complete \`docs/examples/project-secrets-policy.json\`" "$PROJECT_MANAGEMENT" \ + "project initialization does not schedule the project manifest contract" + pass "secrets-management has one precise always-loaded trigger" +} + +test_paid_oidc_remains_captain_owned_recommendation() { + local out + assert_grep "Config-scoped read-only service tokens are the shipped CI default." "$SKILL" \ + "service tokens are not declared as the shipped CI default" + assert_grep "Whether that benefit justifies the paid plan is the captain's decision." "$SKILL" \ + "paid OIDC decision is not reserved for the captain" + assert_grep "Team at \$21 per user per month" "$SKILL" \ + "OIDC recommendation does not record the checked Team-plan price" + out=$(python3 - "$ROLLOUT" <<'PY' +import json +import sys + +with open(sys.argv[1], encoding="utf-8") as fh: + assessment = json.load(fh)["oidcAssessment"] + +assert assessment["decisionOwner"] == "captain" +assert assessment["status"] == "recommendation-only" +assert assessment["teamPriceUsdPerUserMonth"] == 21 +assert assessment["enterprisePrice"] == "custom" +assert "read-only service token" in assessment["shippedDefault"] +assert "no Team-plan" in assessment["shippedDefault"] +print("captain-owned recommendation; service-token default") +PY +) + assert_contains "$out" "service-token default" \ + "rollout does not keep service tokens as the paid-plan-independent default" + pass "paid OIDC stays a costed captain-owned recommendation" +} + +test_schemas_and_example_are_publicly_validated() { + local path out + for path in "$SCHEMA" "$ROLLOUT_SCHEMA" "$ROLLOUT" "$EXAMPLE"; do + assert_present "$path" "required secrets standard artifact is missing: $path" + done + out=$("$CHECK" manifest "$EXAMPLE" 2>&1) \ + || fail "example project secrets manifest did not validate"$'\n'"$out" + assert_contains "$out" "manifest: ok" "checker did not report project manifest validation" + pass "public schemas, rollout, and project example validate" +} + +test_schema_patterns_and_unknown_fields_are_enforced() { + local manifest out rc + manifest="$TMP_ROOT/schema-violations.json" + cat >"$manifest" <<'EOF' +{ + "schemaVersion": 1, + "project": "Bad Name", + "repository": "owner/extra/name", + "classification": "doppler", + "doppler": { + "project": "example", + "configs": ["dev"] + }, + "ci": { + "runner": "owned-linux", + "identity": "doppler-service-token", + "injection": "per-job-doppler-fetch" + }, + "exceptions": [], + "review": { + "owner": "project-maintainer", + "cadenceDays": 90 + }, + "credentialValue": "must-not-exist" +} +EOF + rc=0 + out=$("$CHECK" manifest "$manifest" 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "checker accepted fields forbidden by the public schema" + assert_contains "$out" '$.project must match' "checker did not enforce the project slug pattern" + assert_contains "$out" '$.repository must match' "checker did not enforce owner/name" + assert_contains "$out" '$.credentialValue is not allowed' "checker did not reject an unknown field" + pass "manifest checker executes schema patterns and field closure" +} + +test_undocumented_exception_is_rejected() { + local manifest out rc + manifest="$TMP_ROOT/undocumented-exception.json" + cat >"$manifest" <<'EOF' +{ + "schemaVersion": 1, + "project": "example", + "repository": "example/example", + "classification": "secretless", + "doppler": { + "project": null, + "configs": [] + }, + "ci": { + "runner": "owned-linux", + "identity": "none", + "injection": "none" + }, + "exceptions": [ + { + "kind": "secretless", + "scope": "all", + "reason": "" + } + ], + "review": { + "owner": "project-maintainer", + "cadenceDays": 90 + } +} +EOF + rc=0 + out=$("$CHECK" manifest "$manifest" 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "checker accepted an exception without a reason" + assert_contains "$out" "exceptions[0].reason" "checker did not identify the undocumented exception" + pass "project manifest rejects an exception without a reason" +} + +test_temporary_migration_requires_removal_contract() { + local manifest out rc + manifest="$TMP_ROOT/unbounded-migration.json" + cat >"$manifest" <<'EOF' +{ + "schemaVersion": 1, + "project": "example", + "repository": "example/example", + "classification": "doppler", + "doppler": { + "project": "example", + "configs": ["stg"] + }, + "ci": { + "runner": "owned-linux", + "identity": "doppler-service-token", + "injection": "per-job-doppler-fetch" + }, + "exceptions": [ + { + "kind": "temporary-migration", + "scope": "legacy path", + "reason": "A bounded transition is required." + } + ], + "review": { + "owner": "project-maintainer", + "cadenceDays": 90 + } +} +EOF + rc=0 + out=$("$CHECK" manifest "$manifest" 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "checker accepted an unbounded temporary migration" + assert_contains "$out" "reviewBy" "checker did not require a migration review date" + assert_contains "$out" "removalCondition" "checker did not require a removal condition" + pass "temporary migrations require a dated removal contract" +} + +test_identity_and_exception_cross_fields_are_enforced() { + local platform_manifest injection_manifest out rc + platform_manifest="$TMP_ROOT/platform-native-without-reason.json" + injection_manifest="$TMP_ROOT/service-token-without-injection.json" + + sed 's/"classification": "doppler"/"classification": "platform-native"/' \ + "$EXAMPLE" >"$platform_manifest" + rc=0 + out=$("$CHECK" manifest "$platform_manifest" 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "checker accepted platform-native without a documented exception" + assert_contains "$out" "platform-mandated" \ + "checker did not require the platform-native authority reason" + + sed 's/"injection": "per-job-doppler-fetch"/"injection": "none"/' \ + "$EXAMPLE" >"$injection_manifest" + rc=0 + out=$("$CHECK" manifest "$injection_manifest" 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "checker accepted a service token with no injection path" + assert_contains "$out" "per-job-doppler-fetch" \ + "checker did not bind service-token identity to per-job injection" + pass "identity and exception cross-field contracts are enforced" +} + +test_doppler_injection_requires_owned_runner_or_explicit_workflow_override() { + local rejected hosted incomplete approved public unknown out rc + rejected="$TMP_ROOT/non-owned-doppler.json" + hosted="$TMP_ROOT/non-owned-doppler-hosted-exception.json" + incomplete="$TMP_ROOT/non-owned-doppler-incomplete-override.json" + approved="$TMP_ROOT/non-owned-doppler-override.json" + public="$TMP_ROOT/non-owned-doppler-public.json" + unknown="$TMP_ROOT/non-owned-doppler-unknown-exposure.json" + sed 's/"runner": "owned-linux"/"runner": "github-hosted-public-fork"/' \ + "$EXAMPLE" >"$rejected" + rc=0 + out=$("$CHECK" manifest "$rejected" 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "checker accepted Doppler injection on a non-owned runner" + assert_contains "$out" "owned runner" \ + "checker did not keep the owned-runner predicate authoritative" + + python3 - "$rejected" "$hosted" "$incomplete" <<'PY' +import json +import sys + +with open(sys.argv[1], encoding="utf-8") as fh: + manifest = json.load(fh) +manifest["exceptions"] = [{ + "kind": "hosted-runner", + "scope": "public fork pull requests", + "reason": "Untrusted public fork work must not run on owned hardware.", +}] +with open(sys.argv[2], "w", encoding="utf-8") as fh: + json.dump(manifest, fh, indent=2) + fh.write("\n") +manifest["exceptions"] = [{ + "kind": "non-owned-runner-doppler", + "scope": ".github/workflows/approved.yml", + "reason": "A separately approved provider boundary requires this named workflow.", +}] +with open(sys.argv[3], "w", encoding="utf-8") as fh: + json.dump(manifest, fh, indent=2) + fh.write("\n") +PY + for invalid in "$hosted" "$incomplete"; do + rc=0 + out=$("$CHECK" manifest "$invalid" 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "checker accepted an unauthorized non-owned-runner override: $invalid" + assert_contains "$out" "ci.runner" "checker did not fail closed for $invalid" + done + + python3 - "$rejected" "$approved" <<'PY' +import json +import sys + +with open(sys.argv[1], encoding="utf-8") as fh: + manifest = json.load(fh) +manifest["exceptions"] = [{ + "kind": "non-owned-runner-doppler", + "scope": ".github/workflows/approved.yml", + "workflow": ".github/workflows/approved.yml", + "reason": "A separately approved provider boundary requires this named workflow.", +}] +with open(sys.argv[2], "w", encoding="utf-8") as fh: + json.dump(manifest, fh, indent=2) + fh.write("\n") +PY + rc=0 + out=$("$CHECK" manifest "$approved" 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "checker allowed an override to unlock a non-owned runner" + assert_contains "$out" "owned runner" "checker did not keep the runner predicate authoritative" + python3 - "$approved" "$public" <<'PY' +import json +import sys + +with open(sys.argv[1], encoding="utf-8") as fh: + manifest = json.load(fh) +manifest["repositoryVisibility"] = "public" +with open(sys.argv[2], "w", encoding="utf-8") as fh: + json.dump(manifest, fh, indent=2) + fh.write("\n") +PY + rc=0 + out=$("$CHECK" manifest "$public" 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "checker accepted a public repository override" + assert_contains "$out" "unavailable" "checker did not enforce the public repository override wall" + python3 - "$approved" "$unknown" <<'PY' +import json +import sys + +with open(sys.argv[1], encoding="utf-8") as fh: + manifest = json.load(fh) +manifest["forkExposure"] = "unknown" +with open(sys.argv[2], "w", encoding="utf-8") as fh: + json.dump(manifest, fh, indent=2) + fh.write("\n") +PY + rc=0 + out=$("$CHECK" manifest "$unknown" 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "checker accepted unknown repository exposure" + assert_contains "$out" "forkExposure" "checker did not refuse unknown repository exposure" + pass "Doppler injection keeps the owned-runner predicate authoritative" +} + +test_temporary_migration_review_window_is_deterministic() { + local manifest out rc + manifest="$TMP_ROOT/bounded-migration.json" + cp "$EXAMPLE" "$manifest" + python3 - "$manifest" <<'PY' +import json +import sys + +with open(sys.argv[1], encoding="utf-8") as fh: + manifest = json.load(fh) +manifest["exceptions"] = [{ + "kind": "temporary-migration", + "scope": "legacy path", + "reason": "A bounded transition is required.", + "reviewBy": "2026-07-27", + "removalCondition": "Remove the legacy path.", +}] +with open(sys.argv[1], "w", encoding="utf-8") as fh: + json.dump(manifest, fh, indent=2) + fh.write("\n") +PY + for date in 2026-07-27 2026-10-25; do + sed -i.bak "s/\"reviewBy\": \"[0-9-]*\"/\"reviewBy\": \"$date\"/" "$manifest" + "$CHECK" test-manifest-fixture "$manifest" 2026-07-27 >/dev/null 2>&1 \ + || fail "checker rejected temporary migration reviewBy=$date" + done + for date in 2026-07-26 2026-10-26; do + sed -i.bak "s/\"reviewBy\": \"[0-9-]*\"/\"reviewBy\": \"$date\"/" "$manifest" + rc=0 + out=$("$CHECK" test-manifest-fixture "$manifest" 2026-07-27 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "checker accepted out-of-window temporary migration reviewBy=$date" + assert_contains "$out" "must be today or within the next 90 days" \ + "checker did not report the bounded review window for reviewBy=$date" + done + sed -i.bak 's/"reviewBy": "[0-9-]*"/"reviewBy": "2026-07-26"/' "$manifest" + rc=0 + out=$(FM_SECRETS_CHECK_TODAY=2026-07-25 "$CHECK" manifest "$manifest" 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "normal manifest validation honored the test clock environment" + assert_contains "$out" "must be today or within the next 90 days" \ + "normal manifest validation did not enforce the runtime clock" + rm -f "$manifest.bak" + pass "temporary migration review dates enforce deterministic inclusive 90-day bounds" +} + +test_inventory_rejects_unsafe_doppler_workflow_targets() { + local case_root manifest workflow out rc + for case_root in owned hosted indeterminate quoted fork unknown-trigger override; do + case_root="$TMP_ROOT/workflow-$case_root" + mkdir -p "$case_root/docs" "$case_root/.github/workflows" + cp "$EXAMPLE" "$case_root/docs/secrets-policy.json" + manifest="$case_root/docs/secrets-policy.json" + workflow="$case_root/.github/workflows/deploy.yml" + workflow_trigger=$'on:\n push:\n workflow_dispatch:' + case "$case_root" in + *owned) + runner='[self-hosted, Linux, X64, fleet-ci]' + job_key='deploy' + ;; + *hosted) + runner='ubuntu-latest' + job_key='deploy' + ;; + *indeterminate) + runner="\${{ matrix.runner }}" + job_key='deploy' + ;; + *quoted) + runner='[self-hosted, Linux, X64, fleet-ci]' + job_key='"deploy"' + ;; + *fork) + runner='ubuntu-latest' + job_key='deploy' + workflow_trigger=$'on:\n pull_request:' + python3 - "$manifest" <<'PY' +import json +import sys + +with open(sys.argv[1], encoding="utf-8") as fh: + manifest = json.load(fh) +manifest["ci"]["runner"] = "github-hosted-public-fork" +manifest["exceptions"] = [{ + "kind": "non-owned-runner-doppler", + "scope": ".github/workflows/deploy.yml", + "workflow": ".github/workflows/deploy.yml", + "reason": "A separately approved provider boundary requires this named workflow.", +}] +with open(sys.argv[1], "w", encoding="utf-8") as fh: + json.dump(manifest, fh, indent=2) + fh.write("\n") +PY + ;; + *unknown-trigger) + runner='[self-hosted, Linux, X64, fleet-ci]' + job_key='deploy' + workflow_trigger='' + ;; + *) + runner='ubuntu-latest' + job_key='deploy' + python3 - "$manifest" <<'PY' +import json +import sys + +with open(sys.argv[1], encoding="utf-8") as fh: + manifest = json.load(fh) +manifest["ci"]["runner"] = "github-hosted-public-fork" +manifest["exceptions"] = [{ + "kind": "non-owned-runner-doppler", + "scope": ".github/workflows/deploy.yml", + "workflow": ".github/workflows/deploy.yml", + "reason": "A separately approved provider boundary requires this named workflow.", +}] +with open(sys.argv[1], "w", encoding="utf-8") as fh: + json.dump(manifest, fh, indent=2) + fh.write("\n") +PY + ;; + esac + cat >"$workflow" <&1) || rc=$? + if [ "$case_root" = "$TMP_ROOT/workflow-owned" ]; then + [ "$rc" -eq 0 ] || fail "inventory rejected the positively proven owned workflow"$'\n'"$out" + assert_contains "$out" "doppler_refs=true" "inventory did not inspect the owned workflow" + else + [ "$rc" -ne 0 ] || fail "inventory accepted unsafe workflow target: $case_root" + if [ "$case_root" = "$TMP_ROOT/workflow-quoted" ]; then + assert_contains "$out" "workflow requires" \ + "inventory did not reject the ambiguous quoted job structure" + elif [ "$case_root" = "$TMP_ROOT/workflow-fork" ]; then + assert_contains "$out" "trigger=fork-originated" \ + "inventory did not enforce the fork-exposure override wall" + else + assert_contains "$out" "Doppler injection requires" \ + "inventory did not report the unsafe workflow target: $case_root" + fi + fi + done + + mkdir -p "$TMP_ROOT/not-a-git-worktree" + rc=0 + out=$("$CHECK" inventory "$TMP_ROOT/not-a-git-worktree" 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "inventory accepted a failed authoritative workflow enumeration" + assert_contains "$out" "requires a git worktree" "inventory did not fail authoritative enumeration loudly" + + local unreadable="$TMP_ROOT/workflow-unreadable" + mkdir -p "$unreadable/docs" "$unreadable/.github/workflows" + cp "$EXAMPLE" "$unreadable/docs/secrets-policy.json" + fm_git_init_commit "$unreadable" + git -C "$unreadable" add docs/secrets-policy.json + git -C "$unreadable" -c user.name='Firstmate Tests' -c user.email='tests@example.invalid' commit -qm policy + printf '%s\n' 'jobs:' ' deploy:' ' runs-on: [self-hosted, Linux, X64, fleet-ci]' > \ + "$unreadable/.github/workflows/deploy.yml" + git -C "$unreadable" add .github/workflows/deploy.yml + rc=0 + out=$("$CHECK" inventory "$unreadable" 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "inventory accepted an unreadable tracked workflow" + assert_contains "$out" "unavailable for validation" "inventory did not fail unreadable workflow validation loudly" + pass "inventory completes workflow enumeration and records refusal verdicts" +} + +test_value_leak_scan_reports_rule_not_value() { + local clean bad canary out rc + clean="$TMP_ROOT/clean-policy.txt" + bad="$TMP_ROOT/bad-policy.txt" + printf '%s\n' 'DOPPLER_TOKEN is an allowed secret name.' >"$clean" + "$CHECK" leak-scan "$clean" >/dev/null 2>&1 \ + || fail "leak scanner rejected a secret name without a value" + + canary='synthetic-canary-not-a-real-secret' + printf '%s%s\n' 'DOPPLER_TOKEN=dp.st.' "$canary" >"$bad" + rc=0 + out=$("$CHECK" leak-scan "$bad" 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "leak scanner accepted a secret-shaped synthetic value" + assert_contains "$out" "doppler-service-token" "leak scanner did not name the matched rule" + assert_contains "$out" "$bad:1" "leak scanner did not report the location" + assert_not_contains "$out" "$canary" "leak scanner echoed the matched value" + pass "leak scanner rejects a synthetic value without echoing it" +} + +test_public_entrypoint_validates_tracked_standard +test_policy_owner_has_one_precise_trigger +test_paid_oidc_remains_captain_owned_recommendation +test_schemas_and_example_are_publicly_validated +test_schema_patterns_and_unknown_fields_are_enforced +test_undocumented_exception_is_rejected +test_temporary_migration_requires_removal_contract +test_identity_and_exception_cross_fields_are_enforced +test_doppler_injection_requires_owned_runner_or_explicit_workflow_override +test_temporary_migration_review_window_is_deterministic +test_inventory_rejects_unsafe_doppler_workflow_targets +test_value_leak_scan_reports_rule_not_value diff --git a/tests/fm-test-run.test.sh b/tests/fm-test-run.test.sh index e0714cd3403..76859092c8a 100755 --- a/tests/fm-test-run.test.sh +++ b/tests/fm-test-run.test.sh @@ -513,22 +513,46 @@ exit 1 SH cat >"$repo/$a" <<'SH' #!/usr/bin/env bash -sleep 0.5 +touch "$SCHED_EVIDENCE/slow-started" +attempt=0 +while [ ! -e "$SCHED_EVIDENCE/release-slow" ]; do + attempt=$((attempt + 1)) + if [ "$attempt" -ge 1000 ]; then + echo "not ok - slow fixture was never released" + exit 1 + fi + sleep 0.01 +done touch "$SCHED_EVIDENCE/slow-done" echo "ok - slow fixture" SH cat >"$repo/$b" <<'SH' #!/usr/bin/env bash -sleep 0.05 +attempt=0 +while [ ! -e "$SCHED_EVIDENCE/slow-started" ]; do + attempt=$((attempt + 1)) + if [ "$attempt" -ge 1000 ]; then + echo "not ok - slow fixture never started" + exit 1 + fi + sleep 0.01 +done echo "ok - fast fixture" SH cat >"$repo/$c" <<'SH' #!/usr/bin/env bash +if [ ! -e "$SCHED_EVIDENCE/slow-started" ]; then + touch "$SCHED_EVIDENCE/release-slow" + echo "not ok - replacement fixture started before both initial workers" + exit 1 +fi if [ -e "$SCHED_EVIDENCE/slow-done" ]; then + touch "$SCHED_EVIDENCE/release-slow" echo "not ok - scheduler waited for oldest worker" exit 1 fi -echo "ok - replacement fixture started before slow fixture finished" +touch "$SCHED_EVIDENCE/release-slow" +echo "ok - replacement fixture released the blocked slow fixture" SH chmod +x "$runner" "$repo/$a" "$repo/$b" "$repo/$c" "$fake_bin/stat" set +e diff --git a/tests/fm-toolchain-mirror.test.sh b/tests/fm-toolchain-mirror.test.sh new file mode 100755 index 00000000000..3f1a18fffe6 --- /dev/null +++ b/tests/fm-toolchain-mirror.test.sh @@ -0,0 +1,128 @@ +#!/usr/bin/env bash +# Contract and behavior tests for the offline toolchain mirror. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +MIRROR_SCRIPT="$ROOT/bin/fm-toolchain-mirror.sh" +assert_present "$MIRROR_SCRIPT" "bin/fm-toolchain-mirror.sh is missing" +[ -x "$MIRROR_SCRIPT" ] || fail "fm-toolchain-mirror.sh must be executable" +TMP_ROOT=$(fm_test_tmproot fm-toolchain-mirror) +trap 'rm -rf "$TMP_ROOT"' EXIT + +make_version_tool() { + path=$1 + output=$2 + mkdir -p "$(dirname "$path")" + { + printf '%s\n' '#!/usr/bin/env bash' + printf 'printf '\''%%s\\n'\'' '\''%s'\''\n' "$output" + } > "$path" + chmod +x "$path" +} + +make_npm_tool() { + tool=$1 + version=$2 + bin_relative=$3 + package_dir="$CASE_DIR/npm-root/$tool" + mkdir -p "$package_dir/$(dirname "$bin_relative")" "$package_dir/node_modules/example-dependency" + cat > "$package_dir/package.json" < "$package_dir/$bin_relative" < "$package_dir/node_modules/example-dependency/proof.txt" + ln -s "$package_dir/$bin_relative" "$CASE_DIR/fakebin/$tool" +} + +setup_case() { + CASE_DIR="$TMP_ROOT/$1" + mkdir -p "$CASE_DIR/fakebin" "$CASE_DIR/npm-root" + make_npm_tool tasks-axi 0.2.3 dist/bin/tasks-axi.js + make_npm_tool gh-axi 0.1.27 lib/custom-gh-entry.js + make_npm_tool lavish-axi 0.1.42 dist/cli.mjs + make_npm_tool quota-axi 0.1.6 dist/bin/quota-axi.js + make_npm_tool chrome-devtools-axi 0.1.26 dist/bin/chrome-devtools-axi.js + make_version_tool "$CASE_DIR/fakebin/no-mistakes" \ + 'no-mistakes version v1.40.3 (test) 2026-07-22T01:41:55Z' + make_version_tool "$CASE_DIR/fakebin/herdr" 'herdr 0.7.5' + make_version_tool "$CASE_DIR/fakebin/treehouse" 'v2.1.0' +} + +test_snapshot_and_offline_restore() { + setup_case snapshot-restore + mirror="$CASE_DIR/mirror" + restore="$CASE_DIR/restored" + out=$(PATH="$CASE_DIR/fakebin:$PATH" \ + FM_TOOLCHAIN_NPM_ROOT="$CASE_DIR/npm-root" \ + FM_TOOLCHAIN_SNAPSHOT_ID=test-snapshot \ + "$MIRROR_SCRIPT" snapshot --mirror "$mirror") || fail "snapshot command failed" + + assert_contains "$out" "snapshot: test-snapshot" "snapshot must report its id" + assert_contains "$out" "artifacts: 8" "snapshot must contain all eight tools" + assert_present "$mirror/current" "snapshot must publish the current marker" + assert_present "$mirror/snapshots/test-snapshot/manifest.tsv" "snapshot must publish a manifest" + assert_present "$mirror/snapshots/test-snapshot/versions/no-mistakes.txt" \ + "snapshot must retain raw version output" + assert_contains "$(cat "$mirror/snapshots/test-snapshot/versions/no-mistakes.txt")" \ + "no-mistakes version v1.40.3" "raw version output must be preserved" + + verify_out=$("$MIRROR_SCRIPT" verify --mirror "$mirror") || fail "verify command failed" + assert_contains "$verify_out" "(8 artifacts)" "verify must check every artifact" + + restore_out=$("$MIRROR_SCRIPT" restore --mirror "$mirror" --prefix "$restore") \ + || fail "restore command failed" + assert_contains "$restore_out" "(8 tools)" "restore must reinstall every tool" + assert_present "$restore/bin/no-mistakes" "restore must install binary tools" + assert_present "$restore/lib/node_modules/lavish-axi/node_modules/example-dependency/proof.txt" \ + "restore must retain installed npm dependencies for offline use" + [ "$("$restore/bin/tasks-axi" --version)" = "0.2.3" ] \ + || fail "restored tasks-axi must report the pinned version" + [ "$("$restore/bin/treehouse" --version)" = "v2.1.0" ] \ + || fail "restored treehouse must report the pinned version" + pass "snapshot and restore preserve all pinned tools offline" +} + +test_restore_refuses_existing_prefix() { + setup_case existing-prefix + mirror="$CASE_DIR/mirror" + PATH="$CASE_DIR/fakebin:$PATH" \ + FM_TOOLCHAIN_NPM_ROOT="$CASE_DIR/npm-root" \ + FM_TOOLCHAIN_SNAPSHOT_ID=test-snapshot \ + "$MIRROR_SCRIPT" snapshot --mirror "$mirror" >/dev/null \ + || fail "snapshot setup failed" + mkdir -p "$CASE_DIR/existing" + if "$MIRROR_SCRIPT" restore --mirror "$mirror" --prefix "$CASE_DIR/existing" \ + >"$CASE_DIR/restore.out" 2>"$CASE_DIR/restore.err"; then + fail "restore must refuse an existing prefix" + fi + assert_contains "$(cat "$CASE_DIR/restore.err")" "restore prefix already exists" \ + "restore refusal must explain how to recover safely" + pass "restore refuses to overwrite an existing installation" +} + +test_verify_rejects_tampered_artifact() { + setup_case tampered-artifact + mirror="$CASE_DIR/mirror" + PATH="$CASE_DIR/fakebin:$PATH" \ + FM_TOOLCHAIN_NPM_ROOT="$CASE_DIR/npm-root" \ + FM_TOOLCHAIN_SNAPSHOT_ID=test-snapshot \ + "$MIRROR_SCRIPT" snapshot --mirror "$mirror" >/dev/null \ + || fail "snapshot setup failed" + printf '%s\n' 'tamper' >> "$mirror/snapshots/test-snapshot/artifacts/bin/treehouse" + if "$MIRROR_SCRIPT" verify --mirror "$mirror" >"$CASE_DIR/verify.out" 2>"$CASE_DIR/verify.err"; then + fail "verify must reject a tampered artifact" + fi + assert_contains "$(cat "$CASE_DIR/verify.err")" "checksum mismatch for treehouse" \ + "tamper refusal must name the affected tool" + pass "verify rejects changed mirror bytes" +} + +test_snapshot_and_offline_restore +test_restore_refuses_existing_prefix +test_verify_rejects_tampered_artifact From d77916eaa4757121dadbcb9c03f8daf8c1bb2275 Mon Sep 17 00:00:00 2001 From: juniorlovestmh Date: Fri, 31 Jul 2026 07:41:42 -0300 Subject: [PATCH 02/52] feat(mobile-mode): add verified Moshi Pro mobile review workflow (#6) * docs: separate current guidance from verification evidence (#994) * docs: separate current guides from verification * no-mistakes(review): Restore Herdr 0.7.5 restart-reclaim verification evidence * fix: preserve Claude watcher continuity across Stop hooks (#997) * feat(claude): Stop-owned tokenless watcher continuity via asyncRewake auto-arm Claude primaries (main home and marked secondmate homes) no longer depend on the model remembering to re-arm the watcher after each wake. A tracked Stop asyncRewake hook (bin/fm-claude-stop-autoarm.sh, timeout 28800s) fires on every turn end, claims one home-scoped single-flight owner, foregrounds bin/fm-watch-arm.sh inside the hook-owned process tree, and translates an actionable close or typed watcher failure into exactly one exit-2 rewake. The hook scopes to genuine primary checkouts, requires the session lock to be held by its own harness ancestor, stays inert while AFK owns triage or the home is idle, and hands AFK transitions mid-cycle to the daemon without rewaking. The synchronous turn-end guard gains a --claude cooperative mode: it ignores stop_hook_active (true on every post-continuation stop, which is what re-opened the 2026-07-21 blind window), waits briefly for a watcher health proof, a live auto-arm owner claim, or a fresh rewake epoch, and re-blocks only when the auto-arm genuinely failed to establish - bounded to 3 consecutive blocks per session, safely below Claude Code's 8-block override, then a degraded allow with a visible systemMessage. Codex keeps the previous one-block loop guard byte-identically, and Pi, OpenCode, and Grok adapters are untouched. Continuity PreToolUse gate and durable wake queue are preserved; the gate's recovery guidance now names the Stop-owned re-arm and reserves manual background arms for auto-arm failure. Claude supervision protocol, harness-adapters facts, architecture, configuration, and continuity docs updated; docs/turnend-guard.md records the 2026-07-24 Claude 2.1.218 contract revalidation (tokenless multi-cycle rewake, no-dedup, timeout process-group kill, 8-block cap, interactive non-stall) and the 2.1.219 product live E2Es. Regression matrix: hermetic tests cover scope, identity, AFK, need, single-flight, translation, guard cooperation, budget, and registration; the new live E2E proves two full tokenless auto-arm rewake cycles with zero model arm commands; Pi and OpenCode Option B live E2Es pass unchanged. * no-mistakes(review): Fix Claude X-mode auto-arm continuity backstop * no-mistakes(review): Remove unsupported Claude contract-lab verification claims * no-mistakes(document): Update Claude auto-arm continuity documentation * fix(herdr): clean stale projections at session start (#996) * Clean stale Herdr projections at session start * no-mistakes(document): Document stale Herdr session-start projection cleanup * no-mistakes(review): Enforce locked exact Herdr projection cleanup * no-mistakes(review): Fail closed on unverified session lock ownership * no-mistakes(review): Serialize session lock acquisition atomically * no-mistakes(document): Align session-start and Herdr cleanup documentation * no-mistakes(document): Generalize lock-refusal diagnostics * no-mistakes(lint): Avoid reserved keyword in concurrency test * no-mistakes: apply CI fixes * no-mistakes: apply CI fixes * fix: recover Claude supervision without watcher-status gate (#1001) * fix: recover Claude supervision at session start * fix: remove Claude watcher-status command gate * no-mistakes(document): docs: remove stale continuity gate references * fix: make quota-aware profile selection agent-owned (#1018) * Replace quota dispatch selector instructions * no-mistakes(review): Align bootstrap docs with agent-owned dispatch selection * fix(bin): remove vestigial dispatch selector (#1026) * remove vestigial dispatch selector * no-mistakes(review): Synchronize isolation proof and portable shard evidence * no-mistakes(review): Correct shard history and proof archive date * no-mistakes(review): Remove reintroduced selector documentation reference * no-mistakes(document): Remove stale dispatch strategy documentation * docs(agents): drop superseded interim quota-window rule (#1039) quota-axi 0.1.13 emits schemaVersion 2 with a quotaSemantics object per provider, so the successor named in the interim rule has landed and the rule's own removal condition is satisfied. Keep the ownership clause so quota-axi remains the single owner of how model or product windows relate to bounding account windows, and drop the interim weakest-headroom instruction. The unknown-semantics case is already covered by the existing requirement to stop and report a candidate whose applicable quota data or interpretation cannot be established. Drop the matching assertion phrase from tests/fm-instruction-owners.test.sh; the retained ownership phrase still asserts. * fix(tmux): scope busy detection and recognize current Claude turns (#1049) * fix(tmux): scope Claude busy detection by harness * no-mistakes(review): Separate verified and fallback busy signatures * no-mistakes(test): Scope busy signatures to supplied harnesses * no-mistakes(document): Document harness-scoped busy detection * feat: add verified Kimi crewmate adapter (#1047) * Add verified Kimi crewmate harness adapter * no-mistakes(review): Scope Kimi moon detection to spinner lines * no-mistakes(review): Match only complete Kimi spinner rows * no-mistakes(review): Resolve Kimi binary portably before pane creation * no-mistakes(document): Align Kimi adapter documentation * no-mistakes(lint): Suppress false-positive ShellCheck warning for sourced watcher override * Fix Kimi busy spinner detection * no-mistakes(review): Recognize Kimi session-lock ancestry and holders * no-mistakes(review): Scope pending-reply Kimi busy detection by harness * no-mistakes(document): Correct Kimi spinner capture documentation * no-mistakes(document): Clarify optional Kimi spinner whitespace * no-mistakes(lint): Silence intentional pending-reply test stub warnings * test: align rebased Kimi busy fixtures * no-mistakes: apply CI fixes * Reconcile Kimi busy detection after per-harness scoping * no-mistakes(review): Clarify observed Kimi spinner whitespace contract * no-mistakes(document): Clarify Kimi harness documentation * fix: harden Kimi submission and spinner matching (#1058) * fix kimi pointer submission and spinner conformance * no-mistakes(review): Preserve Kimi submit target ownership guard * feat(bin): add guarded Kimi turn-end wake (#1059) * Add guarded Kimi turn-end hook * no-mistakes(review): Require jq before installing Kimi turn-end hook * no-mistakes(review): Expose jq inside isolated Kimi test fixtures * no-mistakes(review): Preserve Kimi config boundaries during hook removal * no-mistakes(review): Document Kimi removal newline safeguard * no-mistakes(document): Document Kimi shared-home preservation * fix(tmux): classify bordered composers across all rows (#1066) * Fix structural tmux composer reading * Verify Calm compatibility with Pi 0.82 * no-mistakes(review): Harden structural composer classification boundaries * no-mistakes(review): Refresh composer and Kimi regression fixtures * no-mistakes(review): Fail closed on unbounded composer edges * no-mistakes(review): Enforce aligned composer geometry safely * no-mistakes(review): Make composer ambiguity locale-safe * no-mistakes(review): Preserve ambiguity through composer submission * no-mistakes(review): Carry composer proof through retries * no-mistakes(document): Document structural tmux composer delivery guarantees * no-mistakes: apply CI fixes * feat(bin): add verified pi-signed runtime adapter (#1145) * feat: add verified pi-signed adapter * no-mistakes(review): Correct pi-signed maintainer verification date * no-mistakes(review): Correct remaining pi-signed verification dates * no-mistakes(review): Preserve authoritative pi-signed runtime identity * no-mistakes(document): Document pi-signed shared adapter semantics * no-mistakes: apply CI fixes * fix(pi): rearm watcher across session transitions (#1166) * fix(pi): rearm watcher across same-process session transitions Pi emits session_shutdown for ordinary /new, /resume, and /fork replacement as well as terminal quit. The primary watcher extension latched a module-level stopping flag on every shutdown, so a replacement session in the same process could not arm monitoring until Pi restarted. Own arm authority per session generation so only the active live generation may start, stop, or rearm the child. Replacement sessions can arm again without restarting Pi, stale prior-generation callbacks cannot mutate the active cycle, and real quit still blocks late rearm. * no-mistakes(review): Preserve Pi generation isolation and exit cleanup * no-mistakes(document): Correct Pi watcher transition documentation * feat: route crew dispatch using quota-window pace (#1172) * Consume quota-axi pace signals in dispatch profile array selection. Add quota-array-dispatch as the single owner of the pace-aware candidate choice, keep AGENTS.md to the intake boundary and load trigger, and cover the acceptance cases with sanitized schemaVersion 3 fixtures. * no-mistakes(review): Stop and report genuine quota dispatch ties * no-mistakes(document): Document quota pace freshness and uncertainty * fix: adapt Grok Stop continuation and harden endpoint cleanup (#1171) * fix(grok): adapt Stop continuation to runtime capability * no-mistakes(review): Reject ambiguous Grok Stop payloads * no-mistakes(review): Reject duplicate Grok fields and accept spaced tmux sessions * no-mistakes(review): Enforce exact tmux cleanup selectors * no-mistakes(test): Fix historical tmux fixture and validate Grok Stop * no-mistakes: apply CI fixes * fix: restore stock macOS Bash 3.2 brief scaffolding (#1093) * fix(brief): make DOD scaffolding parse-safe on stock macOS Bash 3.2 fm-brief.sh built each Definition-of-done block and the not-enabled Herdr declaration with `VAR=$(cat < 63/544/4068 (about 63%/60%/60% reduction). * feat(bin): inherit backend config into secondmate homes (#1219) * Inherit config/backend into secondmate homes with deliberate-override preservation Add backend to the shared inheritable config allowlist so launch, locked bootstrap, and config-push converge a primary pin into secondmate homes as each home local future-spawn default. Track last-inherited bytes in a private state provenance marker so deliberate per-home overrides survive present and absent primary convergence, keep --backend and FM_BACKEND stronger, and extend the existing inheritance tests plus docs and skill claims. * no-mistakes(review): Preserve equal unprovenanced backend overrides * no-mistakes(review): Preserve symlink overrides and verify spawn precedence * no-mistakes(review): Snapshot backend inheritance for consistent provenance * no-mistakes(review): Simplify backend inheritance to primary-authoritative convergence * no-mistakes(document): Document inherited backend override preservation * fix: restore primary-authoritative backend inheritance after document regression The document step reintroduced provenance and deliberate per-home override semantics after review had simplified config/backend to plain primary-authoritative allowlist membership. Restore the primary-always-wins path: present overwrites, absent removes, no provenance marker, and docs/tests match that contract. * no-mistakes(review): Add divergent backend precedence regression fixtures * no-mistakes(document): Document backend inheritance contract * fix(pi): remove Calm's upper version ceiling (#1226) * fix(pi): remove Calm's exclusive Pi upper-version ceiling tests/fm-calm-pi-extension.test.sh gated on a closed PI_COMPAT_VERSIONS allowlist ("0.81.1 0.82.0") that refused any other installed Pi, and docs described that range as "supported" rather than verified evidence. The Calm CHANGELOG shows no API introduced at either version, so there is no evidence for a real minimum; the presentation adapters already probe the exact method they patch rather than checking a version. Replace the allowlist with dated version evidence that never rejects a newer Pi, and make each presentation adapter degrade independently with a diagnostic if a future Pi removes its API, instead of the whole Calm extension failing to load. Rewrite the feasibility doc's "Pi 0.81.1 through 0.82.0" phrasing to state it as verified evidence, not a ceiling. * no-mistakes(review): Probe missing Calm adapter exports safely * no-mistakes(document): Document Calm's unbounded Pi compatibility * fix(bin): allow session-local todo tools in the subagent guard (#1204) * fix(guard): allow session-local todo tools in the primary The delegation-shape guard denied TaskCreate and TaskUpdate because their normalized names contain the `task` stem. Those tools write only the harness's session-local todo list, which has no executor: it spawns no agent, allocates no worktree, registers no schedule, and starts nothing that outlives the session. That is not the unaccounted work the guard exists to stop, so the stem match was a false positive, and the deny text told the primary to run bin/fm-brief.sh and bin/fm-spawn.sh to create a todo entry. Add a separately-reasoned PLAN_ONLY_TOOLS exact-name exclusion rather than widening OBSERVE_ONLY_TOOLS, whose documented contract is tools that only observe or stop existing work. Both lists stay exact-name so neither can widen by substring. Tests cover the two allowed names and six near-miss names that a substring or shortened-stem widening would release; both mutations were watched red. * no-mistakes(review): drop session-local todo tools from recommended deny list * no-mistakes: apply CI fixes * fix(session-lock): resolve Claude bg-spare ancestry to the outermost claude pid (#1206) * fix(session-lock): resolve Claude bg-spare ancestry to the outermost claude pid fm_harness_ancestry_pid() previously returned the first ancestor process whose command matched a verified harness name. Claude Code's Stop hook fires as a bg-spare worker several levels below the session's actual lock-owning claude process (hook shell -> claude bg-spare -> claude bg-pty-host -> claude -> claude(lock)), so the first match was the bg-spare worker, not the lock owner. fm_session_lock_owned_by_self() then never matched state/.lock, and the Claude Stop auto-arm silently treated its own primary session as an unrelated live owner and never armed the watcher. The walk now keeps going past a claude-named match, looking for a still more ancestral claude-named match, and stops the instant a non-match follows an already-found match (bounding it to a contiguous run rather than the literal ancestry top, so an unrelated claude-named process further up the real process tree is never mistaken for part of this session's own nested chain). Every other harness keeps the original first-match-wins behavior, since e.g. Pi's shared signed-wrapper ancestry actually holds the session at the inner engine pid, not an outer wrapper pid. Hop limit raised from 8 to 16 to cover the deeper bg-spare chain. * no-mistakes(review): Add nested-claude-ancestry regression test; fix nudge doc depth claim * no-mistakes: apply CI fixes * fix: conferma l'avvio del watcher su Windows/MSYS (#1212) * fix: confirm watcher startup on MSYS * no-mistakes(review): gate MSYS arm ready timeout, cache uname, harden locale test * no-mistakes(review): validate OpenCode ready timeout, make uname cache internal * fix(spawn): forward CLAUDE_CONFIG_DIR to claude crewmates (#1195) * fix(spawn): forward firstmate's CLAUDE_CONFIG_DIR to claude crewmates Crewmate panes are created by a long-lived tmux/herdr daemon that does not inherit firstmate's current environment. When firstmate runs under a non-default CLAUDE_CONFIG_DIR (for example a work-vs-personal subscription split), a bare `claude` in the crewmate pane fell back to the default ~/.claude store and launched unauthenticated, blocking the crewmate before it could do any work. fm-spawn now prefixes the claude launch with firstmate's own resolved CLAUDE_CONFIG_DIR when set, so the crewmate uses the same credential/config store firstmate is authenticated with. An unset value is the single-store default and adds no prefix; non-claude harnesses are unaffected. Adds three tests in fm-spawn-dispatch-profile.test.sh (forwarded-when-set, omitted-when-unset, non-claude-ignored) and pins CLAUDE_CONFIG_DIR in the test helper so launch assertions no longer depend on the developer's environment. * no-mistakes: apply CI fixes * fix: preserve dispatch identity across authentication checks (#1233) * fix: preserve dispatch harness identity * no-mistakes(review): Fix Grok counterfactual tuple validation * no-mistakes(document): Scope dispatch authentication to selected tuple * fix: restore dispatch instruction budget * no-mistakes(review): Scope dispatch authentication after candidate selection * fix(bin): normalize relative durable paths (#1256) * fix(bin): handle dash-leading harness process names (#2) * fix: handle dash-leading harness process names * no-mistakes(review): Make dash-leading harness regression hermetic * fix: preserve secondmate reply routes across relative homes Resolve relative home, data, and state inputs before durable charter generation, and fail when caller-relative directories cannot be resolved. Use absolute paths at the related spawn, AFK daemon, and X-mode cross-process handoffs so later processes cannot reinterpret them from another working directory. * no-mistakes(review): Preserve absolute overrides and normalize relative durable paths * no-mistakes(review): Normalize relative home before deriving durable paths * no-mistakes(document): Document relative durable-path normalization * no-mistakes(review): Captain: Ignore inherited CDPATH during relative path normalization * no-mistakes(lint): Fix empty CDPATH assignments for ShellCheck * refactor(skills): make Bearings chat-only by default (#1136) * Add internal status skill * no-mistakes(document): register /status skill in documentation-audiences inventory * no-mistakes(lint): replace grep|wc -l with grep -c in status skill test * test: silence literal status skill patterns * Refactor bearings default to chat-only --------- Co-authored-by: Kun Chen <3233006+kunchenguid@users.noreply.github.com> * Clarify follow-up routing during validation (#1277) * fix: honor concrete approval for project operations (#1272) * docs: add captain-approved project operation exception to hard rule 1 Firstmate stays read-only over projects by default, but when the captain clearly approves a concrete project operation and scope in the moment, firstmate may perform exactly that approved operation with its own tools. The approval is never inferred, broadened, or standing, and it does not relax the existing force, discard, unlanded-work, or merge-authority boundaries. * no-mistakes(review): Clarify captain-approved project operation boundaries * no-mistakes(document): Clarify captain-approved project operation scope * docs: cover directories and preserve the operation-or-scope alternative Widen the captain-approved project operation exception in AGENTS.md to files or directories, and restore the explicit operation-or-scope alternative that a prior pipeline auto-fix had collapsed into "and". Rework project-management SKILL.md's Remove section, which previously told firstmate to refuse project removal until a guarded helper existed; that helper was never built, so the text directly contradicted the new instruction-only exception. It now points at the exception plus the existing removal preflight it still requires unchanged. Update the one instruction-owners test assertion that hard-coded the sentence removed above, so the suite tracks current, not obsolete, text. * docs: add captain-approved project operation exception to hard rule 1 Firstmate stays read-only over projects by default, but when the captain clearly approves a concrete project operation and scope in the moment, firstmate may perform exactly that approved operation with its own tools. The approval is never inferred, broadened, or standing, and it does not relax the existing force, discard, unlanded-work, or merge-authority boundaries. * no-mistakes(review): Clarify captain-approved project operation boundaries * no-mistakes(document): Clarify captain-approved project operation scope * docs: cover directories and preserve the operation-or-scope alternative Widen the captain-approved project operation exception in AGENTS.md to files or directories, and restore the explicit operation-or-scope alternative that a prior pipeline auto-fix had collapsed into "and". Rework project-management SKILL.md's Remove section, which previously told firstmate to refuse project removal until a guarded helper existed; that helper was never built, so the text directly contradicted the new instruction-only exception. It now points at the exception plus the existing removal preflight it still requires unchanged. Update the one instruction-owners test assertion that hard-coded the sentence removed above, so the suite tracks current, not obsolete, text. * no-mistakes(review): Align project removal preflight with approved exception * no-mistakes(document): Align project removal documentation with approved exception * fix: restore removal test byte-for-byte and preserve the default sentence tests/fm-instruction-owners.test.sh had been changed to assert different text; restore it byte-for-byte to origin/main. project-management SKILL.md's Remove section now keeps the exact default "Never issue a raw removal command from Firstmate." sentence that test still asserts, immediately followed by the already-approved captain-operation-or-scope exception, so the default and the exception both stay explicit and consistent. * no-mistakes(document): Align project-write boundary documentation * fix(skills): route new project intake through secondmate scopes (#1275) * Route project intake through secondmate scopes * no-mistakes(test): Guard all main-home project registry mutations * no-mistakes(document): Consolidate secondmate routing documentation * no-mistakes: apply CI fixes * Restore new-project routing scope * no-mistakes(document): Clarify secondmate routing for new-project intake * no-mistakes: apply CI fixes * fix: scope validation corrections by accepted behavior (#1281) * fix: scope validation corrections by accepted behavior * no-mistakes(review): Classify stale delivery evidence as an autonomous correction * test: replace source assertions with behavioral coverage (#1282) * test: remove source-content assertions * no-mistakes(review): Replace source assertions with runtime behavior coverage * no-mistakes(review): Isolate Kimi task temp runtime coverage * no-mistakes(document): Refresh test cleanup documentation * no-mistakes: apply CI fixes * fix(watch): escalate busy workers with no completed turn (#1286) * fix(watch): bound how long a busy pane may run with no completed turn A busy pane (backend busy state or the harness's rendered footer) was unconditional, unbounded proof of liveness in every escalation path, so a hung foreground tool call behind a busy signature could run for hours undetected (2026-07 hibit-agent-focus-nonsteal-r1 incident: a catastrophic- backtracking regex hung one bash call for 25h behind an unchanging "Working..." footer). FM_BUSY_TURN_MAX_SECS (default 3600s) now bounds how long a busy pane may run with no completed turn (state/.turn-ended, or its spawn record before any turn has completed). Past the bound, busy_turn_over_age routes the pane through the existing wedge_timer_check, reusing the identical stale reason, escalation counter, and demand-deep-inspection marker for human inspection only - never an automatic interrupt, signal, or restart of the worker or its tool process. A completed turn resets the age. Reproduced end-to-end against the real installed Pi TUI: a foreground `sleep 999999` bash call with no timeout renders the actual busy footer, and two captures ~15s apart show the elapsed counter changing the pane hash while the same turn stays unfinished. Running the pre-fix watcher against the real captures showed it never starts a wedge timer no matter how long the pane stays busy; the fixed watcher starts and escalates the timer through the same mechanism, while the real hung process remained untouched and alive throughout. * no-mistakes(review): fix: parse enriched AFK stale reasons * no-mistakes(review): fix: preserve enriched wedges during AFK supervision * no-mistakes(review): fix: route all enriched AFK wedges * no-mistakes(document): Clarify busy-turn age supervision documentation * fix(gitignore): ignore config/ as a directory, not by exact filename (#1261) A name-by-name list of config/ entries silently stops ignoring any new or home-local file placed there, which makes the working tree read as dirty and blocks guarded sync paths that refuse to touch a dirty home. AGENTS.md already documents config/ as captain-private and gitignored as a category; this makes .gitignore match that contract. * fix(tests): replace source-content .gitignore assertion with behavioral coverage (#1304) The second assertion in fm-gitignore-config.test.sh (added by #1261) greps .gitignore for a specific spelling of the config/ ignore pattern. It fails on a semantically equivalent pattern like config/** and does not prove Git actually ignores anything, per the completed source-content-test audit. Replace it with a real git check-ignore control test on a generated unrelated path, and strengthen the existing directory-coverage test with generated unpredictable direct and nested config/ paths. * feat: bound and consolidate startup memory during stow (#1303) * Add bounded startup memory curation * no-mistakes(review): Record reproducible stow verification evidence * no-mistakes(review): Validate inherited secondmate stow evidence * no-mistakes(document): Document editable startup-memory budget propagation * no-mistakes(review): Captain: restored ADHD coverage and Doppler CLI validation * no-mistakes(review): Classify CLI-only Doppler jobs as injecting * no-mistakes(document): Refresh test-isolation documentation terminology * fix: satisfy pinned shellcheck for merge tests * docs: add Moshi mobile review mode Red proof: the pre-change public agent surface had no Moshi/mobile-mode trigger or durable Browser Preview handoff. Verification: fm-doc-audience-check passes; focused captain-translation and documentation-audience tests pass; fm-test-run --changed passes 26 of 27 selected scripts. The unrelated live Pi 0.83 Calm /export interaction timed out, while this branch changes no Calm or Pi test surface. * no-mistakes(review): Captain: align Chat View with numbered Firstmate fallback * no-mistakes(test): Documented unavailable Moshi dogfood; preserved captain checklist * no-mistakes(document): Consolidated Moshi documentation ownership --------- Co-authored-by: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Co-authored-by: Christopher McKay <101884182+karotkriss@users.noreply.github.com> Co-authored-by: Daniel Kuykendall IV Co-authored-by: Trillium Smith Co-authored-by: Unknownzed <45267749+Unknownzed@users.noreply.github.com> Co-authored-by: lhalbert Co-authored-by: AG Co-authored-by: deeto15 <92119640+deeto15@users.noreply.github.com> Co-authored-by: juniorlovestmh <272474227+juniorlovestmh@users.noreply.github.com> --- .agents/skills/afk/SKILL.md | 14 +- .agents/skills/ask-user-authority/SKILL.md | 4 +- .agents/skills/bearings/SKILL.md | 90 +- .agents/skills/bootstrap-diagnostics/SKILL.md | 7 +- .../firstmate-coding-guidelines/SKILL.md | 31 +- .agents/skills/firstmate-orca/SKILL.md | 5 +- .agents/skills/harness-adapters/SKILL.md | 152 ++- .agents/skills/mobile-mode/SKILL.md | 54 + .agents/skills/project-management/SKILL.md | 17 +- .agents/skills/quota-array-dispatch/SKILL.md | 65 + .../skills/secondmate-provisioning/SKILL.md | 7 +- .agents/skills/stow/SKILL.md | 104 +- .claude/settings.json | 12 +- .github/workflows/ci.yml | 13 +- .gitignore | 11 +- .no-mistakes.yaml | 13 + .opencode/plugins/fm-primary-watch-arm.js | 6 +- .pi/extensions/fm-calm.ts | 25 +- .pi/extensions/fm-primary-pi-watch.ts | 170 ++- .../lib/fm-calm-assistant-layout.ts | 15 +- .../lib/fm-calm-operational-user-layout.ts | 27 +- AGENTS.md | 56 +- CONTRIBUTING.md | 13 +- README.md | 50 +- bin/backends/cmux.sh | 4 +- bin/backends/herdr.sh | 20 +- bin/backends/tmux.sh | 20 +- bin/backends/zellij.sh | 4 +- bin/fm-afk-launch.sh | 22 + bin/fm-backend.sh | 188 ++- bin/fm-bootstrap.sh | 49 +- bin/fm-brief.sh | 45 +- bin/fm-claude-stop-autoarm.sh | 194 +++ bin/fm-composer-lib.sh | 11 +- bin/fm-config-inherit-lib.sh | 57 +- bin/fm-config-push.sh | 4 +- bin/fm-continuity-command-policy.mjs | 145 -- bin/fm-continuity-pretool-check.sh | 113 -- bin/fm-crew-state.sh | 17 +- bin/fm-dispatch-select.sh | 337 ----- bin/fm-doc-audience-check.sh | 269 ++++ bin/fm-guard.sh | 6 +- bin/fm-harness.sh | 17 +- bin/fm-herdr-session-cleanup.sh | 383 ++++++ bin/fm-kimi-turnend-hook.sh | 276 ++++ bin/fm-lint.sh | 33 +- bin/fm-lock.sh | 96 +- bin/fm-pending-reply-lib.sh | 12 +- bin/fm-secrets-check.mjs | 8 +- bin/fm-send.sh | 18 +- bin/fm-session-lock-lib.sh | 96 ++ bin/fm-session-start.sh | 48 +- bin/fm-spawn.sh | 210 ++- bin/fm-startup-memory-budget-lib.sh | 224 +++ bin/fm-startup-memory-budget.sh | 94 ++ bin/fm-subagent-pretool-check.sh | 25 +- bin/fm-supervise-daemon.sh | 73 +- bin/fm-supervision-instructions.sh | 7 +- bin/fm-supervision-lib.sh | 27 +- bin/fm-teardown.sh | 53 +- bin/fm-test-isolation-proof.sh | 17 +- bin/fm-test-run.sh | 90 +- bin/fm-tmux-lib.sh | 378 +++++- bin/fm-turnend-guard-grok.sh | 61 +- bin/fm-turnend-guard.sh | 197 ++- bin/fm-wake-lib.sh | 18 +- bin/fm-watch-arm.sh | 15 +- bin/fm-watch.sh | 104 +- docs/architecture.md | 39 +- docs/arm-pretool-check.md | 17 +- docs/calm-mode-feasibility.md | 51 +- docs/calm.md | 37 + docs/cd-guard.md | 6 +- docs/cmux-backend.md | 430 ++---- docs/codex-app-backend.md | 230 +--- docs/configuration.md | 90 +- docs/documentation-audiences.json | 375 +++++ docs/documentation-audiences.md | 28 + docs/examples/crew-dispatch.json | 2 +- docs/fm-test-isolation-proof.json | 216 +-- docs/fm-test-isolation-proof.md | 182 +-- docs/fm-test-portable-shards.md | 132 +- docs/herdr-backend.md | 1202 +++-------------- docs/moshi-mobile-review.md | 92 ++ docs/orca-backend.md | 137 +- docs/scripts.md | 8 +- docs/sessionstart-nudge.md | 143 +- docs/subagent-guard.md | 41 +- docs/supervision-protocols/claude.md | 38 +- docs/supervision-protocols/grok.md | 7 +- docs/supervision-protocols/opencode.md | 5 - docs/supervision-protocols/pi.md | 38 +- docs/tmux-backend.md | 169 +-- docs/turnend-guard.md | 247 ++-- docs/verification/moshi-mobile-review.md | 67 + docs/verification/runtime-backends.md | 411 ++++++ docs/verification/stow-memory.md | 217 +++ docs/verification/supervision.md | 198 +++ docs/watcher-continuity.md | 76 +- docs/wedge-alarm.md | 106 +- docs/zellij-backend.md | 271 ++-- .../fixtures/quota-array-dispatch/cases.json | 394 ++++++ .../quota-array-dispatch/schema-v3-shape.json | 103 ++ tests/fm-afk-launch.test.sh | 49 + tests/fm-arm-pretool-check.test.sh | 95 -- tests/fm-ask-user-authority.test.sh | 128 +- tests/fm-backend-cmux.test.sh | 6 +- .../fm-backend-herdr-presentation-e2e.test.sh | 2 +- tests/fm-backend-herdr.test.sh | 158 --- tests/fm-backend-orca.test.sh | 73 +- tests/fm-backend-zellij.test.sh | 22 +- tests/fm-backend.test.sh | 94 +- tests/fm-bearings-snapshot.test.sh | 30 - tests/fm-bootstrap.test.sh | 26 +- tests/fm-brief.test.sh | 266 +++- tests/fm-calm-pi-extension.test.sh | 259 +++- tests/fm-captain-translation-contract.test.sh | 357 +---- tests/fm-cd-pretool-check.test.sh | 67 - tests/fm-claude-continuity-live-e2e.test.sh | 74 - tests/fm-claude-stop-autoarm-live-e2e.test.sh | 164 +++ tests/fm-claude-stop-autoarm.test.sh | 435 ++++++ tests/fm-composer-ghost.test.sh | 348 ++++- tests/fm-continuity-pretool-check.test.sh | 125 -- tests/fm-daemon.test.sh | 157 ++- tests/fm-dispatch-select.test.sh | 327 ----- tests/fm-documentation-audiences.test.sh | 141 ++ tests/fm-gate-refuse.test.sh | 40 +- tests/fm-gitignore-config.test.sh | 47 + tests/fm-gotmp.test.sh | 44 +- tests/fm-grok-stop-live-e2e.test.sh | 219 +++ tests/fm-herdr-lab.test.sh | 4 +- tests/fm-herdr-session-cleanup-e2e.test.sh | 146 ++ tests/fm-herdr-session-cleanup.test.sh | 309 +++++ tests/fm-install-herdr.test.sh | 106 -- tests/fm-instruction-owners.test.sh | 260 ---- tests/fm-kimi-harness.test.sh | 676 +++++++++ tests/fm-lint.test.sh | 77 +- tests/fm-nm-test-contract.test.sh | 127 -- tests/fm-no-mistakes-ownership.test.sh | 39 - tests/fm-pending-reply.test.sh | 42 +- tests/fm-pi-watch-extension.test.sh | 298 ++-- tests/fm-pr-check-security.test.sh | 32 +- tests/fm-secondmate-harness.test.sh | 318 ++++- tests/fm-secondmate-lifecycle-e2e.test.sh | 1 + tests/fm-secondmate-liveness.test.sh | 26 +- tests/fm-secondmate-safety.test.sh | 6 +- tests/fm-secondmate-sync.test.sh | 10 - tests/fm-secrets-check.test.sh | 30 +- tests/fm-send-popup-settle.test.sh | 4 +- ...m-send-secondmate-marker-herdr-e2e.test.sh | 44 +- tests/fm-send-secondmate-marker.test.sh | 4 +- tests/fm-send-settle.test.sh | 4 +- tests/fm-send-strict.test.sh | 5 +- tests/fm-session-start.test.sh | 148 +- tests/fm-sessionstart-nudge.test.sh | 39 - tests/fm-spawn-dispatch-profile.test.sh | 304 ++++- tests/fm-startup-memory-budget.test.sh | 310 +++++ tests/fm-stow-contract.test.sh | 37 - tests/fm-subagent-pretool-check.test.sh | 70 +- tests/fm-supervision-instructions.test.sh | 26 +- tests/fm-teardown-endpoint-safety.test.sh | 275 ++++ tests/fm-teardown.test.sh | 10 +- tests/fm-test-isolation-proof.test.sh | 139 +- tests/fm-test-run.test.sh | 94 +- tests/fm-tmux-submit-busy.test.sh | 155 ++- tests/fm-turnend-guard.test.sh | 367 +++-- tests/fm-watch-triage.test.sh | 275 ++++ tests/fm-watcher-lock.test.sh | 94 +- tests/fm-x-mode.test.sh | 18 + tests/lib.sh | 15 +- tests/no-mistakes-required-workflow.test.sh | 96 -- tests/secondmate-helpers.sh | 5 +- tests/wake-helpers.sh | 25 +- 173 files changed, 12511 insertions(+), 7163 deletions(-) create mode 100644 .agents/skills/mobile-mode/SKILL.md create mode 100644 .agents/skills/quota-array-dispatch/SKILL.md create mode 100755 bin/fm-claude-stop-autoarm.sh delete mode 100755 bin/fm-continuity-command-policy.mjs delete mode 100755 bin/fm-continuity-pretool-check.sh delete mode 100755 bin/fm-dispatch-select.sh create mode 100755 bin/fm-doc-audience-check.sh create mode 100755 bin/fm-herdr-session-cleanup.sh create mode 100755 bin/fm-kimi-turnend-hook.sh create mode 100644 bin/fm-session-lock-lib.sh create mode 100644 bin/fm-startup-memory-budget-lib.sh create mode 100755 bin/fm-startup-memory-budget.sh create mode 100644 docs/calm.md create mode 100644 docs/documentation-audiences.json create mode 100644 docs/documentation-audiences.md create mode 100644 docs/moshi-mobile-review.md create mode 100644 docs/verification/moshi-mobile-review.md create mode 100644 docs/verification/runtime-backends.md create mode 100644 docs/verification/stow-memory.md create mode 100644 docs/verification/supervision.md create mode 100644 tests/fixtures/quota-array-dispatch/cases.json create mode 100644 tests/fixtures/quota-array-dispatch/schema-v3-shape.json delete mode 100755 tests/fm-claude-continuity-live-e2e.test.sh create mode 100755 tests/fm-claude-stop-autoarm-live-e2e.test.sh create mode 100755 tests/fm-claude-stop-autoarm.test.sh delete mode 100755 tests/fm-continuity-pretool-check.test.sh delete mode 100755 tests/fm-dispatch-select.test.sh create mode 100755 tests/fm-documentation-audiences.test.sh create mode 100755 tests/fm-gitignore-config.test.sh create mode 100755 tests/fm-grok-stop-live-e2e.test.sh create mode 100755 tests/fm-herdr-session-cleanup-e2e.test.sh create mode 100755 tests/fm-herdr-session-cleanup.test.sh delete mode 100755 tests/fm-install-herdr.test.sh delete mode 100755 tests/fm-instruction-owners.test.sh create mode 100755 tests/fm-kimi-harness.test.sh delete mode 100755 tests/fm-nm-test-contract.test.sh delete mode 100755 tests/fm-no-mistakes-ownership.test.sh create mode 100755 tests/fm-startup-memory-budget.test.sh delete mode 100755 tests/fm-stow-contract.test.sh create mode 100755 tests/fm-teardown-endpoint-safety.test.sh delete mode 100755 tests/no-mistakes-required-workflow.test.sh diff --git a/.agents/skills/afk/SKILL.md b/.agents/skills/afk/SKILL.md index 97083c8d50d..95f64b11e03 100644 --- a/.agents/skills/afk/SKILL.md +++ b/.agents/skills/afk/SKILL.md @@ -41,8 +41,8 @@ batched digest rather than per-wake injections. in as `FM_SUPERVISOR_TARGET` so the daemon injects into the captain, not its own new pane. **Never manufacture a terminal by splitting the captain's active pane** (`herdr pane split`): a split co-tenants the tab and visibly - shrinks the captain's pane (docs/herdr-backend.md "Away-mode daemon terminal - launch"). + shrinks the captain's pane (docs/herdr-backend.md "Away-mode supervisor + support"). Both paths share `bin/fm-afk-start.sh` as the daemon entry. The native path tells it that the launcher already prepared lifecycle state; the terminal-backed path lets the entry perform its existing state setup inside the new terminal. It exits immediately if the identity-backed daemon lock already names a live process, otherwise it execs `bin/fm-supervise-daemon.sh` in the foreground. @@ -84,7 +84,7 @@ The daemon constructs every current injection as the `away-supervisor` kind owne The bare `FM_INJECT_MARK` form remains accepted for legacy daemon escalations during rollout. U+2063 has no normal keyboard keystroke and survives terminal transport as UTF-8 text. This is how firstmate tells a daemon escalation apart from a real message in the same pane. -The operational prefix travels with the message text; it does not rely on harness-level typed-vs-injected detection, which is not portable across claude, codex, opencode, pi, and grok. +The operational prefix travels with the message text; it does not rely on harness-level typed-vs-injected detection, which is not portable across claude, codex, opencode, pi, pi-signed, grok, and kimi. ## Busy-guard and composer guard @@ -96,7 +96,7 @@ backend (tmux or herdr; see "Auto-discovered supervisor pane" below): - **Composer-state guard** - `inject_msg` reads the full `empty`/`pending`/`unknown` verdict from `fm_backend_composer_state` and injects only when it is affirmatively `empty`. `pending` means real unsubmitted text, while `unknown` includes an unreadable pane and a bare shell prompt left after the agent exits, so both defer. The shared `bin/fm-composer-lib.sh` owns the content decision after each backend captures and structurally identifies its own composer row. - It preserves idle bordered composers such as claude's `│ > … │` and bare agent glyphs as empty, but a bare shell glyph is unknown unless inside a genuine bordered composer box; see `docs/herdr-backend.md` "Composer-emptiness safety" for the complete contract. + It preserves idle bordered composers such as claude's `│ > … │` and bare agent glyphs as empty, but a bare shell glyph is unknown unless inside a genuine bordered composer box; see `docs/herdr-backend.md` "Composer and injection safety" for the complete contract. `pane_input_pending` remains the tested predicate for callers that only need to know whether real unsubmitted text is present, but it is insufficient for an injection-safety decision because it cannot distinguish `empty` from `unknown`. Either condition, or any composer verdict other than `empty`, defers the injection; the buffered escalation survives in `state/.subsuper-escalations` and is retried on the next housekeeping tick. @@ -110,7 +110,7 @@ If that submit cannot be confirmed, it raises a loud, rate-limited wedge alarm: an ERROR in the daemon log, a durable `state/.subsuper-inject-wedged` marker (surface it on the "while you were out" catch-up if present), a tmux status-line flash when applicable, and a configurable backend-independent active alert. -`docs/wedge-alarm.md` owns the alert channel setup and verification record. +`docs/wedge-alarm.md` owns the alert channel setup, and `docs/verification/supervision.md` "Wedge-alarm channels" owns active evidence. So a guard false-positive becomes a visible stall, never an unbounded silent no-op. ## Submit model @@ -225,14 +225,14 @@ the operational prefix lets firstmate distinguish it from a real captain message backends, including zellij, orca, and cmux, are not yet supported as supervisor backends; the daemon refuses loudly at startup instead of misapplying tmux primitives to a pane that isn't one - (docs/herdr-backend.md "Away-mode daemon: herdr supervisor-pane support"). + (docs/herdr-backend.md "Away-mode supervisor support"). ## Stale-artifact lifecycle Treat `state/.subsuper-escalations`, its `.since` sidecar, and `state/.subsuper-inject-wedged` as session-scoped delivery artifacts, not as the durable work record. Always enter through `bin/fm-afk-launch.sh`, which clears prior-session artifacts only for a fresh entry and preserves the current session's buffer on refresh. Always exit through `bin/fm-afk-launch.sh stop`, which keeps `state/.afk` present through the daemon's shutdown flush and clears it last. -`docs/herdr-backend.md` "Stale-artifact lifecycle fix" owns the mechanism and verification evidence. +`docs/herdr-backend.md` "Away-mode supervisor support" owns the current mechanism, and `docs/verification/runtime-backends.md` "Away-mode transport" owns active evidence. ## Reliability properties diff --git a/.agents/skills/ask-user-authority/SKILL.md b/.agents/skills/ask-user-authority/SKILL.md index d4b63d525bf..38761e6d98a 100644 --- a/.agents/skills/ask-user-authority/SKILL.md +++ b/.agents/skills/ask-user-authority/SKILL.md @@ -19,7 +19,9 @@ The concise standing authority boundary remains always loaded in `AGENTS.md` sec With `yolo` off, every ask-user finding belongs to the captain, and the remaining steps structure that escalation rather than authorize an autonomous answer. 2. Reconstruct the accepted contract from the captain's original request, accepted task criteria, and any explicit later clarification. Reviewer language cannot amend that contract. -3. Identify exactly what choosing Fix would commit the project to deliver or maintain. +3. Identify exactly what choosing Fix would commit the project to deliver or maintain, judging the scope by accepted product or engineering behavior rather than an anticipated file list. + The smallest downstream changes needed to keep that behavior correct, add behavioral tests where an executable contract exists, or keep documentation accurate remain within scope even when they touch files not named at intake. + Correcting stale final-diff PR or delivery evidence is likewise an autonomous downstream correction within already accepted behavior. 4. Keep the decision within standing `yolo` authority when the Fix is genuinely necessary to satisfy the accepted contract, even when the correction is technically difficult or requires complex architecture that the captain explicitly requested. 5. Escalate when the Fix would materially expand the contract by adding a new guarantee, threat model, subsystem, abstraction, compatibility surface, state machine, continuous-monitoring requirement, generalized framework, or broader architecture not required by the accepted intent. 6. Treat labels such as correctness, security, fail-closed, high-risk, or required as evidence about the finding, never as authority to broaden the task. diff --git a/.agents/skills/bearings/SKILL.md b/.agents/skills/bearings/SKILL.md index b2804c7290b..42990edd04f 100644 --- a/.agents/skills/bearings/SKILL.md +++ b/.agents/skills/bearings/SKILL.md @@ -1,6 +1,9 @@ --- name: bearings -description: Generate a "pick up where I left off" status report from firstmate's live fleet state. Use when the captain invokes /bearings or asks for a bearings report, morning brief, status report, catch-up, "where did I leave off", or "what's in the works". Reads bounded local fleet state cheaply, optionally checks open PRs when requested, composes a scannable dated report to data/status-report-.md, and surfaces a concise version in chat; it is read-mostly and must not tear down, merge, or mutate task state as a side effect of producing the brief. +description: >- + Generate a "pick up where I left off" fleet digest from firstmate's live fleet state. + Use when the captain invokes /bearings or asks for a bearings report, morning brief, status report, catch-up, "where did I leave off", or "what's in the works". + Plain /bearings is chat-only by default, while /bearings file explicitly writes the dated data/status-report-.md artifact; live PR enrichment remains opt-in and composes with file mode. user-invocable: true metadata: internal: true @@ -8,42 +11,58 @@ metadata: # bearings -Generate a complete standalone snapshot from the fleet's current state, so the captain can resume in one read after a break, a night, or a context reset. -The deliverable is a dated markdown file plus a concise chat summary that each stand on the current snapshot rather than an earlier report. -This skill is read-mostly. -It reads fleet state and writes exactly one report file. -It never tears down a task, merges a PR, dispatches new work, or mutates any task state as a side effect of producing the brief - those belong to the captain's explicit word and the normal task lifecycle. +Generate a complete current snapshot from the fleet's current state, so the captain can resume in one read after a break, a night, or a context reset. +Plain `/bearings` returns only the concise four-section chat digest. +Only `/bearings file` writes the dated markdown report artifact and then returns the concise four-section chat digest linked to that report. +This skill is operationally read-only in both modes. +It never tears down a task, merges a PR, dispatches new work, steers a worker, answers a decision, cleans up work, mutates backlog or task state, or writes any file except the single dated report in explicit file mode. + +## Invocation modes + +- Plain `/bearings` gathers a fresh bounded snapshot and renders the four-section chat digest without creating, deleting, reading, or replacing `data/status-report-.md`. +- `/bearings file` gathers a fresh bounded snapshot, replaces today's `data/status-report-.md` from scratch, and renders the four-section chat digest with a link or path to that report. +- Treat `file` only as an explicit invocation option in the slash command. +- Do not treat natural-language requests such as "write a report", "save this", "persist it", or "make a file" as file mode unless the invocation explicitly includes the standalone `file` option. +- When the captain asks to include PRs, pass the snapshot command's live-PR opt-in. +- `/bearings include PRs` remains chat-only and makes the live-PR opt-in. +- `/bearings file include PRs` writes the dated report and makes the live-PR opt-in. ## What it does 1. **Gather live fleet state with one deterministic command.** - Run `bin/fm-bearings-snapshot.sh` and read its compact output. - It is the single bounded, deterministic source for this report and renders TOON by default. - Do not hand-probe the snapshot schema and do not make ad-hoc `gh-axi`/`gh` calls to assemble fleet facts; this command already assembles them. + Run `bin/fm-bearings-snapshot.sh` at invocation time and read its compact output. + It is the single bounded, deterministic fleet-state source for Bearings and renders TOON by default. + Do not create or consult a second fleet-state reader, parser contract, status-event-tail interpretation, visible-session recap, ad-hoc project probe, or ad-hoc `gh-axi`/`gh` query. The command's header and `--help` output own its exact fields, bounds, opt-ins, and output contract. - When the captain asks to include PRs, use the command's live-PR opt-in; otherwise keep the default local-only read. - If the command is unavailable, fall back to `bin/fm-fleet-snapshot.sh --json` and `bin/fm-crew-state.sh `; never infer current state from a raw `tail` of `state/.status`, which is append-only wake-event history whose last line goes stale. - For registered secondmates, use the snapshot's structured-home classification and provenance; a parent event or bounded terminal contradiction is fallback evidence, never authority over readable structured home state. - Structured captain-held decisions come from `decision-hold-lifecycle` and appear under `decisions_open`; do not scrape reports or visual-review artifacts to supplement them. - A queued item under `gates` only becomes "next work" when its blocker is gone and its time/date gate has arrived; until then it stays queued with the reason. - The `(main-inventory)` gate is an action-free integrity warning rather than queued work: render it under Charted Next with the related `omitted` disclosure, never invent an Underway row from backlog-only state, and never move it into Captain's Call. - -2. **Compose the detailed report file around the four-section spine, adding the richer detail the chat leaves out.** - The gather step is deterministic; your judgment is scoped to the last mile only - ranking the command's facts by what matters right now and writing the scannable prose. + Keep the default local-only read unless the captain asks to include PRs. + For registered secondmates, use the snapshot's structured-home classification and provenance. + A parent event or bounded terminal contradiction is fallback evidence, never authority over readable structured home state. + Structured captain-held decisions come from `decision-hold-lifecycle` and appear under `decisions_open`. + Do not scrape reports, visual-review artifacts, raw status-event tails, or visible conversation history to supplement current state. + A queued item under `gates` only becomes "next work" when its blocker is gone and its time/date gate has arrived. + Until then it stays queued with the reason. + The `(main-inventory)` gate is an action-free integrity warning rather than queued work. + Render it under Charted Next with the related `omitted` disclosure, never invent an Underway row from backlog-only state, and never move it into Captain's Call. + +2. **Compose the four-section chat digest from the fresh snapshot.** + The gather step is deterministic; your judgment is scoped to ranking the command's facts by what matters right now and writing scannable captain-facing prose. + The chat response uses the four complete sections in the chat-response contract below, in the same order, each always present. + Plain mode stops here and writes no report artifact. + +3. **In explicit file mode only, compose and replace the detailed report file.** + The report uses the same four complete sections as the chat, in the same order, and adds the detail the chat omits. Never read an earlier `data/status-report-*.md` to decide what to omit, include, describe as changed, or call current. - The report uses the same four complete sections as the chat (see the chat-response contract below), in the same order, each always present, and adds the detail the chat omits: + Write the full report to `data/status-report-.md` using today's date. + If today's file already exists, delete it first, then create a new file from scratch. + This is the only write allowed by the skill. + The detailed report includes: - **Title** - `# Bearings - ` (use "Morning status" only when the captain specifically asks for a morning brief), followed by two or three sentences framing where things stand. - **Captain's Call** - every open decision summarized with its options from the structured decision record, plus each PR ready to merge and each needed credential or login, every PR with the full `https://...` URL, never a bare `#number`. - **Recently Landed** - the bounded current recent-completions baseline from structured state across the main fleet and every registered secondmate home, rendered in full on every run. - - **Underway** - each live direct report making progress, with its current state, and the plans / main pickup pointers worth reopening (`data//report.md` files, `.lavish/*.html` boards). + - **Underway** - each live direct report making progress, with its current state, and the plans or main pickup pointers worth reopening (`data//report.md` files, `.lavish/*.html` boards). - **Charted Next** - queued or gated work, including any main-inventory integrity warning, with each item's blocker, date, or integrity reason. - -3. **Write the dated report file so it persists, then surface the mandatory four-section digest in chat.** - - Write the full report to `data/status-report-.md` using today's date. - This is the required artifact; it lives in gitignored `data/`. - If today's file already exists, delete it first, then create a new file from scratch. - - The chat response is the concise four-section digest defined by the contract below: materially shorter than the report file, complete as a current snapshot, internally consistent with the file, and linked to that file for the full picture. - - For a richer review surface, optionally offer a Lavish board with `lavish-axi` when the report has enough structure to deserve one, but the markdown file is the required artifact and the four-section chat digest is the required minimum. + After writing the file, return the concise four-section chat digest and include the report path or link without adding a fifth section. + For a richer review surface, optionally offer a Lavish board with `lavish-axi` when the report has enough structure to deserve one, but only after the required digest is ready. ## Chat-response contract @@ -62,22 +81,27 @@ Every `/bearings` chat response renders EXACTLY these four sections, in THIS ord Rules that keep the contract unambiguous: - Every section ALWAYS renders, even when empty, with its short empty-state sentence; never omit a section. -- Every report and chat digest is a complete current snapshot, never a delta against a prior report. +- Every chat digest and file-mode report is a complete current snapshot, never a delta against a prior report. - Recently Landed always renders the bounded current baseline, even when the same completions appeared in an earlier report. - The four buckets are mutually exclusive, so every item is forced into exactly one: needs-your-action is Captain's Call, done is Recently Landed, self-progressing is Underway, and not-yet-started work or an action-free fleet-integrity warning is Charted Next. - The strict boundary keeps action-free items OUT of Captain's Call: a working or validating task, a queued item blocked on another task or a date, landed work, a completed scout's report pointer, a declared `paused:` external wait, and a bare recorded PR with no merge-ready signal each belong to one of the other three sections, never Captain's Call. - A secondmate's own row appears Underway only for `active_child_work`; `externally_held` belongs in Charted Next, and `unknown` belongs there as an unavailable-state gate unless its reason requires the captain's action. - Do not suppress separately projected decisions, landed records, or gates from a `partial-structured` home merely because that secondmate's own row is `unknown`. -- The chat follows `AGENTS.md` section 9 and carries one scannable line per item, each PR as the full `https://...` URL; detailed decisions, plans, full gate reasons, and evidence live only in the report file, which the chat links to, so the chat stays materially shorter than that file. +- Include the required direct address to the captain inside one item or empty-state sentence. +- Every PR appears as the full `https://...` URL; a shorthand `#number` is fine only as a back-reference after the full URL has already appeared in the same digest. +- The chat follows `AGENTS.md` section 9 and carries one scannable line per item. +- Detailed decisions, plans, full gate reasons, and evidence belong in the file only when file mode is explicit, so plain chat stays concise and file-mode chat stays materially shorter than that file. +- In file mode, include the report path or link inside the four-section digest without adding another heading. ## Tone and content rules -- This report is a private, captain-facing internal artifact that lives in gitignored `data/`, so unlike normal captain chat it MAY reference task ids, PR URLs, and repo names - the captain works with these directly and needs them to resume; keep it organized and scannable, not a raw dump. -- Every PR reference is a full `https://...` URL, never a bare `#number`; a shorthand `#number` is fine only as a back-reference after the full URL has already appeared in the same report. +- The optional file-mode report is a private, captain-facing internal artifact that lives in gitignored `data/`, so unlike normal captain chat it MAY reference task ids, PR URLs, and repo names. +- The captain works with those directly and needs them to resume; keep the report organized and scannable, not a raw dump. +- Every PR reference is a full `https://...` URL, never a bare `#number`. - Never include PHI or secret values; the report is an operational artifact, but it is still subject to the same security and compliance rules that govern everything else in this fleet. ## Supervision discipline -This skill is read-mostly and changes no fleet state. -Do not tear down a task, merge a PR, dispatch queued work, or mutate any `state/` or `data/` file other than the single report file as a side effect of generating the brief. +This skill changes no fleet state. +Do not tear down a task, merge a PR, dispatch queued work, steer a worker, answer a queued decision, clean up work, or mutate any `state/` or `data/` file other than the single report file in explicit file mode. If the state you read suggests an action - a PR ready to merge, a queued item whose gate has arrived, or a needs-decision finding - name it in its section and leave the action to the normal lifecycle and configured authority rather than taking it from inside this skill. diff --git a/.agents/skills/bootstrap-diagnostics/SKILL.md b/.agents/skills/bootstrap-diagnostics/SKILL.md index 5379ec9702f..477980b8df1 100644 --- a/.agents/skills/bootstrap-diagnostics/SKILL.md +++ b/.agents/skills/bootstrap-diagnostics/SKILL.md @@ -2,7 +2,7 @@ name: bootstrap-diagnostics description: >- Agent-only handling playbook for session-start bootstrap diagnostics. - Use whenever the session-start digest's bootstrap section prints an actionable diagnostic line - MISSING, MISSING_MANUAL, BACKEND_INVALID, NEEDS_GH_AUTH, TANGLE, CREW_DISPATCH invalid, FLEET_SYNC, PR_CHECK_MIGRATION, SECONDMATE_SYNC, SECONDMATE_LIVENESS, NUDGE_SECONDMATES, or FMX - or when a standalone bin/fm-bootstrap.sh run prints one of those lines. + Use whenever the session-start digest's bootstrap section prints an actionable diagnostic line - MISSING, MISSING_MANUAL, BACKEND_INVALID, NEEDS_GH_AUTH, TANGLE, STARTUP_MEMORY_BUDGET, CREW_DISPATCH invalid, FLEET_SYNC, PR_CHECK_MIGRATION, SECONDMATE_SYNC, SECONDMATE_LIVENESS, NUDGE_SECONDMATES, or FMX - or when a standalone bin/fm-bootstrap.sh run prints one of those lines. A silent bootstrap section, or a BOOTSTRAP_INFO fact, means no skill load. user-invocable: false metadata: @@ -20,14 +20,15 @@ When any diagnostic needs captain attention, report the plain consequence and re For `treehouse`, this also covers an installed version whose `treehouse get` lacks `--lease`; treat it as an upgrade request. For `no-mistakes`, this also covers an installed version older than 1.31.2, because crewmate validation briefs delegate gate mechanics to no-mistakes' version-matched guidance. For `tasks-axi`, this also covers an installed build that fails the compatibility probe (`docs/configuration.md` "Backlog backend" owns the definition); `config/backlog-backend=manual` only suppresses the verbose `BOOTSTRAP_INFO: tasks-axi available` fact, not this missing-tool report. - For `quota-axi`, bootstrap requires it because every crew-dispatch profile array calls it automatically; `bin/fm-dispatch-select.sh` still selects uniformly from valid candidates with OS-backed randomness when quota data is unavailable. + For `quota-axi`, bootstrap requires it because firstmate reads its current output directly before resolving every crew-dispatch profile array; without it, report the missing requirement and do not choose around an unexamined candidate. - `MISSING_MANUAL: (instructions: )` - tell the captain why the tool is required and give them the printed instructions URL, but do not pass the tool to `bin/fm-bootstrap.sh install`; wait for the captain to complete the manual installation, then rerun session start to confirm the dependency is present. - `BACKEND_INVALID: (known: )` - the resolved runtime backend has no verified dependency or lifecycle contract, so do not dispatch work until the invalid `FM_BACKEND` or `config/backend` value is corrected to one of the listed backends. - `NEEDS_GH_AUTH` - ask the captain to run `! gh auth login` (interactive; you cannot run it for them). - `TANGLE: ` - the primary checkout is stranded on a feature branch instead of its default branch; `AGENTS.md` section 8 explains why this guard exists and what it protects. The work is safe on that branch ref; restore the primary to its default branch with the printed `git -C checkout `, then re-validate that branch in a proper worktree. This is the only sanctioned firstmate-initiated git write to the primary, and it is a non-destructive branch switch that strands nothing. -- `CREW_DISPATCH: invalid config/crew-dispatch.json - ` - the optional dispatch profile file exists but failed low-cost bootstrap validation; continue with the normal fallback chain, resolve and pass the chosen fallback harness explicitly while the file remains present, fix the malformed schema, unverified harness name, unknown selector, or invalid harness/effort pair when convenient, and do not select a bad profile. +- `STARTUP_MEMORY_BUDGET: invalid config/startup-memory-budget - ` - the visible startup-memory budget is not a safe one-line positive decimal file; do not infer the default or propagate it. Correct the local primary file, then rerun session start so the normal convergence path can deliver the validated value to secondmate homes. +- `CREW_DISPATCH: invalid config/crew-dispatch.json - ` - the optional dispatch profile file exists but failed low-cost bootstrap validation; stop profile-based dispatch, report the actionable error, and require correction of the malformed schema, unverified harness name, or invalid harness/effort pair rather than falling back around it or selecting a bad profile. - `FLEET_SYNC: : skipped: ` - a benign one-off skip (offline, no origin, local-only); bootstrap continued, investigate only if it blocks work. A skip can also report the bounded fleet-refresh timeout (`FM_FLEET_SYNC_BOOTSTRAP_TIMEOUT`, or a fleet-size-aware default with a 20 second floor); a timeout never blocks startup. - `FLEET_SYNC: : recovered: ` - the clone had drifted onto a clean detached HEAD holding no unique commits and the sync self-healed it (re-attached the default branch and fast-forwarded); no action needed, it is reported only so the self-heal is visible. diff --git a/.agents/skills/firstmate-coding-guidelines/SKILL.md b/.agents/skills/firstmate-coding-guidelines/SKILL.md index b58e6ea41f4..8bbb275dae8 100644 --- a/.agents/skills/firstmate-coding-guidelines/SKILL.md +++ b/.agents/skills/firstmate-coding-guidelines/SKILL.md @@ -3,7 +3,7 @@ name: firstmate-coding-guidelines description: >- Agent-only reference for changing firstmate's shared, tracked material per AGENTS.md section 1. Use before editing any of that material, whether working as firstmate directly or as a crewmate briefed on a firstmate-repo task. - Covers the knowledge-placement decision tree, the one-owner rule for contracts, the inline-stub pattern for content moved into a skill, AGENTS.md size discipline, trigger hygiene for new skills, and repo style rules (one sentence per line, plain dash, no agent co-author, shellcheck-clean bin scripts, colocated tests, and backend-verification evidence). + Covers the knowledge-placement decision tree, the one-owner rule for contracts, the inline-stub pattern for content moved into a skill, AGENTS.md size discipline, trigger hygiene for new skills, and repo style rules (one sentence per line, plain dash, no agent co-author, shellcheck-clean bin scripts, colocated tests, and maintainer-verification evidence). user-invocable: false metadata: internal: true @@ -23,13 +23,20 @@ Before writing a new fact anywhere in this repo, ask where it belongs, in this o If yes: `AGENTS.md`, inline. 2. Does the agent need it only in a nameable situation - a spawn, a recovery, a specific wake type, a specific lifecycle step? If yes: an agent-only skill under `.agents/skills/`, plus a one-line trigger pointer left inline in `AGENTS.md` (usually section 13). -3. Is it human/reference detail - a wire format, a verification record, a mechanism narrative, an incident writeup? - If yes: `docs/`. -4. Is it mechanics - exact flags, exact commands, exact paths? - If yes: the script's own header comment plus its `--help` output, not prose in `AGENTS.md` or a skill. +3. Is it public product, setup, or user/operator reference? + If yes: the surface classified for that audience in [`docs/documentation-audiences.md`](../../../docs/documentation-audiences.md), limited to current behavior, setup, supported limits, stable invariants, concise rationale, and current verification entry points. +4. Is it contributor/maintainer architecture? + If yes: the classified maintainer-architecture owner for stable ownership, extension points, mechanism boundaries, and safety rationale. +5. Is it active reusable verification for a current guarantee? + If yes: an explicitly classified maintainer-verification record may keep current dates, versions, exact commands, and exact output. +6. Is it task or incident evidence - chronology, transcripts, branches, temporary paths, failed hypotheses, or delivery proof? + If yes: keep it in the private task report or PR evidence by default, after distilling every unique current fact into its authoritative owner. +7. Is it mechanics - exact flags, exact commands, exact paths? + If yes: the script's own header comment plus its `--help` output, not prose in `AGENTS.md`, a skill, or a second documentation owner. Stop at the first tier that answers yes. Do not place a fact at a more convenient tier than the one this tree gives you. +The machine-consumed inventory in [`docs/documentation-audiences.json`](../../../docs/documentation-audiences.json) is the single classification owner for maintained prose surfaces; do not add parallel front matter or a second audience list. ## One-owner rule @@ -74,6 +81,13 @@ Mark an axis not applicable only after inspecting its integration surface, and u For critical safety, routing, startup, and supervision infrastructure, prefer deterministic and idempotent enforcement over relying on agent memory alone. Keep instructions as the authority and discovery layer, but make repeated execution converge safely and make invalid or unsafe states fail closed wherever the runtime can enforce them. +## Documentation change review + +For every changed maintained prose surface, identify its inventory audience, authoritative owner, current-behavior relevance, destination for supporting evidence, and any unique safety fact that removal could lose. +Move or delete evidence only after the current owner and regression pointer are verified. +After all documentation, review-fix, and lint-fix commits, review the complete branch diff again against those criteria rather than reviewing only the latest commit. +Run `bin/fm-doc-audience-check.sh`; it enforces classification, README setup routing, local link targets, and owner pointers without keyword-linting legitimate evidence prose. + ## Repo style rules - Put one full sentence per line in tracked Markdown. @@ -83,6 +97,7 @@ Keep instructions as the authority and discovery layer, but make repeated execut - `bin/*.sh` and `bin/backends/*.sh` must pass `shellcheck`. - Run `bin/fm-lint.sh` before treating a script change as done; it is the single owner of the lint definition (file set, config, and pinned shellcheck version) that CI and the no-mistakes pre-push gate both invoke, and it refuses to run under any other shellcheck version. - Colocate tests with the existing pattern in `tests/`, name them `.test.sh`, and extend an existing script rather than inventing a new runner. -- A backend-verification doc (`docs/*-backend.md`) records empirical facts, not assumptions. -- Include the date, version, exact commands run, and exact output. -- Write incidents the same way, as evidence, not narrative alone. +- Tests must exercise behavior through an executable or public interface and must never assert implementation-source bytes, including through parsers, regexes, snapshots, or indirect wrappers. +- A maintainer-verification record under `docs/verification/` records active empirical facts, not assumptions or task chronology. +- Include the date, version, exact commands run, and exact output needed to support the current guarantee. +- Keep incident chronology and delivery evidence in private task reports or PR evidence unless a concise rationale is required to maintain a current safety boundary. diff --git a/.agents/skills/firstmate-orca/SKILL.md b/.agents/skills/firstmate-orca/SKILL.md index 854d4979399..d8d50b07b47 100644 --- a/.agents/skills/firstmate-orca/SKILL.md +++ b/.agents/skills/firstmate-orca/SKILL.md @@ -13,10 +13,11 @@ It does not replace `AGENTS.md`, `docs/orca-backend.md`, or `harness-adapters`. Orca is a runtime backend, not an agent harness. The runtime backend owns the task endpoint and, for Orca, the task worktree. -The harness is the agent process launched inside that endpoint, such as `claude`, `codex`, `opencode`, `pi`, or `grok`. +The harness is the agent process launched inside that endpoint, such as `claude`, `codex`, `opencode`, `pi`, `pi-signed`, `grok`, or `kimi`. Load `harness-adapters` for harness-specific launch, interrupt, resume, trust-dialog, and skill-invocation facts. -Implementation details, metadata fields, teardown guarantees, limitations, and smoke evidence live in `docs/orca-backend.md`. +Implementation details, metadata fields, teardown guarantees, and limitations live in `docs/orca-backend.md`. +`docs/verification/runtime-backends.md` "Orca" owns active smoke evidence. Prefer the `bin/fm-*` helpers over raw `orca` commands. Use raw `orca` only when the helper surface cannot answer the inspection question, and keep the recorded firstmate metadata as the task identity. diff --git a/.agents/skills/harness-adapters/SKILL.md b/.agents/skills/harness-adapters/SKILL.md index e4cdb0263b9..0ae4ee05b52 100644 --- a/.agents/skills/harness-adapters/SKILL.md +++ b/.agents/skills/harness-adapters/SKILL.md @@ -1,6 +1,6 @@ --- name: harness-adapters -description: Agent-only reference for firstmate harness operations. Use before spawning or recovering a crewmate or secondmate, handling a trust dialog, sending a harness-specific skill invocation, interrupting or exiting an agent, resuming an exited agent, or verifying a new harness adapter. Contains verified facts for claude, codex, opencode, pi, and grok. +description: Agent-only reference for firstmate harness operations. Use before spawning or recovering a crewmate or secondmate, handling a trust dialog, sending a harness-specific skill invocation, interrupting or exiting an agent, resuming an exited agent, or verifying a new harness adapter. Contains verified facts for claude, codex, opencode, pi, pi-signed, grok, and kimi. user-invocable: false metadata: internal: true @@ -12,6 +12,7 @@ Use this reference before any harness-specific firstmate operation: spawn, recov Crewmates default to the same harness firstmate is running on unless `config/crew-harness` records an adapter name. Optional dispatch profiles in `config/crew-dispatch.json` can override that static default for one crewmate or scout dispatch by selecting concrete harness, model, and effort axes at intake. +When a matched rule or default is a profile array, load `quota-array-dispatch` for the pace-aware candidate choice after this skill establishes harness and model/provider facts. The captain may override that file at session start or later; a per-task instruction such as "run this one on codex" overrides it for that dispatch only. `default` means mirror firstmate's own harness. @@ -25,7 +26,7 @@ If `config/crew-harness` is unset or `default`, there is no concrete value to in Inheritance also copies the literal `config/crew-dispatch.json` file, so secondmates apply the same best-fit profile rules for their own crewmates. Each adapter splits into mechanics and knowledge. -The per-task mechanics, including launch command, autonomy flag, and crewmate turn-end hook, live in `bin/fm-spawn.sh`. +The per-task mechanics, including launch command, autonomy flag, and any enabled crewmate turn-end hook, live in `bin/fm-spawn.sh`. The primary-session "no turn ends blind" guard contract and harness hook installation paths live in `docs/turnend-guard.md`. The primary-session watcher wake protocols are rendered from `docs/supervision-protocols/` by `bin/fm-supervision-instructions.sh`. The supervision knowledge lives here: busy signature, exit command, interrupt, dialogs, resume behavior, skill invocation, and quirks. @@ -38,6 +39,7 @@ If the captain asks for a new harness, propose verifying it first: spawn a trivi ## Detection `bin/fm-harness.sh` prints firstmate's own harness, using verified env markers first and then process ancestry. +Within the Pi family, only the exact launch-boundary marker `FM_PI_HARNESS=pi-signed` alongside `PI_CODING_AGENT=true` selects the signed identity; unmarked shared launcher ancestry remains `pi`. `bin/fm-harness.sh crew` resolves the effective crewmate harness from `config/crew-harness` (absent or `default` -> own). `bin/fm-harness.sh secondmate` resolves the secondmate-launch harness through the chain `config/secondmate-harness` -> `config/crew-harness` -> own, so an unset `config/secondmate-harness` matches the crew harness. `bin/fm-spawn.sh` uses `crew` mode for a crewmate/scout launch and `secondmate` mode for a `--secondmate` launch, re-resolving on every spawn so the split is durable across respawns; an explicit per-spawn harness arg overrides either. @@ -50,17 +52,20 @@ Use that value for interrupt, exit, resume, and skill-invocation facts. ## Primary turn-end guard -Every verified primary harness has an empirically validated hook path for the "no turn ends blind" guard. +The primary integrations for `claude`, `codex`, `opencode`, `pi`, `pi-signed`, and `grok` have empirically validated hook paths for the "no turn ends blind" guard. `claude` and `codex` block directly through Stop hooks that preserve exit status 2 and stderr from `bin/fm-turnend-guard.sh`. -`opencode`, `pi`, and `grok` expose passive lifecycle callbacks for this purpose, so their tracked primary adapters force one bounded follow-up or resume when the shared predicate blocks. -The exact hook files, commands, validation transcripts, scoping rules, and fail-open tradeoffs are owned by `docs/turnend-guard.md`. +`opencode`, `pi`, and `pi-signed` expose passive lifecycle callbacks and force one bounded follow-up when the shared predicate blocks. +Grok selects native blocking or its pre-native bounded resume fallback from the exact running Stop payload; [`docs/turnend-guard.md`](../../../docs/turnend-guard.md) owns that contract. +Kimi is outside the primary turn-end guard scope, while `docs/turnend-guard.md` owns its separate guarded global hook for crew wake signals. +The exact hook files, commands, scoping rules, and fail-open tradeoffs are owned by `docs/turnend-guard.md`. +`docs/verification/supervision.md` "Turn-end guard" owns active validation evidence. When changing any primary turn-end hook, validate the real harness behavior in a scratch project or throwaway home before trusting it, then update that doc and the relevant concise fact below. ## Primary pre-arm (PreToolUse) seatbelt -Every verified primary harness also has a wired PreToolUse-equivalent hook that denies a watcher-arm anti-pattern (shell `&`, truncating pipe, bundling, broad `pkill -f fm-watch`) before it runs. +The primary integrations for `claude`, `codex`, `opencode`, `pi`, `pi-signed`, and `grok` also have wired PreToolUse-equivalent hooks that deny a watcher-arm anti-pattern (shell `&`, truncating pipe, bundling, broad `pkill -f fm-watch`) before it runs. `claude` and `codex` block directly through PreToolUse hooks; `grok` blocks the same way but requires every `$VAR` reference in its hook `command` string to carry an inline `:-default` or it fails to launch the hook entirely. -`opencode` and `pi` block by throwing from `tool.execute.before` / returning `{block: true}` from `tool_call`. +`opencode`, `pi`, and `pi-signed` block by throwing from `tool.execute.before` / returning `{block: true}` from `tool_call`. The exact hook files, commands, output-shaping quirks (Claude Code only honors the deny when stdout is empty), and validation transcripts are owned by `docs/arm-pretool-check.md`. When changing any watcher-arm PreToolUse hook, validate the real harness behavior in a scratch project before trusting it, then update that doc. ## Primary delegation-shape guard @@ -79,22 +84,23 @@ The subagent tool presents to the model as `Agent`, and on Claude Code 2.1.217 b AGENTS.md section 3 remains the behavioral owner for session start, while tracked native adapters invoke `bin/fm-sessionstart-nudge.sh` as an idempotent enforcement layer. The wrapper prints one canonically typed `session-start` instruction to run `bin/fm-session-start.sh`; it never runs the digest, wake drain, bootstrap sweeps, lock, or supervision arm itself. -Full mechanics, scoping, dated commands, payloads, and fail-open evidence live in `docs/sessionstart-nudge.md`. +Full mechanics, scoping, and fail-open behavior live in `docs/sessionstart-nudge.md`. +`docs/verification/supervision.md` "Native session-start delivery" owns active dated commands, payloads, and evidence. - `claude`: verified native `SessionStart` stdout injection; `.claude/settings.json` matches `startup`, `resume`, and `clear`, but not `compact`. - `codex`: verified on 0.144.4; `.codex/hooks.json` receives `source=startup`, and wrapper stdout reaches model context. - `opencode`: verified on 1.17.18; `session.created` plus `client.session.promptAsync` starts the nudge turn in the TUI, while `opencode run` remains fail-open headless. -- `pi`: verified native `session_start`; the existing primary extension handles `startup`, `new`, and `resume` and uses `pi.sendMessage` to inject context without racing a positional launch prompt. +- `pi` and `pi-signed`: verified native `session_start`; the existing primary extension handles `startup`, `new`, and `resume` and uses `pi.sendMessage` to inject context without racing a positional launch prompt. - `grok`: the 0.2.103 project `SessionStart` event fires with `source=new`, but stdout does not reach model context; the tracked project hook remains fail-open, and a global token-guarded fallback requires a captain decision. ## Primary watcher supervision At session start, `bin/fm-session-start.sh` prints exactly one watcher supervision block for the detected primary harness. Do not substitute another harness's wait shape when resuming supervision. -Claude and Grok use tracked background-notify cycles around `bin/fm-watch-arm.sh`. +Claude's Stop `asyncRewake` hook (`bin/fm-claude-stop-autoarm.sh`) owns tokenless re-arm around `bin/fm-watch-arm.sh`, and Grok uses tracked background-notify cycles around `bin/fm-watch-arm.sh`. Codex uses bounded foreground checkpoints through `bin/fm-watch-checkpoint.sh` because Codex cannot reason while a foreground tool call is running. OpenCode uses `.opencode/plugins/fm-primary-watch-arm.js`, which coordinates with the turn-end guard plugin and wakes the TUI with `client.session.promptAsync`. -Pi uses the tracked `.pi/extensions/fm-primary-turnend-guard.ts` plus the tracked `.pi/extensions/fm-primary-pi-watch.ts`, both project-local extensions Pi auto-discovers once trusted. +Pi and pi-signed use the tracked `.pi/extensions/fm-primary-turnend-guard.ts` plus the tracked `.pi/extensions/fm-primary-pi-watch.ts`, both project-local extensions the Pi engine auto-discovers once trusted. When changing any primary watcher adapter, update `docs/supervision-protocols/`, `docs/turnend-guard.md` if a shared idle or turn-end hook changed, and the relevant concise fact below. ## Launch profile axes @@ -117,8 +123,28 @@ The supported launch-profile flags below are verified locally; each row records | claude | `--model ` | `--effort ` | Verified on Claude Code 2.1.196. | | codex | `--model ` | `-c 'model_reasoning_effort=""'` | Verified on codex-cli 0.142.1. The installed binary schema contains `model_reasoning_effort`, the active config uses it, and the bundled model catalog advertises only low/medium/high/xhigh. `max` is omitted. | | grok | `--model ` | `--reasoning-effort ` | Verified on grok 0.2.99 (2026-07-13). `--effort` is an alias, but firstmate's profile axis is reasoning effort. As of 0.2.99 the ceiling is `high`; both `xhigh` and `max` are rejected with `use one of: high, medium, low`, so firstmate omits them. | -| pi | `--model ` | `--thinking ` | Verified 2026-07-13 on Pi 0.80.6. `pi --help` advertises `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, and `max`; `pi --print --model openai-codex/gpt-5.6-sol --thinking max 'Reply with exactly OK.'` completed successfully. | +| pi / pi-signed | `--model ` | `--thinking ` | Verified 2026-07-27 on Pi and pi-signed 0.82.0. Both expose the same accepted thinking levels and completed the same model-qualified max-thinking smoke. | | opencode | `--model ` | none for firstmate's interactive launch | Verified on opencode 1.17.6. `opencode run` has `--variant`, but firstmate launches the interactive `opencode --prompt` path, which has no verified effort flag. | +| kimi | `--model ` | none | Verified 2026-07-25 on Kimi Code CLI 0.29.1. | + +The concrete `harness` field owns adapter identity independently of the model provider: `harness=pi` with `model=xai/grok-*` is Pi using xAI, not `harness=grok`, and does not require Grok CLI login; `harness=grok` remains the standalone Grok Build CLI adapter. + +### Model support discovery + +Treat model and provider knowledge as current source-of-truth discovery, not as a permanent namespace or provider mapping. +Use the discovery surface in the current authenticated environment because supported and available models can change by version, account, and configuration. + +| Harness | Authoritative discovery surface | +|---|---| +| claude | Open the current interactive session's `/model` picker; `claude --help` documents the accepted alias or full-model-name input shape. | +| codex | Open the current interactive session's `/model` picker. | +| opencode | Run `opencode models [provider]`, which lists available provider/model identifiers. | +| pi / pi-signed | Run the selected executable as ` --list-models [search]`; Pi's installed `docs/models.md` owns how built-in, extension-registered, and custom provider/model entries reach that list. | +| grok | Run `grok models`, which lists the models available to the current Grok installation and account. | +| kimi | Run `kimi provider list --json`, which lists the current provider and model configuration. | + +For an unfamiliar harness or model namespace, establish support and provider identity from that harness's authoritative CLI help, model listing, or current documentation rather than guessing from a name or prefix. +If those sources do not establish the relationship needed for dispatch, fail loudly and report the unresolved candidate. When a requested effort value is outside the harness-specific accepted set, `fm-spawn` records the requested `effort=` in meta but emits no effort flag for that harness. This preserves launch success instead of passing a known-bad value. @@ -131,14 +157,21 @@ Natural language is acceptable if uncertain. - claude: `/`, for example `/no-mistakes`. - codex: `$`, for example `$no-mistakes`; `/` is claude-only and codex rejects it as "Unrecognized command". - opencode: no separate verified skill invocation beyond normal slash-command behavior; use natural language if the exact skill command is uncertain. -- pi: no separate verified skill invocation beyond normal command behavior; use natural language if the exact skill command is uncertain. -- grok: `/`, for example `/no-mistakes` (same form as claude). Verified end to end: grok discovers the user-level `no-mistakes` skill, `/no-mistakes` invokes it, and grok drives a real `no-mistakes axi run`. Like codex's `$`/`/` popups, typing `/` opens grok's slash-autocomplete, so a too-fast Enter selects the popup entry instead of sending, and for an argument-taking command (like `/no-mistakes`'s optional task-first argument) that first Enter only expands the popup selection into an argument-hint placeholder rather than submitting - a genuine second Enter is required (see the grok section below for the 2026-07-03 incident and fix). `fm_tmux_submit_core`'s retried Enter (used by `fm-send` on the tmux backend) already handles this correctly by reading the cursor row; the herdr backend needed a dedicated fix (`fm_backend_herdr_composer_state`, docs/herdr-backend.md) because its prior delta-based verification false-positived on that same popup-close content change. +- pi and pi-signed: no separate verified skill invocation beyond normal command behavior; use natural language if the exact skill command is uncertain. +- grok: `/`, for example `/no-mistakes` (same form as claude). Verified end to end: grok discovers the user-level `no-mistakes` skill, `/no-mistakes` invokes it, and grok drives a real `no-mistakes axi run`. Like codex's `$`/`/` popups, typing `/` opens grok's slash-autocomplete, so a too-fast Enter selects the popup entry instead of sending, and for an argument-taking command (like `/no-mistakes`'s optional task-first argument) that first Enter only expands the popup selection into an argument-hint placeholder rather than submitting - a genuine second Enter is required (see the grok section below for the 2026-07-03 incident and fix). `fm_tmux_submit_core`'s retried Enter (used by `fm-send` on the tmux backend) handles this through the structural composer reader; the herdr backend needed a dedicated fix (`fm_backend_herdr_composer_state`, docs/herdr-backend.md) because its prior delta-based verification false-positived on that same popup-close content change. +- kimi: `/`, for example `/no-mistakes`. + +## Submission acknowledgement hazards -## claude (VERIFIED) +A send or key action reporting success is not proof that the intended action happened. +OpenCode can accept and queue an Enter while leaving text visible, Grok can consume Enter in its slash popup without submitting, and Kimi can silently drop a message sent before readiness even though the send returns success. +The shared symptom is a healthy-looking pane with no work in progress, so each adapter must verify the observable postcondition that is specific to its TUI. + +## claude (VERIFIED; busy signature re-verified 2026-07-25 on Claude Code 2.1.220) | Fact | Value | |---|---| -| Busy-pane signature | `esc to interrupt` | +| Busy-pane signature | Current turns match the harness-scoped `…[[:space:]]+\([0-9]+[smh]` shape after a rotating glyph and word, for example `✢ Pollinating… (16s · ...)`; legacy `esc to interrupt` remains accepted, while `Worked for 31s` is idle. | | Exit command | `/exit` | | Interrupt | single Escape | | Skill invocation | `/` (e.g. `/no-mistakes`) | @@ -152,17 +185,17 @@ A plain `tmux capture-pane` cannot tell that ghost text apart from typed text. Firstmate launches every claude crewmate and secondmate with `CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false`, scoped to firstmate-launched agents through `bin/fm-spawn.sh`, so it never touches the captain's global config. The CLI's `--prompt-suggestions` flag is print/SDK-mode only and does not suppress the interactive composer ghost text, verified empirically on v2.1.186. As defense in depth for any pane that flag cannot reach, including the captain's own firstmate composer that away-mode reads, the shared `fm_composer_strip_ghost` extractor in `bin/fm-composer-lib.sh` removes dim/faint SGR 2 ghost runs before pending-input classification on both ANSI-capable readers (tmux and herdr). -Its broader dark-TRUECOLOR placeholder handling and dark-theme tradeoff are documented in `docs/herdr-backend.md`'s 2026-07-10 incident record. +Its broader dark-TRUECOLOR placeholder handling and dark-theme tradeoff are documented in `docs/herdr-backend.md` "Composer and injection safety", with active captures in `docs/verification/runtime-backends.md`. That styled capture is internal to the boolean detector only. `fm-peek` and every other human or LLM-facing capture path stays plain `tmux capture-pane` with no escape codes. -**Primary-session guard fact (verified 2026-07-04, Claude Code 2.1.201; preserved 2026-07-08, Claude Code 2.1.204).** +**Primary-session guard fact (verified 2026-07-04, Claude Code 2.1.201; preserved 2026-07-08, Claude Code 2.1.204; Stop-owned auto-arm revalidated 2026-07-24, Claude Code 2.1.219).** This is separate from the per-task crewmate turn-end hook above (that one just `touch`es a marker file in a task's own `.claude/settings.local.json`). -The firstmate PRIMARY's own `.claude/settings.json` registers `bin/fm-turnend-guard.sh` as a Stop hook, and exiting with status 2 plus stderr reliably forces the model to continue. -Claude Code's stdin payload to a Stop hook carries a `stop_hook_active` boolean that is `true` exactly when the current stop attempt is itself a forced continuation from an earlier block this turn; a hook can and should use that as its own loop-guard (always allow the stop when it is already `true`) rather than tracking state itself. +The firstmate PRIMARY's own `.claude/settings.json` registers two Stop hooks: `bin/fm-turnend-guard.sh --claude` and the Stop-owned auto-arm `bin/fm-claude-stop-autoarm.sh` (`asyncRewake: true`, `timeout: 28800`), and exiting the guard with status 2 plus stderr reliably forces the model to continue. +Claude Code's stdin payload to a Stop hook carries a `stop_hook_active` boolean that is `true` when the current stop attempt follows ANY stop-hook-driven continuation, including `asyncRewake` rewakes; the primary guard therefore ignores it in `--claude` mode and uses the cooperative claim/epoch check plus a bounded re-block budget instead, while the codex-mode default still treats it as a one-block loop guard. A project-level `.claude/settings.json` only takes effect when Claude Code's project root is that exact directory - it does not walk up from a subdirectory looking for one, so firstmate launches the primary from the repo root. -After those settings are loaded, hook command resolution is still cwd-sensitive because Claude Code runs commands through `/bin/sh` against the session's current cwd; keep the tracked command anchored through `"$CLAUDE_PROJECT_DIR"/bin/fm-turnend-guard.sh` and see `docs/turnend-guard.md` for the verified Stop-hook details. -Claude Code's primary watcher protocol is the lowest-friction path: run `bin/fm-watch-arm.sh` as its own Claude Code background task and treat background-task completion as the wake. +After those settings are loaded, hook command resolution is still cwd-sensitive because Claude Code runs commands through `/bin/sh` against the session's current cwd; keep the tracked commands anchored through `"$CLAUDE_PROJECT_DIR"/bin/...` and see `docs/turnend-guard.md` for the verified Stop-hook details. +Claude Code's primary watcher protocol is Stop-owned: the auto-arm hook fires on every Stop and foregrounds `bin/fm-watch-arm.sh` when the home is eligible and still needs supervision, and its exit-2 `asyncRewake` rewake is the wake; the model drains and handles wakes but never runs a routine re-arm command. ## codex (VERIFIED 2026-06-11, codex-cli 0.139.0) @@ -232,7 +265,7 @@ Throwing from `session.idle` does not block `opencode run`, so the primary adapt The companion `.opencode/plugins/fm-primary-watch-arm.js` owns normal TUI watcher wake supervision and coordinates with the guard plugin before the guard tries a blind-turn follow-up. The follow-up was verified in the interactive TUI; `opencode run` can exit before displaying a queued follow-up, so the adapter is fail-open in headless mode. -## pi (VERIFIED 2026-06-11) +## pi and pi-signed (VERIFIED 2026-07-27) | Fact | Value | |---|---| @@ -241,6 +274,11 @@ The follow-up was verified in the interactive TUI; `opencode run` can exit befor | Interrupt | single Escape | Pi has no permission system, so crewmates are always autonomous. +`pi-signed` is the signed wrapper identity verified on version 0.82.0 and exposes the same CLI and TUI behavior as Pi. +Firstmate launches the selected executable name from `PATH`, records `pi-signed` without normalization, and refuses rather than falling back to `pi` when that wrapper is unavailable. +The observed signed process tree is an exact `pi-signed` wrapper parent with the Pi application as its child, while tmux reports the foreground command as the exact `pi-launcher` name for both selected executables. +The installed plain `pi` command also execs that signed launcher, so `FM_PI_HARNESS=pi-signed` is the authoritative selection marker and shared unmarked ancestry remains `pi`. +Firstmate sets `FM_PI_HARNESS` explicitly for both worker launch identities, and a signed primary uses the README launch command to establish the same boundary. Keep the brief as one positional argument. Multiple positional args become separate queued messages; `fm-spawn`'s template already does this correctly. @@ -257,8 +295,8 @@ The firstmate PRIMARY's own `.pi/extensions/fm-primary-turnend-guard.ts` listens Without `deliverAs: "followUp"`, Pi rejects the send while the agent is still processing. Pi's primary watcher protocol also requires the tracked `.pi/extensions/fm-primary-pi-watch.ts` extension, same trust-once discovery as the turn-end guard. The model arms through `fm_watch_arm_pi`, never a foreground bash arm; the watcher tool result and clean-exit fallback are owned by `docs/supervision-protocols/pi.md`. -`bin/fm-session-start.sh` reports when the live Pi session has not loaded both the turn-end guard and watcher extensions, and points at plain `pi` after project trust as the fix, with `-e` as a trust-free fallback. -When a secondmate is launched on Pi, `fm-spawn.sh --secondmate` launches Pi with both `-e .pi/extensions/fm-primary-turnend-guard.ts` and `-e .pi/extensions/fm-primary-pi-watch.ts`, both already present in the secondmate home's git worktree. +`bin/fm-session-start.sh` reports when the live Pi-family session has not loaded both the turn-end guard and watcher extensions, and points at the selected executable after project trust as the fix, with `-e` as a trust-free fallback. +When a secondmate is launched on Pi or pi-signed, `fm-spawn.sh --secondmate` launches the selected executable with both `-e .pi/extensions/fm-primary-turnend-guard.ts` and `-e .pi/extensions/fm-primary-pi-watch.ts`, both already present in the secondmate home's git worktree. ## grok (VERIFIED 2026-06-29, grok 0.2.73; slash-submit re-verified 2026-07-03 on 0.2.82; reasoning-effort ceiling re-verified 2026-07-13 on 0.2.99; exit paths re-verified 2026-07-19 on grok 0.2.103) @@ -278,8 +316,8 @@ For Grok's supported reasoning-effort values and omission behavior, see the [lau **Incident (2026-07-03, herdr backend only, grok 0.2.82):** two grok/herdr crewmates were sent `/no-mistakes` via `fm-send`; both left it fully typed but unsubmitted in the composer for minutes (footer still `Enter:send`), and `fm-send` exited 0 with no error. Reproduced live: the herdr adapter's submit-verification at the time treated ANY pane-content change after Enter as "submitted", and the popup-close-with-placeholder-fill described above IS a visible content change even though nothing was actually sent. -The tmux backend was never affected - `fm_tmux_composer_state` reads the actual cursor row, correctly sees the placeholder text as still-pending, and its retry loop already sends the needed second Enter. -Fixed in the herdr adapter (`fm_backend_herdr_composer_state`, `bin/backends/herdr.sh`) by classifying the composer's own row structurally instead of diffing raw content; see `docs/herdr-backend.md`'s "Incident (2026-07-03)" section for the full account and `tests/fm-backend-herdr.test.sh` for the regression coverage. +The tmux backend's structural `fm_tmux_composer_state` read sees placeholder-filled text on any content row as still pending, so its retry loop sends the needed second Enter. +The Herdr adapter (`fm_backend_herdr_composer_state`, `bin/backends/herdr.sh`) classifies the composer's own row structurally instead of diffing raw content; see `docs/herdr-backend.md` "Composer and injection safety" for the current boundary and `tests/fm-backend-herdr.test.sh` for regression coverage. Startup dialog: the "Run Grok Build in a project directory?" project picker appears ONLY when grok is launched from a non-project directory (home, Desktop, Downloads, `/tmp`). `fm-spawn` launches inside the treehouse worktree (a git repo root), so the picker never appears and grok treats the worktree as a trusted project automatically - no post-launch keystroke is needed. @@ -292,11 +330,10 @@ Verified live against grok 0.2.93: real input is the bright `38;2;224;222;244` ( This assumes a dark terminal theme, the fleet reality; the SGR-2 signal stays theme-independent. Regression coverage: `tests/fm-composer-ghost.test.sh` (`test_strip_ghost_drops_dark_truecolor_ghost`, `test_dark_truecolor_ghost_only_composer_is_not_pending`) and `tests/fm-backend-herdr.test.sh` (`test_composer_state_grok_dark_truecolor_placeholder_is_empty`, `test_composer_state_grok_bright_truecolor_real_text_is_pending`). -**Residual gap, tmux-only (unfixed):** -in that same pristine placeholder-only state, tmux's own `#{cursor_y}` points at the composer box's BOTTOM BORDER row, one row below the actual text row (the box appears to render one row lower before any real typing starts); once real text is typed the cursor correctly aligns with the text row again. -This is a row-SELECTION quirk, orthogonal to the styling fix above, and affects only the tmux path (herdr uses a structural composer-row scan, not `cursor_y`, so it is unaffected). -A correct fix needs a row-window read near `cursor_y` rather than the single `cursor_y` row. -In practice `fm-spawn` launches grok with the brief as its initial prompt, so a live task's composer is never observed in this pristine pre-typing state - but this is unverified for every path (e.g. a steer sent before grok's first real turn settles) and needs dedicated investigation before relying on it. +**Tmux bottom-border cursor quirk (fixed):** +In a pristine placeholder-only composer, tmux's `#{cursor_y}` can point at the box's bottom border instead of its text row. +The shared tmux reader now locates the complete box structurally and classifies every content row, so the cursor may sit on a content row or the bottom border without changing the result. +The same structural read covers multi-row composers without fixed cursor offsets, while Herdr retains its own structural composer-row scan. Turn-end hook: grok fires a `Stop` hook at every turn boundary, giving firstmate a precise per-turn wake instead of only stale-pane detection. grok loads PROJECT hooks (`/.grok/hooks/`, `/.claude/settings.local.json`) only after the folder is granted hook-trust in `~/.grok/trusted_folders.toml`, which is not automatic and which firstmate will not establish by editing grok's own managed trust store. @@ -309,10 +346,51 @@ This keeps the hook outside the worktree, needs no trust grant, and writes only `fm-teardown` removes the worktree pointer before returning a pooled worktree. Secondmate spawns skip the pointer (idle panes are healthy, no stale-pane detection for them). -**Primary-session guard fact (verified 2026-07-08, Grok 0.2.91).** +**Primary-session guard fact (verified 2026-07-28, Grok 0.2.112 and 0.2.73).** The firstmate PRIMARY's own `.grok/hooks/fm-primary-turnend-guard.json` invokes `bin/fm-turnend-guard-grok.sh`. -Grok Stop hooks are passive for this purpose: exit 2 does not make the model continue. -The adapter therefore runs the shared predicate and, when it returns 2, forces one same-session follow-up with `grok --resume -p ` while setting `GROK_TURNEND_GUARD_ACTIVE=1` so the nested Stop hook does not recurse. -It does not pass `--permission-mode`, so the passive hook cannot escalate the primary session's tool permissions. +Grok 0.2.112 exposes native same-process Stop continuation in its running payload, while the genuine pre-native 0.2.73 payload omits that capability and still needs one guarded `grok --resume`. +The exact adaptive and malformed-input contract is owned by `docs/turnend-guard.md`. +The tracked Claude Stop hooks skip themselves under `GROK_AGENT`, because Grok also loads Claude-compatible project settings and otherwise creates a second blocking path. Project-local Grok hooks require folder trust, verified with launch-time `--trust`; if the primary firstmate checkout is not trusted for Grok hooks, this primary guard fails open and `fm-guard.sh` remains the next-command alarm. -Grok's primary watcher protocol is Claude-shaped background-notify around `bin/fm-watch-arm.sh`; the passive Stop hook is only a backstop for blind turn ends. +Grok's primary watcher protocol remains background-notify around `bin/fm-watch-arm.sh`; native Stop continuation does not provide Pi-like extension ownership. + +## kimi (VERIFIED 2026-07-25, kimi 0.29.1) + +Kimi Code CLI launches from the absolute path resolved from `PATH`, falling back to the executable `$HOME/.kimi-code/bin/kimi`. + +| Fact | Value | +|---|---| +| Binary | Executable `kimi` from `PATH`, then executable `$HOME/.kimi-code/bin/kimi`; spawning refuses if neither exists. | +| Launch | Bare interactive TUI with `--auto`, followed by readiness-gated pointer delivery; positional prompts are rejected. | +| Models | `kimi-code/kimi-for-coding` (default), `kimi-code/kimi-for-coding-highspeed`, `kimi-code/k3`, and `kimi-code/k3-256k`. | +| Busy-pane signature | A transient line with optional leading whitespace, a rotating moon-phase glyph, required whitespace on both sides of `·`, and optional trailing content; the line is absent when idle. | +| Exit command | `/exit` | +| Interrupt | Single Escape, which prints `Interrupted by user`. | +| Skill invocation | `/`, for example `/no-mistakes`; firstmate skills are discovered. | +| Autonomy | `--auto`; `-y` and `--yolo` are weaker and are not used. | +| Trust dialog | None on a clean first launch in a fresh pooled worktree. | +| Slash submission | One Enter submits, with no popup swallow or settle hazard. | +| Environment marker | None; detection relies on process ancestry command name `kimi`. | +| Composer | Bordered box with a bare `>` prompt glyph and no observed ghost or placeholder text. | +| Effort | No reasoning-effort flag exists, so requested effort is recorded in task metadata but omitted from launch. | + +`fm-spawn.sh` launches Kimi bare, waits for the composer box or `Welcome to Kimi Code!`, sends only `Read the brief at and follow it exactly.`, and requires a cleared composer plus either the echoed `✨` submission or nonzero context before accepting delivery. +This launch-then-send shape is mandatory because Kimi rejects a positional brief as an unknown command. +Sending before readiness was reproduced as a silent drop with a zero exit status, an empty composer, `context: 0%`, no echoed user message, and a healthy-looking idle pane. +The brief path must be absolute because the brief lives outside the task worktree, and Kimi reads it there without `--add-dir`. + +Observed live spinner captures included optional leading whitespace, a moon-phase glyph, whitespace around `·`, and rotating tip text, with the same shape observed during tool execution. +Because every captured spinner row had whitespace on both sides of `·`, the matcher requires that whitespace, deliberately does not match the never-observed zero-whitespace form, and does not require trailing tip text. +The startup input-readiness window is the established cause of Kimi's first-Enter delivery defect, while the banner is not the cause. +An early Enter can expand Kimi's composer to multiple content rows, leaving the pointer text on the first row and the cursor on an empty later row, which is the same single-cursor-row reading defect exposed by Grok's bottom-border cursor quirk. +The shared tmux reader now locates the complete bordered composer and treats real text on any content row as positive evidence that submission is still pending. +No rendering signal is trustworthy for proving that Kimi will accept input during this window, so delivery retries Enter through the shared submit core and retains the existing postcondition verification rather than relaxing readiness or delivery checks. +Kimi's footer tip rotates independently and can display `ctrl+c: cancel` while completely idle, so tip text is never used as its busy signature without the leading moon-plus-middot spinner structure. +The idle status bar can contain lowercase `thinking`, which is the model's effort label rather than a busy signal. +The spinner match covers the full moon-phase glyph set rather than one frame, but it remains locale- and emoji-font-sensitive because Kimi exposes no stable ASCII busy token. + +[`docs/turnend-guard.md`](../../../docs/turnend-guard.md) owns Kimi's verified global hook surface and captain-approved crew wake integration. +`fm-spawn.sh` installs one marker-delimited Firstmate entry in `$HOME/.kimi-code/config.toml`, one silent always-zero hook script, and one private token registry under `$HOME/.kimi-code/fm-turn-end.d/`. +Each Kimi crew worktree receives a gitignored `.fm-kimi-turnend` token pointer, and the global hook touches that task's `state/.turn-ended` only when the Stop payload's `cwd`, pointer, and registry entry all agree. +A guarded silent hook cannot be verified from absence of effect, so prove invocation with an unguarded probe before concluding that the hook did not fire. +The guarded turn-end signal supplements the pane busy signature, whose locale- and emoji-font-sensitive limits still apply while a turn is running. diff --git a/.agents/skills/mobile-mode/SKILL.md b/.agents/skills/mobile-mode/SKILL.md new file mode 100644 index 00000000000..dd316e46926 --- /dev/null +++ b/.agents/skills/mobile-mode/SKILL.md @@ -0,0 +1,54 @@ +--- +name: mobile-mode +description: >- + Shape captain-facing Firstmate messages and review handoffs for a phone, especially when Moshi is the active surface. +user-invocable: false +metadata: + internal: true +--- + +# mobile-mode + +Load this skill when the captain says they are in mobile mode or identifies Moshi as the active surface. +Continue following it until the captain says they are back on desktop or requests normal mode. + +This is a presentation profile over the same host-side Firstmate session. +It does not create a Moshi runtime backend, supervision path, authority channel, or webhook integration. +The always-loaded [`i-have-adhd`](../i-have-adhd/SKILL.md) contract remains the owner of general captain-facing presentation, and `AGENTS.md` section 9 remains the owner of outcome translation and approval escalation. +This skill owns only the mobile delta and the review-surface choice. + +## Message shape + +- Apply the `i-have-adhd` outcome-first rule, then keep the outcome, consequence, evidence, and requested action within one ordinary phone scroll whenever the required facts fit. +- Make choices answerable with one low-typing reply such as `1`, `2`, `yes`, `merge`, or `hold`. +- Number choices, put the recommendation first, and end with the exact short reply that will select it. +- Keep full `https://...` pull-request links under section 9's existing rule so the captain can open the review directly. +- Put long logs and secondary evidence in the existing private report, then summarize the consequence in chat. +- Do not require terminal copy mode, pane navigation, punctuation-heavy commands, or multi-step text entry to answer a decision. + +## Review-surface choice + +Use plain chat for a simple decision or whenever a rich surface is unnecessary or unavailable. +Use a host-local Lavish surface through Moshi Pro Browser Preview when several options or structured feedback benefit from a touch-friendly review. +Follow the operator runbook in [`docs/moshi-mobile-review.md`](../../../docs/moshi-mobile-review.md). + +After Lavish starts on the host, tell the captain to open Browser Preview in Moshi and choose the Lavish server. +Never present a raw `127.0.0.1`, `localhost`, `file://`, or desktop-only LAN URL as if the phone can open it directly. +If Browser Preview is unavailable, restate the complete decision in chat with numbered replies instead of suggesting public sharing. + +Use Moshi Pro Diff to inspect the connected working tree, while keeping a full HTTPS pull-request link in chat for a hosted PR review. +Use Chat View only when Moshi recognizes the active agent and session. For unsupported prompts, incomplete cards, or an unrecognized session, keep the same Moshi/Firstmate session and present the complete fallback as concise numbered Firstmate chat; the terminal remains the source of truth, but is not a separate mobile handoff surface. + +Private fleet reviews stay host-local. +Never invoke or suggest `lavish-axi share` for private fleet state, even with a password. +Public sharing of separately sanitized public material requires an explicit request and the ordinary outward-facing consent boundary. + +## `moshi-hook` authority + +The already-installed `moshi-hook` may surface the running agent's native inbox events and approvals in Moshi or Apple Watch. +An approval button may answer only the exact native agent prompt that Moshi can map safely to the same live session. +It never authorizes a Firstmate merge, product or scope decision, destructive or irreversible action, credential use, permission change, or security-sensitive choice. +Route those decisions through Firstmate chat under the existing authority contract. + +Do not add a Firstmate-to-Moshi webhook, change hook configuration, or place secrets in notification summaries. +When an approval is unavailable, ambiguous, or unsupported, return to the terminal or ask through Firstmate chat without weakening the underlying boundary. diff --git a/.agents/skills/project-management/SKILL.md b/.agents/skills/project-management/SKILL.md index f2cc0b11a17..b6211f56b95 100644 --- a/.agents/skills/project-management/SKILL.md +++ b/.agents/skills/project-management/SKILL.md @@ -3,6 +3,7 @@ name: project-management description: >- Agent-only procedure for Firstmate project management. Use before adding, creating, removing, or initializing a project. + Cloning or registering a project is add intake and uses the same trigger. Owns project add, create, clone, remove, initialization, registry, delivery-mode, autonomy, outward-consent decisions, and the secrets-intake handoff. user-invocable: false metadata: @@ -12,6 +13,7 @@ metadata: # project-management Use this procedure before adding, creating, removing, or initializing a project. +Cloning or registering a project is add intake and uses the same trigger. This skill is the single owner of Firstmate's project-management procedure. It does not replace `secondmate-provisioning`, which owns project clones inside persistent secondmate homes. @@ -22,6 +24,11 @@ Use the registry format and parser contract owned by the header of `bin/fm-proje Keep each registry description useful for identifying the project, but keep delivery posture, captain-private state, and detailed project knowledge in their existing designated homes. Do not turn the registry into project documentation. +Before adding, cloning, creating, or registering any project in the main home, inspect the authoritative `data/secondmates.md` routing table and judge every existing natural-language `scope:` against the proposed project or domain. +Apply `AGENTS.md` section 7's authoritative secondmate routing rules; if an existing scope owns that domain, route the new-project operation or work there instead of creating or registering a duplicate main-home clone. +Absence from the main `data/projects.md` registry is never evidence that no second mate owns the domain. +If the owning second mate cannot accept the route, report that concrete blocker or obtain an explicit captain redirection rather than silently duplicating the project in the main home. + Resolve the project name, destination, delivery mode, and autonomy posture before changing local or remote state. Keep a newly added clone and its registry entry consistent, and roll back only artifacts created by the incomplete operation when a later initialization step fails and that rollback is safe. Do not overwrite or repurpose an existing path. @@ -76,9 +83,9 @@ Firstmate never hand-writes that project file. ## Remove -Project removal is destructive and is not one of Firstmate's current direct-write exceptions under `projects/`. -Never issue a raw removal command from Firstmate. +Project removal is destructive. First obtain the captain's explicit removal decision, then inspect the current digest and authoritative repositories for in-flight or queued work, registered secondmate clones, linked worktrees, dirty files, unpushed commits, and any other unlanded work. -If any dependency or unlanded work exists, stop and report it before changing the registry. -Until a guarded removal helper and corresponding prime-directive exception exist, report that implementation gap instead of bypassing the project-write boundary. -When a clone has already been removed through an approved guarded path, or the registry is provably stale because no clone exists, remove its registry line so navigation matches reality. +If any dependency or unlanded work exists, stop and report it before changing anything. +Never issue a raw removal command from Firstmate. +Once that preflight confirms none of the above and the captain's approval is concrete, AGENTS.md hard rule 1's captain-approved project operation exception authorizes firstmate to remove the clone directly and update its registry entry to match. +When a clone has already been removed through an approved removal, or the registry is provably stale because no clone exists, remove its registry line so navigation matches reality. diff --git a/.agents/skills/quota-array-dispatch/SKILL.md b/.agents/skills/quota-array-dispatch/SKILL.md new file mode 100644 index 00000000000..c384553a859 --- /dev/null +++ b/.agents/skills/quota-array-dispatch/SKILL.md @@ -0,0 +1,65 @@ +--- +name: quota-array-dispatch +description: >- + Agent-only decision procedure for resolving a matched crew-dispatch profile + array from current quota-axi output, including quota-window pace signals. + Load when a dispatch rule or default resolves to more than one profile candidate. +user-invocable: false +metadata: + internal: true +--- + +# quota-array-dispatch + +This skill is the single owner of the pace-aware profile-array selection procedure. +`AGENTS.md` section 4 owns the always-loaded intake boundary, load trigger, malformed-config refusal, every-candidate accounting, and strongest-reasoning/tie safety rules. +`harness-adapters` owns harness verification, model/provider discovery, and effort fallback. +`quota-axi` remains data-only and never recommends a route. +Do not add a daemon, opaque composite score, routing wrapper, hard-coded model-specific policy, or producer-side route recommendation. + +## Collect facts + +Run `quota-axi --json` once per intake and reuse that snapshot for every candidate. +For each candidate, preserve explicit `harness`, `model`, and `provider`; `harness-adapters` owns identity, and model/provider never infer harness: + +- task/profile fit and required reasoning class +- raw applicable headroom (`effectivePercentRemaining` or tightest applicable percentage) +- effective pace, signed reserve per window, and worst reserve (`worstReservePercentPoints` or minimum signed reserve) +- whether applicable windows/summary are ahead, or pace is `unknown` +- schema note when pace fields are absent + +Stale raw windows are diagnostic, never headroom. +Read all windows named by `boundedBy`, `limitingWindowIds`, `aheadWindowIds`, `behindWindowIds`, `onPaceWindowIds`, and `unknownWindowIds`. + +## Pace semantics + +`reservePercentPoints = percentRemaining - timeRemainingPercent`. +Negative reserve means usage is ahead of reset pace and creates conservation pressure. +Positive reserve means usage is behind reset pace. +`on_pace` is neutral. +Conservation pressure is present for effective pace status `ahead`, effective pace status is `mixed` and any `aheadWindowIds` remain, or a bounding window is `ahead`. +`unknown` is valid explicit uncertainty from quota-axi, not parser failure or permission to assume health. + +## Selection order + +Apply only among candidates satisfying required fit and strongest reasoning class. +Never use pace or raw headroom to silently replace that reasoning class. + +1. Unresolved relationship or quota: stop and report the tuple and concrete evidence. +2. All-tight: keep strongest reasoning; dispatch inside it or report if blocked. +3. Comparable fit/reasoning: prefer no ahead pressure over pressure, even with higher raw headroom. +4. Among pressured candidates, prefer the least-negative worst applicable reserve. +5. Sustainable candidates: use known pace plus raw headroom. + Prefer known sustainable evidence over `unknown` when comparable. + Do not collapse those facts into an opaque composite score. +6. If unresolved pace changes the choice, report uncertainty. +7. Absent pace or older schema: do not crash, fabricate pace, or treat absence as healthy/`on_pace`. + Compare raw headroom only, state pace is unavailable, and keep safety rules. +8. Genuine ties: stop and report every tied candidate for captain choice. + Do not select by array order, harness name, or another arbitrary identity ordering. + Report duplicate concrete profiles as a configuration error. + +Name the inspectable facts used for every candidate. +After selecting, check auth only through that tuple's surface; another harness CLI cannot block it. +A blocked credential report must name `harness`, `model`, authentication surface, and concrete failure evidence; never emit a bare `Grok unauthenticated` statement. +Never conclude with an unexplained "best quota" label. diff --git a/.agents/skills/secondmate-provisioning/SKILL.md b/.agents/skills/secondmate-provisioning/SKILL.md index ecc364939ac..44caa0cb1c1 100644 --- a/.agents/skills/secondmate-provisioning/SKILL.md +++ b/.agents/skills/secondmate-provisioning/SKILL.md @@ -78,10 +78,13 @@ This section is the single owner of the secondmate sync and inherited-local-mate Before launch, `fm-spawn.sh --secondmate` locally fast-forwards the home to the primary firstmate checkout's current default-branch commit when it is safe; dirty, diverged, or in-flight homes launch unchanged with a warning. The locked session-start bootstrap sweep runs the same guarded fast-forward for every live secondmate home, discovered from `state/.meta` records with `kind=secondmate` (`data/secondmates.md` only backfills `home=` for older records). That no-fetch path is a purely local fast-forward of tracked files, never an origin fetch, and it never touches the gitignored operational dirs, so a secondmate's backlog, projects, and in-flight work are never disturbed; a linked worktree advances immediately, while a standalone clone that lacks the target receives firstmate updates through `/updatefirstmate`'s origin refresh. -The same launch and the same locked bootstrap sweep also propagate the primary's declared inherited local material: `config/crew-dispatch.json`, `config/crew-harness`, `config/backlog-backend`, `config/herdr-presentation-spaces`, and the one shared captain-preference file `data/captain-shared.md`. +The same launch and the same locked bootstrap sweep also propagate the primary's declared inherited local material: `config/crew-dispatch.json`, `config/crew-harness`, `config/backlog-backend`, `config/backend`, `config/herdr-presentation-spaces`, `config/startup-memory-budget`, and the one shared captain-preference file `data/captain-shared.md`. Because these paths are gitignored, that propagation is a separate, primary-authoritative copy independent of the tracked-files fast-forward: it re-converges every live home whether or not its tracked files advanced, and it touches only the declared items. Propagation failures warn without blocking secondmate launch or session-start continuation, and the destination keeps whatever safely validated state the helper left behind. Inheritance copies the literal `config/crew-harness` file, so a secondmate's own crewmates use the primary's crewmate harness only when it names a concrete adapter such as `codex`; an unset or `default` value has nothing concrete to inherit, and the secondmate's own crewmates fall back to the secondmate's own or detected harness instead. +Inherited `config/backend` becomes that secondmate home's local runtime-backend default for future spawns only; it never retargets, rewrites, migrates, stops, or restarts an already-live worker endpoint. +A present primary value always converges byte-exact into validated secondmate homes, and primary absence removes the destination so those homes keep runtime auto-detection. +Explicit per-spawn `--backend` and `FM_BACKEND` remain stronger than every home's local `config/backend`, including an inherited default. `config/secondmate-harness` is not inherited because it is only the primary's knob for launching secondmate agents. `data/captain-shared.md` is main-authoritative in the primary home and read-only in secondmate homes. Its primary file header must state that the file is main-authoritative, read-only in secondmate homes, must not be edited there, and that new captain-preference discoveries are routed to the main firstmate through marked status or a document pointer. @@ -96,7 +99,7 @@ Keep every `data/learnings.md` fully local by captain decision; route fleet-gene No AGENTS.md reread nudge is needed at spawn or respawn because the agent reads instructions fresh on launch; only the bootstrap sweep's running-home instruction-surface advance needs that AGENTS.md re-read. Bootstrap reports successful AGENTS.md re-read sends as `BOOTSTRAP_INFO:` and only emits `NUDGE_SECONDMATES:` when that send fails and needs retry. A separate, literal-content config reread is required whenever inherited `config/*` material changes under an already-running secondmate. -After each successful allowlisted config write, both the locked bootstrap convergence path and mid-session `bin/fm-config-push.sh` use the shared propagation report to build one per-home generation-specific private instruction file from the validated destination post-write bytes for only the allowlisted config items that actually changed for that home (`config/crew-dispatch.json`, `config/crew-harness`, `config/backlog-backend`, `config/herdr-presentation-spaces`), in deterministic allowlist order. +After each successful allowlisted config write, both the locked bootstrap convergence path and mid-session `bin/fm-config-push.sh` use the shared propagation report to build one per-home generation-specific private instruction file from the validated destination post-write bytes for only the allowlisted config items that actually changed for that home (`config/crew-dispatch.json`, `config/crew-harness`, `config/backlog-backend`, `config/backend`, `config/herdr-presentation-spaces`, `config/startup-memory-budget`), in deterministic allowlist order. Each changed path is printed with clear begin/end delimiters and the destination file's full exact new bytes unparsed, or the explicit token `ABSENT` when propagation removed the destination copy. The instruction uses only minimal framing that these are defaults/rules and do not remove judgment; it never includes SHA values, selected profiles, parsed summaries, or any other generated interpretation. `data/captain-shared.md` is not a config file and is never inlined into this instruction file or message. diff --git a/.agents/skills/stow/SKILL.md b/.agents/skills/stow/SKILL.md index 4c2c2a337ae..672894bd56d 100644 --- a/.agents/skills/stow/SKILL.md +++ b/.agents/skills/stow/SKILL.md @@ -10,56 +10,76 @@ metadata: # stow -Sweep this session for durable knowledge that only exists in conversation right now, and write it to the disk locations firstmate already prints in the next session-start context digest. -The goal is a session that is safe to reset or destroy because everything durable has already been captured. +Sweep this session for durable knowledge that exists only in conversation, then leave the next session with a compact current operating map rather than an accumulating journal. +This skill writes only through the existing Firstmate ownership and write boundaries. -## What it does +## Required startup-memory pass -1. **Sweep the session for uncaptured durable knowledge.** - Read back over this conversation and look for: - - Operational learnings: fleet-local facts and gotchas discovered while operating firstmate (a script's sharp edge, a harness quirk, a recurring false alarm and its real cause). - - Captain preferences expressed in passing: a working-style or approval preference the captain stated conversationally rather than through the destination selected by AGENTS.md's knowledge-routing table. - - Project-intrinsic facts discovered: build, test, release, or architecture facts about a project that belong in that project's own `AGENTS.md`. - - Decisions made: a standing choice the captain made this session that should outlive it. - - Undone next steps: anything left open that has not yet been filed as backlog work. +Every `/stow` invocation performs this complete pass, even when the session contains no new finding: -2. **Route each finding using AGENTS.md's knowledge-routing table.** - AGENTS.md (section 6, "Knowledge routing") is the single source of truth for where each kind of knowledge belongs. - Read that table and route each finding there instead of re-deriving the mapping here. +1. Run `bin/fm-startup-memory-budget.sh report` before considering a write. + Record its effective budget and each file's estimated-token total. + The helper's stable estimate is the documented conservative local approximation, not provider-exact accounting. + If it rejects the setting or a memory file, do not infer a default or silently continue. + Report that concrete exception and do not call the session reset-safe. +2. Read every current memory file completely: `data/captain.md`, `data/captain-shared.md`, and `data/learnings.md`. + Treat an absent local file as absent, not as an invitation to manufacture content. + In a primary home, all three are curation inputs under their existing ownership rules. + In a secondmate home, `data/captain-shared.md` is a read-only primary-owned input: count it, never edit it, and curate only the editable local files. +3. Build one whole-file retention plan before editing. + Retain, in order: current captain preferences, authority and safety boundaries, and recurring working style; stable home-local operating facts that repeatedly affect future work and are expensive to rediscover; then concise pointers to an existing authoritative report, project document, configuration, or backlog item. + Retain lower-priority material only while budget remains. +4. Consolidate every editable memory file as needed, not only the file apparently related to a new finding. + Prefer one concise current rule or authoritative pointer over duplicate prose. + Remove, merge, or route completed incident and release chronology, stale versions and paths, transient task state, resolved alternatives, old metrics, superseded claims, duplicates, and report-sized procedures. + Do not remove a unique current fact unless it is preserved directly elsewhere through a stronger existing owner. +5. Run `bin/fm-startup-memory-budget.sh report` again after the complete pass. + Finish at or below the effective budget unless a concrete inability remains. + A secondmate must explicitly report `primary-owned-shared-file-alone-exceeds-budget` when the inherited shared file alone exceeds its allowance, because local curation cannot resolve it. + Any other unresolved excess must identify the fact that cannot safely be removed or routed and why. + +A net increase is allowed only for a genuinely new current fact with no stronger owner. +Before allowing it, consolidate enough lower-priority material to remain within budget. +Never describe the session as reset-safe while the memory total is over budget or an exception is unresolved. + +## Knowledge sweep and routing -3. **Write within firstmate's existing write boundaries.** - This skill does not grant any new write permission; it only prompts firstmate to use the boundaries that already exist (AGENTS.md section 1): - - Captain preferences and fleet-local operational facts: hand-write directly to the destination selected by AGENTS.md's knowledge-routing table, using inspect-then-update every time. - Before writing, inspect the destination, find the existing bullet or section the finding duplicates or supersedes, and rewrite it in place rather than adding a new trailing entry. - `data/learnings.md` may not exist yet; create it on first local learning, in the same dated, evidence-backed, curated style as the captain-preference files. - - Project-intrinsic knowledge: never hand-write a project's `AGENTS.md`. - Route it through a normal ship task so a crewmate records it via `bin/fm-ensure-agents-md.sh` and commits it through that project's delivery pipeline, exactly as section 6 describes. - If the fleet is live, delegate this to a crewmate rather than doing it inline. - - Knowledge generalizable to every firstmate user: this repo's own `AGENTS.md` (or other shared, tracked material), shipped through the normal branch -> no-mistakes -> PR -> captain-merge pipeline for this repo (section 1), never hand-committed straight to `main`. - - Task-scoped notes: inspect the relevant backlog item with `tasks-axi show --full`, judge whether the new note is new, duplicate, superseding, or obsolete, then write a considered replacement body with `tasks-axi update --body-file `. - When the replacement intentionally supersedes prior state that should remain recoverable, add `--archive-body` to that update command so the prior body stays recoverable without copying it into the replacement. +1. **Sweep the session for uncaptured durable knowledge.** + Look for operational learnings, captain preferences expressed in passing, project-intrinsic facts, standing decisions, and undone next steps. +2. **Route each finding using AGENTS.md's knowledge-routing table.** + AGENTS.md section 6 is the source of truth for destinations. + Do not re-derive or duplicate that mapping here. +3. **Write within the existing boundaries.** + - Captain preferences and fleet-local operational facts belong in the destination selected by AGENTS.md after the required whole-file curation pass. + Create `data/learnings.md` only for a genuinely new local learning with no stronger owner. + - In a primary home, curate shared captain preferences only under the existing primary-authoritative shared-preference contract. + In a secondmate home, route a newly discovered shared preference to the main firstmate through marked status or a document pointer instead of editing the inherited file. + - Project-intrinsic knowledge never goes directly into a project's `AGENTS.md`. + Route it through a normal ship task so a crewmate records it with `bin/fm-ensure-agents-md.sh` and the project's delivery path. + - Knowledge general to every Firstmate user belongs in this repo's shared tracked material through the normal branch, no-mistakes, PR, and captain-merge path. + - For task-scoped notes, inspect the item with `tasks-axi show --full`, classify the change as new, duplicate, superseding, or obsolete, then use a considered replacement body through `tasks-axi update --body-file `. + Use `--archive-body` when recoverability matters. Never append. - If hand-editing `data/backlog.md` per the active backend, make the same inspect-then-update edit in place. - - Undone next steps: file each as a queued backlog item (section 10), with `blocked-by` recorded if it genuinely depends on something else. + - File each undone next step as a queued backlog item with a genuine `blocked-by` dependency when applicable. +4. **Use inspect-then-update.** + For every retained fact, ask which current statement it supersedes, whether it can be a one-sentence rewrite, and whether a stale entry should be deleted, retired, or routed to an existing stronger owner. + The only graduation moves are promotion to tracked shared material through a PR, folding a learning into the captain-preference destination selected by AGENTS.md, or deletion of a stale entry. + Do not invent another graduation path. + +## Completion receipt + +Report the outcome in plain captain-facing language with all of these facts: -4. **Curate with inspect-then-update.** - Every write starts by reading the current destination and deciding how the finding changes what is already there. - Use this checklist before writing: - - Which existing bullet, section, or task body does this supersede? - - Can this be a one-sentence rewrite instead of a new entry? - - Should an older bullet or note be deleted, retired, or archived because it is now obsolete? - When a finding overlaps or supersedes something already on disk, rewrite or prune the existing entry instead of piling on a new one. - Graduation moves are limited to exactly three: promote a learning to the shared `AGENTS.md` via PR, fold it into the captain-preference destination selected by AGENTS.md, or delete a stale entry. - Do not invent other graduation paths. +- effective startup-memory budget and total estimated tokens before and after; +- one or more actions for each of `data/captain.md`, `data/captain-shared.md`, and `data/learnings.md`: `unchanged`, `added`, `rewritten`, `pruned`, or `routed`; +- each durable finding filed outside memory and its authoritative owner; +- every unresolved exception, including a primary-owned shared-file constraint in a secondmate home; +- whether the session is safe to reset, only when all durable findings are captured and the post-pass result is within budget with no exception. -5. **Report to the captain.** - Summarize, in plain outcome language (section 9): what was stowed and where, what was filed to the backlog, and whether the session is now safe to reset or destroy - i.e. whether every durable finding from this sweep now lives on disk rather than only in this conversation. - If something could not be captured yet (for example, project-intrinsic knowledge waiting on a crewmate to land it), say so explicitly rather than reporting the session fully safe. +Do not hide an over-budget result behind a reset-safe claim. ## Scope exclusion: no skill storage -`/stow` must **never** store, create, or edit a skill as a destination for any finding. +`/stow` must never store, create, or edit a skill as a destination for any finding. There is no "graduate this to a skill" move in this skill's routing. -This is a deliberate, standing exclusion, not an oversight: even with the two-tier skill layout, a stow sweep is a memory-routing operation, not a way to author or mutate skills. -Writing learnings into either `.agents/skills/` or public `skills/` would still risk mixing fleet-local material with shared firstmate behavior or standalone installer-facing behavior. -Until a human deliberately scopes a skill change as firstmate repo work, route generalizable knowledge to the shared `AGENTS.md` (or other shared, tracked material) via the pipeline, and fleet-local knowledge to `data/`, never to a skill. +Until a human deliberately scopes a skill change as Firstmate repository work, route generalizable knowledge to shared tracked material through its pipeline and fleet-local knowledge to `data/`, never to `.agents/skills/` or public `skills/`. diff --git a/.claude/settings.json b/.claude/settings.json index 4ad0e1acaab..0be379c46b7 100644 --- a/.claude/settings.json +++ b/.claude/settings.json @@ -22,10 +22,6 @@ { "type": "command", "command": "\"$CLAUDE_PROJECT_DIR\"/bin/fm-cd-pretool-check.sh --claude" - }, - { - "type": "command", - "command": "\"$CLAUDE_PROJECT_DIR\"/bin/fm-continuity-pretool-check.sh" } ] }, @@ -44,7 +40,13 @@ "hooks": [ { "type": "command", - "command": "\"$CLAUDE_PROJECT_DIR\"/bin/fm-turnend-guard.sh" + "command": "[ -z \"${GROK_AGENT:-}\" ] || exit 0; exec \"$CLAUDE_PROJECT_DIR\"/bin/fm-turnend-guard.sh --claude" + }, + { + "type": "command", + "command": "[ -z \"${GROK_AGENT:-}\" ] || exit 0; exec \"$CLAUDE_PROJECT_DIR\"/bin/fm-claude-stop-autoarm.sh", + "asyncRewake": true, + "timeout": 28800 } ] } diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 8ae5f669bcd..f85fcc28c03 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -324,7 +324,14 @@ jobs: esac /bin/bash --version | head -1 command -v jq >/dev/null || { echo "::error::jq is required"; exit 1; } - /bin/bash -n bin/fm-fleet-snapshot.sh + + shell_inventory="$RUNNER_TEMP/fm-shell-inventory" + bin/fm-lint.sh --list-files > "$shell_inventory" + parse_fail=0 + while IFS= read -r f; do + /bin/bash -n "$f" || { echo "::error::stock macOS Bash 3.2 failed to parse $f"; parse_fail=1; } + done < "$shell_inventory" + [ "$parse_fail" -eq 0 ] || { echo "::error::stock macOS Bash 3.2 parse sweep failed"; exit 1; } snapshot_output=$(/bin/bash tests/fm-fleet-snapshot-view.test.sh) printf '%s\n' "$snapshot_output" @@ -337,8 +344,8 @@ jobs: bearings_output=$(/bin/bash tests/fm-bearings-snapshot.test.sh) printf '%s\n' "$bearings_output" bearings_count=$(printf '%s\n' "$bearings_output" | grep -c '^ok - ') - [ "$bearings_count" -eq 42 ] || { - echo "::error::expected 42 Bearings tests, got $bearings_count" + [ "$bearings_count" -eq 41 ] || { + echo "::error::expected 41 Bearings tests, got $bearings_count" exit 1 } diff --git a/.gitignore b/.gitignore index 372af4735f3..1e5e8642efd 100644 --- a/.gitignore +++ b/.gitignore @@ -8,13 +8,4 @@ data/ __pycache__/ *.pyc .env -config/crew-harness -config/crew-dispatch.json -config/secondmate-harness -config/backlog-backend -config/backend -config/calm -config/x-mode.env -config/cmux-socket-password -config/wedge-alarm -config/herdr-presentation-spaces +config/ diff --git a/.no-mistakes.yaml b/.no-mistakes.yaml index 7cb021b0f3e..02e6128f2e9 100644 --- a/.no-mistakes.yaml +++ b/.no-mistakes.yaml @@ -9,6 +9,19 @@ # HEAD-continuity guard; see docs/architecture.md "No-mistakes gate authority boundary." disable_project_settings: true +# Trusted documentation placement policy for the Document step. +# The audience inventory and coding guideline own the detail; keep this as a +# pointer so gate instructions cannot become a second prose policy. +document: + instructions: | + Read docs/documentation-audiences.md and its machine-consumed + docs/documentation-audiences.json inventory before changing documentation. + Apply the knowledge-placement policy in + .agents/skills/firstmate-coding-guidelines/SKILL.md. + For changed prose, verify audience, authoritative owner, current relevance, + evidence destination, and unique safety facts, then review the complete + branch diff again after every documentation or lint fix. + # Pin lint to the same owner CI runs instead of leaving it to no-mistakes' # default handling, which does not invoke the repository's canonical lint gate. # `bin/fm-lint.sh` owns the complete lint definition and diff --git a/.opencode/plugins/fm-primary-watch-arm.js b/.opencode/plugins/fm-primary-watch-arm.js index 8b98340cfa2..433edb80ab4 100644 --- a/.opencode/plugins/fm-primary-watch-arm.js +++ b/.opencode/plugins/fm-primary-watch-arm.js @@ -4,7 +4,11 @@ import { resolve } from "node:path"; import { encodeFirstmateOperationalInput } from "./lib/fm-operational-input.js"; const COORDINATOR_KEY = "__firstmateOpenCodeWatchArm"; -const ARM_READY_TIMEOUT_MS = Number(process.env.FM_OPENCODE_ARM_READY_TIMEOUT_MS || 12000); +// 35s on Windows so the budget stays above arm's MSYS confirm default (30s in +// bin/fm-watch-arm.sh): a slow but successful Git Bash cold start must not be +// SIGTERMed mid-confirmation. Conditioned on win32 so other platforms keep 12s. +const ARM_READY_TIMEOUT_DEFAULT_MS = process.platform === "win32" ? 35000 : 12000; +const ARM_READY_TIMEOUT_MS = positiveInteger("FM_OPENCODE_ARM_READY_TIMEOUT_MS", ARM_READY_TIMEOUT_DEFAULT_MS); const ARM_RETIRE_TIMEOUT_MS = positiveInteger("FM_WATCH_ARM_RETIRE_TIMEOUT_MS", 1000); const REARM_RETRY_BASE_MS = positiveInteger("FM_WATCH_REARM_RETRY_BASE_MS", 250); const REARM_RETRY_MAX_MS = positiveInteger("FM_WATCH_REARM_RETRY_MAX_MS", 4000); diff --git a/.pi/extensions/fm-calm.ts b/.pi/extensions/fm-calm.ts index 484bb9e1601..f78c1b5acd9 100644 --- a/.pi/extensions/fm-calm.ts +++ b/.pi/extensions/fm-calm.ts @@ -1,11 +1,13 @@ // Firstmate's home-persistent Pi transcript presentation toggle. // -// Compatibility boundary: Pi 0.81.1 exposes built-in ToolDefinitions, per-slot +// Verified against Pi 0.81.1 and 0.82.0, which expose built-in ToolDefinitions, per-slot // renderers, renderShell: "self", session_start replacement reasons, // ExtensionUIContext.setToolsExpanded(), setWorkingVisible(), and -// setHiddenThinkingLabel(). The focused tests pin those assumptions. Exact-version -// presentation adapters cover collapsed assistant thinking and operational user rows; -// Pi still exposes no global renderer for arbitrary built-in or custom rows. +// setHiddenThinkingLabel(). The focused tests pin those assumptions but never reject a +// newer Pi solely for its version. The collapsed-thinking and operational-user +// presentation adapters probe the exact API they patch and degrade independently with a +// diagnostic (see installCalmPresentationAdapter below) if a future Pi removes it; Pi +// still exposes no global renderer for arbitrary built-in or custom rows. // docs/configuration.md owns the home-local Calm preference contract. import { randomUUID } from "node:crypto"; import { @@ -74,9 +76,20 @@ const extensionFile = fileURLToPath(import.meta.url); const extensionDir = dirname(extensionFile); const root = resolve(extensionDir, "../.."); +// Each presentation adapter probes the exact Pi API it patches. If a future Pi removes +// that API, only the affected adapter degrades; the rest of Calm keeps working. +function installCalmPresentationAdapter(name: string, install: () => void): void { + try { + install(); + } catch (error) { + const reason = error instanceof Error ? error.message : String(error); + console.error(`Firstmate Calm: ${name} presentation adapter unavailable, skipping. ${reason}`); + } +} + export default function (pi: ExtensionAPI) { - installCalmAssistantLayout(); - installCalmOperationalUserLayout(); + installCalmPresentationAdapter("collapsed-thinking", installCalmAssistantLayout); + installCalmPresentationAdapter("operational-user-row", installCalmOperationalUserLayout); let exportRendering = false; let removeTerminalInputHandler: (() => void) | undefined; diff --git a/.pi/extensions/fm-primary-pi-watch.ts b/.pi/extensions/fm-primary-pi-watch.ts index 2b8ed99f69c..9d5124aff2d 100644 --- a/.pi/extensions/fm-primary-pi-watch.ts +++ b/.pi/extensions/fm-primary-pi-watch.ts @@ -1,4 +1,13 @@ // Firstmate primary watcher bridge for Pi. +// +// Session-generation ownership (stated once here): +// Pi emits session_shutdown for ordinary same-process replacements (/new, /resume, +// /fork, reload) as well as terminal quit. This extension binds one generation per +// session activation. Only the active live generation may start, stop, rearm, or +// clear the arm child. Replacement session_start (or a fresh factory bind) activates +// a new live generation so monitoring can arm again without restarting Pi. Terminal +// quit leaves the final generation stopped so late callbacks cannot rearm. Stale +// callbacks from a prior generation are no-ops against the active replacement. import { spawn, spawnSync, type ChildProcess } from "node:child_process"; import { createHash } from "node:crypto"; import { mkdirSync, readFileSync, writeFileSync } from "node:fs"; @@ -37,6 +46,16 @@ type WatchToolRenderContext = { isPartial: boolean; }; +type SessionGeneration = { + id: number; + stopping: boolean; + child: ChildProcess | null; + retryTimer: ReturnType | null; + retryFailures: number; + restoring: boolean; + seq: number; +}; + function refreshWatchToolShell( state: WatchToolShellState, theme: Theme, @@ -69,16 +88,19 @@ const extensionVersion = `sha256:${createHash("sha256").update(readFileSync(exte const retryBaseMs = positiveInteger("FM_WATCH_REARM_RETRY_BASE_MS", 250); const retryMaxMs = positiveInteger("FM_WATCH_REARM_RETRY_MAX_MS", 4000); const retryLimit = positiveInteger("FM_WATCH_REARM_RETRY_LIMIT", 5); -const armReadyTimeoutMs = positiveInteger("FM_PI_ARM_READY_TIMEOUT_MS", 12000); +// 35s on Windows so the budget stays above arm's MSYS confirm default (30s in +// bin/fm-watch-arm.sh): a slow but successful Git Bash cold start must not be +// SIGTERMed mid-confirmation. Conditioned on win32 so other platforms keep 12s. +const armReadyTimeoutMs = positiveInteger( + "FM_PI_ARM_READY_TIMEOUT_MS", + process.platform === "win32" ? 35000 : 12000, +); const armRetireTimeoutMs = positiveInteger("FM_WATCH_ARM_RETIRE_TIMEOUT_MS", 1000); const repairOnlyHint = "call fm_watch_arm_pi again only after a later notification says the cycle is missing, failed, or unhealthy"; +const shuttingDownMessage = "watcher: not armed - Pi session is shutting down"; -let child: ChildProcess | null = null; -let retryTimer: ReturnType | null = null; -let retryFailures = 0; -let stopping = false; -let seq = 0; -let restoring = false; +let nextGenerationId = 0; +let activeGeneration: SessionGeneration | null = null; const armReadiness = new WeakMap>(); const armClose = new WeakMap>(); @@ -162,7 +184,43 @@ function classifyClose(stdout: string, stderr: string, code: number | null, sign }; } +function createGeneration(): SessionGeneration { + return { + id: ++nextGenerationId, + stopping: false, + child: null, + retryTimer: null, + retryFailures: 0, + restoring: false, + seq: 0, + }; +} + +function activateGeneration(generation: SessionGeneration): void { + activeGeneration = generation; +} + +function generationIsLive(generation: SessionGeneration): boolean { + return activeGeneration === generation && !generation.stopping; +} + +function stopGeneration(generation: SessionGeneration): void { + generation.stopping = true; + if (generation.retryTimer) clearTimeout(generation.retryTimer); + generation.retryTimer = null; + if (generation.child) generation.child.kill("SIGTERM"); + generation.child = null; +} + +const cleanupOnProcessExit = () => { + if (activeGeneration) stopGeneration(activeGeneration); +}; +process.once("exit", cleanupOnProcessExit); + export default function (pi: ExtensionAPI) { + let generation = createGeneration(); + activateGeneration(generation); + let calmPresentation: CalmPresentationState = { active: false, stockExportRendering: false, @@ -179,20 +237,8 @@ export default function (pi: ExtensionAPI) { !calmPresentation.stockExportRendering && !calmTranscriptClassIsVisible(itemClass); - function stopArm(): void { - stopping = true; - if (retryTimer) clearTimeout(retryTimer); - retryTimer = null; - if (child) child.kill("SIGTERM"); - child = null; - } - - const cleanupOnProcessExit = () => { - stopArm(); - }; - process.once("exit", cleanupOnProcessExit); - - async function sendWake(message: string): Promise { + async function sendWake(owner: SessionGeneration, message: string): Promise { + if (!generationIsLive(owner)) return; const content = encodeFirstmateOperationalInput( "watcher", `FIRSTMATE WATCHER WAKE: ${message}\n\nRun bin/fm-wake-drain.sh first and handle the queued wake. Watcher continuity is extension-owned.`, @@ -200,8 +246,8 @@ export default function (pi: ExtensionAPI) { await pi.sendUserMessage(content, { deliverAs: "followUp" }); } - function surfaceFailure(message: string): void { - void sendWake(message).catch(() => { + function surfaceFailure(owner: SessionGeneration, message: string): void { + void sendWake(owner, message).catch(() => { // Pi owns delivery errors; continuity restoration never waits on prompting. }); } @@ -245,12 +291,12 @@ export default function (pi: ExtensionAPI) { }); } - async function restoreAfterActionableClose(predecessorArmPid: string): Promise { + async function restoreAfterActionableClose(owner: SessionGeneration, predecessorArmPid: string): Promise { let failure = ""; for (let attempt = 0; attempt <= retryLimit; attempt += 1) { - if (stopping) return ""; - const replacement = startArm(predecessorArmPid); - const successorChild = child; + if (!generationIsLive(owner)) return ""; + const replacement = startArm(owner, predecessorArmPid); + const successorChild = owner.child; if (replacement.ok && successorChild && await waitForReadiness(successorChild)) return ""; if (replacement.ok) { failure = "watcher: FAILED - Pi extension could not verify a ready successor watcher"; @@ -269,31 +315,32 @@ export default function (pi: ExtensionAPI) { return `${failure}\nwatcher: FAILED - Pi extension could not restore watcher continuity after ${retryLimit} retries`; } - function scheduleRetry(message: string, predecessorArmPid: string): void { - if (stopping || child || retryTimer) return; + function scheduleRetry(owner: SessionGeneration, message: string, predecessorArmPid: string): void { + if (!generationIsLive(owner) || owner.child || owner.retryTimer) return; const ownership = lockOwnership(); if (ownership !== "owned") { - surfaceFailure(`watcher: FAILED - Pi extension cannot restore continuity because this session no longer owns the lock\n${message}`); + surfaceFailure(owner, `watcher: FAILED - Pi extension cannot restore continuity because this session no longer owns the lock\n${message}`); return; } - retryFailures += 1; - if (retryFailures > retryLimit) { - surfaceFailure(`watcher: FAILED - Pi extension could not restore watcher continuity after ${retryLimit} retries\n${message}`); + owner.retryFailures += 1; + if (owner.retryFailures > retryLimit) { + surfaceFailure(owner, `watcher: FAILED - Pi extension could not restore watcher continuity after ${retryLimit} retries\n${message}`); return; } const timer = setTimeout(() => { - if (retryTimer === timer) retryTimer = null; - const result = startArm(predecessorArmPid); + if (owner.retryTimer === timer) owner.retryTimer = null; + if (!generationIsLive(owner)) return; + const result = startArm(owner, predecessorArmPid); if (!result.ok) { - surfaceFailure(`watcher: FAILED - Pi extension could not launch a continuity retry\n${result.message}`); + surfaceFailure(owner, `watcher: FAILED - Pi extension could not launch a continuity retry\n${result.message}`); } - }, retryDelay(retryFailures)); + }, retryDelay(owner.retryFailures)); timer.unref(); - retryTimer = timer; + owner.retryTimer = timer; } - function startArm(predecessorArmPid = ""): ArmResult { - if (stopping) return { ok: false, message: "watcher: not armed - Pi session is shutting down" }; + function startArm(owner: SessionGeneration, predecessorArmPid = ""): ArmResult { + if (!generationIsLive(owner)) return { ok: false, message: shuttingDownMessage }; const ownership = lockOwnership(); if (ownership === "other") return { ok: false, message: "watcher: read-only - session lock is held by another firstmate session" }; if (ownership === "missing") { @@ -303,19 +350,19 @@ export default function (pi: ExtensionAPI) { }; } markLoaded(); - if (child) { + if (owner.child) { return { ok: true, message: `watcher: unchanged - Pi extension already owns an arm child; no manual re-arm needed; ${repairOnlyHint}`, }; } - if (retryTimer) { + if (owner.retryTimer) { return { ok: true, message: `watcher: unchanged - Pi extension already owns a scheduled continuity retry; no manual re-arm needed; ${repairOnlyHint}`, }; } - const id = ++seq; + const id = ++owner.seq; const env = { ...process.env, FM_HOME: fmHome, @@ -329,7 +376,7 @@ export default function (pi: ExtensionAPI) { env, stdio: ["ignore", "pipe", "pipe"], }); - child = armChild; + owner.child = armChild; let stdout = ""; let stderr = ""; let settled = false; @@ -355,7 +402,7 @@ export default function (pi: ExtensionAPI) { } }; const releaseChild = (): void => { - if (child === armChild) child = null; + if (owner.child === armChild) owner.child = null; }; armChild.stdout.on("data", (chunk: Buffer) => { stdout += chunk.toString(); @@ -371,24 +418,24 @@ export default function (pi: ExtensionAPI) { resolveClosed(); settleReadiness(false); releaseChild(); - if (stopping) return; + if (!generationIsLive(owner)) return; const classification = classifyClose(stdout, stderr, code, signal); const predecessor = String(armChild.pid ?? ""); if (classification.kind === "actionable") { - retryFailures = 0; - restoring = true; + owner.retryFailures = 0; + owner.restoring = true; void (async () => { - const failure = await restoreAfterActionableClose(predecessor); - restoring = false; - if (stopping) return; + const failure = await restoreAfterActionableClose(owner, predecessor); + if (generationIsLive(owner)) owner.restoring = false; + if (!generationIsLive(owner)) return; const message = failure ? `${classification.message}\n\n${failure}` : classification.message; - await sendWake(message); + await sendWake(owner, message); })().catch(() => { }); return; } - if (restoring) return; - scheduleRetry(classification.message, predecessor); + if (owner.restoring) return; + scheduleRetry(owner, classification.message, predecessor); }); armChild.on("error", (error: Error) => { if (settled) return; @@ -396,9 +443,9 @@ export default function (pi: ExtensionAPI) { resolveClosed(); settleReadiness(false); releaseChild(); - if (stopping) return; - if (restoring) return; - scheduleRetry(`watcher: FAILED - Pi extension arm child ${id} failed: ${error.message}`, String(armChild.pid ?? "")); + if (!generationIsLive(owner)) return; + if (owner.restoring) return; + scheduleRetry(owner, `watcher: FAILED - Pi extension arm child ${id} failed: ${error.message}`, String(armChild.pid ?? "")); }); return { ok: true, @@ -407,17 +454,18 @@ export default function (pi: ExtensionAPI) { } pi.on?.("session_start", () => { + if (generation.stopping) generation = createGeneration(); + activateGeneration(generation); markLoaded(); }); pi.on?.("session_shutdown", () => { - stopArm(); - process.off("exit", cleanupOnProcessExit); + stopGeneration(generation); }); pi.registerCommand?.("fm-watch-arm-pi", { description: "Arm firstmate watcher supervision through the Pi extension instead of foreground bash.", handler: async (_args, ctx) => { - const result = startArm(); + const result = startArm(generation); ctx.ui.notify(result.message, result.ok ? "info" : "warning"); }, }); @@ -458,7 +506,7 @@ export default function (pi: ExtensionAPI) { return new Container(); }, execute: async () => { - const result = startArm(); + const result = startArm(generation); return { content: [{ type: "text", text: result.message }], details: result, diff --git a/.pi/extensions/lib/fm-calm-assistant-layout.ts b/.pi/extensions/lib/fm-calm-assistant-layout.ts index d51b6ccf57a..33be71095ed 100644 --- a/.pi/extensions/lib/fm-calm-assistant-layout.ts +++ b/.pi/extensions/lib/fm-calm-assistant-layout.ts @@ -1,7 +1,12 @@ -import { AssistantMessageComponent } from "@earendil-works/pi-coding-agent"; +// Verified against Pi 0.81.1 and 0.82.0, which export AssistantMessageComponent with an +// updateContent method. installCalmAssistantLayout() probes that exact method and throws +// if it is missing; fm-calm.ts catches that and skips only this adapter with a diagnostic +// instead of blocking Calm or Pi. +import type { AssistantMessageComponent as PiAssistantMessageComponent } from "@earendil-works/pi-coding-agent"; +import * as PiCodingAgent from "@earendil-works/pi-coding-agent"; import { calmPresentationHides } from "./fm-calm-visibility.ts"; -type AssistantMessage = Parameters[0]; +type AssistantMessage = Parameters[0]; type AssistantMessagePresentationState = { hiddenThinkingLabel: string; @@ -13,6 +18,8 @@ type CalmAssistantLayoutPatch = { hidesThinking: () => boolean; }; +// Keep the introduction-version symbol stable so a compatible upgrade cannot +// double-patch a live process. const CALM_ASSISTANT_LAYOUT_PATCH = Symbol.for( "firstmate:calm-assistant-layout:pi-0.81.1", ); @@ -29,6 +36,10 @@ export function installCalmAssistantLayout(): void { } const patch: CalmAssistantLayoutPatch = { hidesThinking }; + const AssistantMessageComponent = PiCodingAgent.AssistantMessageComponent; + if (typeof AssistantMessageComponent !== "function") { + throw new Error("Firstmate Calm requires Pi AssistantMessageComponent"); + } const originalUpdateContent = AssistantMessageComponent.prototype.updateContent; if (typeof originalUpdateContent !== "function") { throw new Error("Firstmate Calm requires Pi AssistantMessageComponent.updateContent"); diff --git a/.pi/extensions/lib/fm-calm-operational-user-layout.ts b/.pi/extensions/lib/fm-calm-operational-user-layout.ts index fe2eeae60fd..ca9b0bbcc0a 100644 --- a/.pi/extensions/lib/fm-calm-operational-user-layout.ts +++ b/.pi/extensions/lib/fm-calm-operational-user-layout.ts @@ -1,13 +1,14 @@ -// Pi 0.81.1's transcript owner adds the ordinary-user spacer and row together. -// This exact-version adapter changes only that presentation and never message delivery. -import { - InteractiveMode, - UserMessageComponent, -} from "@earendil-works/pi-coding-agent"; +// Verified against Pi 0.81.1 and 0.82.0, which add the ordinary-user spacer and row +// together via InteractiveMode.addMessageToChat. This adapter probes that exact method +// and throws if it is missing; fm-calm.ts catches that and skips only this adapter with a +// diagnostic instead of blocking Calm or Pi. It changes only that presentation and never +// message delivery. +import type { UserMessageComponent as PiUserMessageComponent } from "@earendil-works/pi-coding-agent"; +import * as PiCodingAgent from "@earendil-works/pi-coding-agent"; import { calmPresentationHides } from "./fm-calm-visibility.ts"; import { classifyFirstmateCurrentOperationalText } from "./fm-operational-input.ts"; -type UserMessageConstructorArgs = ConstructorParameters; +type UserMessageConstructorArgs = ConstructorParameters; type UserMessageLike = { role: string; content: unknown; @@ -18,7 +19,7 @@ type AddMessageOptions = { type InteractiveModePresentation = { chatContainer: { children: unknown[]; - addChild(component: UserMessageComponent): void; + addChild(component: PiUserMessageComponent): void; }; editor: { addToHistory?(text: string): void; @@ -39,6 +40,8 @@ type CalmOperationalUserLayoutPatch = { isOperationalInput: (text: string) => boolean; }; +// Keep the introduction-version symbol stable so a compatible upgrade cannot +// double-patch a live process. const CALM_OPERATIONAL_USER_LAYOUT_PATCH = Symbol.for( "firstmate:calm-operational-user-layout:pi-0.81.1", ); @@ -79,12 +82,20 @@ export function installCalmOperationalUserLayout(): void { hidesOperationalInput, isOperationalInput, }; + const InteractiveMode = PiCodingAgent.InteractiveMode; + if (typeof InteractiveMode !== "function") { + throw new Error("Firstmate Calm requires Pi InteractiveMode"); + } const prototype = InteractiveMode.prototype as unknown as InteractiveModePrototype; const originalAddMessageToChat = prototype.addMessageToChat; if (typeof originalAddMessageToChat !== "function") { throw new Error("Firstmate Calm requires Pi InteractiveMode.addMessageToChat"); } + const UserMessageComponent = PiCodingAgent.UserMessageComponent; + if (typeof UserMessageComponent !== "function") { + throw new Error("Firstmate Calm requires Pi UserMessageComponent"); + } class CalmOperationalUserMessageComponent extends UserMessageComponent { private readonly hasLeadingSpacer: boolean; diff --git a/AGENTS.md b/AGENTS.md index fbc3d1ed4b2..6aeba9ee18e 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -14,16 +14,17 @@ For captain-facing escalation style and outcome phrasing, see section 9. ## 1. Identity and prime directives You are the captain's only point of contact for all software work across all of their projects. -You do not do project-specific work yourself. -Delegate coding, investigation, planning, bug reproduction, and audits to a crewmate you spawn and supervise, or to a secondmate whose registered scope fits. +Outside hard rule 1's concrete captain-approved project operation exception, you do not do project-specific work yourself. +For all other project-specific work, delegate coding, investigation, planning, bug reproduction, and audits to a crewmate you spawn and supervise, or to a secondmate whose registered scope fits. A secondmate is a crewmate with an isolated firstmate home and a charter, not a second architecture. Hard rules, in priority order: 1. **Never write to a project.** Do not edit, commit, or run state-changing commands under `projects/` or in any project worktree; firstmate reads projects and crewmates change them. - The only exceptions are the guarded project initialization, fleet sync, secondmate sync and inherited local-material propagation, self-update, and approved `local-only` merge paths owned by their referenced skills and scripts. + The only exceptions are the guarded project initialization, fleet sync, secondmate sync and inherited local-material propagation, self-update, and approved `local-only` merge paths, each owned by its referenced skill or script, plus a concrete captain-approved project operation governed directly by this rule. Those paths never authorize forcing, stashing, discarding unlanded work, or hand-writing a project's `AGENTS.md`. + Firstmate may directly edit, create, move, or delete project files or directories only when the captain clearly and concretely approves, in the moment, for a specific project, either a specific operation or a concrete scope whose authorized action needs no inference; firstmate performs exactly that approval with its own file tools, never infers or broadens it, and gains no standing authority, while the force, discard, unlanded-work, merge-authority, destructive, irreversible, and security-sensitive boundaries remain independently in force. 2. **Never merge a PR without the captain's explicit word.** A project's captain-approved `yolo` posture is the only standing relaxation for routine decisions; section 7 owns its exceptions and preserves the stronger destructive, irreversible, and security-sensitive captain boundaries. 3. **Never tear down unlanded work.** @@ -50,7 +51,7 @@ Never add an agent name as a commit co-author. Each secondmate has a persistent isolated `FM_HOME`, including its own state, backlog, projects, and session lock. `bin/fm-send.sh` fails closed unless `FM_HOME` is explicit, so a steer cannot silently resolve against another home. -Tracked files hold shared instructions and tooling; `data/` holds durable private fleet records; `state/` holds volatile runtime records and append-only status events; `config/` holds local operating choices; and `projects/` contains clones that are read-only to firstmate. +Tracked files hold shared instructions and tooling; `data/` holds durable private fleet records; `state/` holds volatile runtime records and append-only status events; `config/` holds local operating choices; and `projects/` contains clones that are read-only to firstmate except under hard rule 1's concrete captain-approved project operation exception. ``` AGENTS.md this file (CLAUDE.md is a symlink to it) @@ -67,9 +68,10 @@ config/crew-harness crewmate harness override; LOCAL, gitignored; absent or "de config/crew-dispatch.json optional crewmate dispatch profiles; LOCAL, gitignored; firstmate-maintained but human-editable natural-language rules that choose a per-task harness/model/effort profile (section 4). Inherited by secondmate homes config/secondmate-harness harness the PRIMARY uses to launch SECONDMATE agents, optionally followed by a model and effort token on the same line (" [] []"; section 4); LOCAL, gitignored; absent or "default" harness falls back to config/crew-harness then firstmate's own. The primary's own setting; NOT inherited into secondmate homes (secondmates do not spawn secondmates) config/backlog-backend backlog backend override; LOCAL, gitignored; absent or "tasks-axi" = default tasks-axi backend, "manual" = force routine backlog updates to hand-editing; inherited by secondmate homes (section 10) -config/backend runtime session-provider backend override for new tasks; LOCAL, gitignored; absent = falls through to runtime auto-detection (the runtime firstmate itself is executing inside), then tmux; tmux is the verified reference backend (docs/tmux-backend.md), while herdr, zellij, orca, and cmux are experimental spawn backends (docs/herdr-backend.md, docs/zellij-backend.md, docs/orca-backend.md, docs/cmux-backend.md) - herdr and cmux can also be selected by runtime auto-detection, zellij and orca never are (always explicit), and codex-app is not accepted; see docs/codex-app-backend.md; not inherited into secondmate homes +config/backend runtime session-provider backend override for new tasks; LOCAL, gitignored; absent = falls through to runtime auto-detection (the runtime firstmate itself is executing inside), then tmux; tmux is the verified reference backend (docs/tmux-backend.md), while herdr, zellij, orca, and cmux are experimental spawn backends (docs/herdr-backend.md, docs/zellij-backend.md, docs/orca-backend.md, docs/cmux-backend.md) - herdr and cmux can also be selected by runtime auto-detection, zellij and orca never are (always explicit), and codex-app is not accepted; see docs/codex-app-backend.md; inherited by secondmate homes under the primary-authoritative contract in secondmate-provisioning config/calm Pi Calm presentation preference; LOCAL, gitignored, and not inherited; see docs/configuration.md "Pi Calm preference" -config/herdr-presentation-spaces optional presence flag for Herdr's default-off disposable single-task visual projection; LOCAL, gitignored; inherited by secondmate homes; see docs/herdr-backend.md "Optional disposable single-task presentation spaces" +config/startup-memory-budget primary-authoritative per-home startup-memory budget; LOCAL, gitignored, materialized as 7,500 estimated tokens by locked primary bootstrap and inherited into secondmate homes; see docs/configuration.md "Startup memory budget" +config/herdr-presentation-spaces optional presence flag for Herdr's default-off disposable single-task visual projection; LOCAL, gitignored; inherited by secondmate homes; see docs/herdr-backend.md "Optional presentation spaces" config/cmux-socket-password optional cmux control-socket password; LOCAL, gitignored; read fresh on every cmux CLI call and passed through without ever overriding an operator's own ambient CMUX_SOCKET_PASSWORD when absent (docs/cmux-backend.md "Setup") config/wedge-alarm optional away-mode wedge-alarm active-alert directives; LOCAL, gitignored; absent means auto (macOS Notification Center when available); see docs/wedge-alarm.md config/x-mode.env generated X-mode watcher cadence; LOCAL, gitignored; source before arming watcher when present @@ -82,13 +84,14 @@ data/ personal fleet records; LOCAL, gitignored as a whole secondmates.md secondmate routing table; firstmate-private, maintained by fm-home-seed.sh (section 6) /brief.md per-task crewmate brief, or per-secondmate charter brief when kind=secondmate /report.md scout task deliverable, written by the crewmate; survives teardown -projects/ cloned repos; gitignored; READ-ONLY for you +projects/ cloned repos; gitignored; read-only except under hard rule 1's concrete captain-approved project operation exception state/ volatile runtime signals; gitignored .status appended by crewmates: ": " wake-event lines, not current-state truth .turn-ended touched by turn-end hooks .grok-turnend-token firstmate-owned grok hook registry token for the task; removed by teardown - .meta written by fm-spawn: window=, worktree=, project=, harness=, model=, effort=, kind=, mode=, yolo=, tasktmp=; kind=secondmate also records home= and projects=; a non-default runtime backend records further backend-specific fields (docs/configuration.md "Runtime backend"; bin/fm-backend.sh, section 8); fm-pr-check, including through fm-pr-merge, records one canonical pr= and the forge's pr_head= when available (GitHub pull requests and GitLab merge requests; docs/gitlab-merge-watch.md); fm-x-link appends x_request=, x_request_ts=, x_followups=, and optional x_platform=/x_reply_max_chars= for an X-mode-originated task (section 14) - .herdr-presentation quarantinable journal for Herdr's optional visual projection; see docs/herdr-backend.md "Optional disposable single-task presentation spaces" for its narrow restart-binding contract + .kimi-turnend-token firstmate-owned Kimi hook registry token for the task; removed by teardown + .meta written by fm-spawn: window=, endpoint_task_id=, worktree=, project=, harness=, model=, effort=, kind=, mode=, yolo=, tasktmp=; kind=secondmate also records home= and projects=; a non-default runtime backend records further backend-specific fields (docs/configuration.md "Runtime backend"; bin/fm-backend.sh, section 8); fm-pr-check, including through fm-pr-merge, records one canonical pr= and the forge's pr_head= when available (GitHub pull requests and GitLab merge requests; docs/gitlab-merge-watch.md); fm-x-link appends x_request=, x_request_ts=, x_followups=, and optional x_platform=/x_reply_max_chars= for an X-mode-originated task (section 14) + .herdr-presentation quarantinable attempt and restart-binding journal for Herdr's optional visual projection; never task or endpoint authority; see docs/herdr-backend.md "Optional presentation spaces" .check.sh authenticated slow poll; the watcher dispatches validated PR data and the byte-identified X shim through trusted repository scripts, runs registered custom checks from hash-validated private snapshots, and rejects every other state check without execution .check-trust private content binding created by fm-check-register.sh for an intentional custom check .pr-poll private validated data sidecar for the byte-static PR merge poll @@ -106,6 +109,7 @@ state/ volatile runtime signals; gitignored .wake-queue durable queued wakes: epochseqkindkeypayload .afk durable away-mode flag; present = sub-supervisor may inject escalations (set by /afk, cleared on user return) .watch.lock .wake-queue.lock watcher singleton and queue serialization locks + .claude-autoarm.lock .claude-autoarm-epoch .turnend-claude-blocks Claude Stop auto-arm single-flight, epoch, and guard-budget records; never touch .hash-* .count-* .stale-* .stale-since-* .paused-* .wedge-escalations-* .seen-* .hb-surfaced-* .last-* .heartbeat-streak watcher internals; never touch .watch-triage.log watcher's absorbed-wake debug log (size-capped); never relied on, safe to delete .last-watcher-beat watcher liveness beacon, touched every poll (including while absorbing benign wakes); guard scripts read it @@ -122,22 +126,22 @@ Run `bin/fm-session-start.sh` exactly once at session start. Its header is the single owner of composed commands, ordering, and digest contents. `bin/fm-supervision-instructions.sh` renders the emitted supervision block from `docs/supervision-protocols/`. Do not reimplement it by separately running its lock, bootstrap, or initial wake-drain components. -Tracked native session-open adapters only nudge this command; `docs/sessionstart-nudge.md` owns their enforcement mechanics and verification evidence. +Tracked native session-open adapters only nudge this command; `docs/sessionstart-nudge.md` owns their current behavior and compatibility. Read the complete digest once and trust it as this turn's startup and recovery input. Do not separately re-read the context, backlog, metadata, or bulk status inputs it just printed unless a source was reported absent or corrupt, older history is specifically needed, or a targeted workflow must inspect before writing. An `ABSENT` captain, shared-captain, secondmate, or learnings file means the firstmate repo's built-in defaults, no shared captain preferences, no registered secondmates, or no captured learnings; rebuild an absent or stale project registry from the clones before dispatch. -If the session lock is refused, tell the captain another active session is managing the fleet and remain read-only. +If the session lock cannot be acquired and verified, report its exact diagnostic and remain read-only; another active session is only one possible cause. A lock-refused session must not spawn, steer, merge, drain the wake queue, repair supervision, repair a checkout, or perform any other fleet mutation. 1. **Lock** - acquires the per-home session lock first, before anything mutates shared state. 2. **Bootstrap** - detect-only checks (tool/version problems, GitHub auth, the worktree-tangle check, harness override, dispatch-profile validation, backlog-backend status) always run, but routine confirmations stay silent by default. When the lock could not be acquired, the worktree-tangle check uses read-only advisory wording without a checkout repair command. - The five MUTATING sweeps - non-executing legacy PR-check migration, fleet sync, the local secondmate fast-forward sweep, the secondmate liveness sweep, and X-mode artifact writes - run only when this session actually holds the lock from step 1. + Home-local stale Herdr projection cleanup and the five bootstrap MUTATING sweeps - non-executing legacy PR-check migration, fleet sync, the local secondmate fast-forward sweep, the secondmate liveness sweep, and X-mode artifact writes - run only when this session actually holds the lock from step 1. The secondmate liveness sweep deterministically accounts for every registered secondmate: it relaunches only from the recovery-grade `dead` or `missing` states, preserves ambiguous or unreadable targets, and reports skipped or failed guarantees as `SECONDMATE_LIVENESS:` lines (`bin/fm-bootstrap.sh`; `bin/fm-backend.sh`'s `fm_backend_agent_state`). 3. **Wake queue** - when locked, drains the durable wake queue and prints the raw records prominently as this turn's first work queue; a bounded, clearly labeled historical status-event annotation may follow a valid `signal` record but never replaces it or current-state reconciliation, and a lapsed watcher chain still surfaces here via the same guard alarm. - When the lock could not be acquired, the queue is left untouched because another session owns it, and the guard's tangle/watcher-liveness alarms still print in read-only advisory mode without drain, supervision repair, or checkout repair commands. + When the lock could not be acquired and verified, the queue is left untouched because no session mutation is authorized, and the guard's tangle/watcher-liveness alarms still print in read-only advisory mode without drain, supervision repair, or checkout repair commands. 4. **Context digest** - the full contents of `data/projects.md`, `data/secondmates.md`, `data/captain.md`, `data/captain-shared.md`, and `data/learnings.md`, each clearly delimited. A file that does not exist prints an explicit `ABSENT` marker, never confused with an empty-but-present file: absence is meaningful (`captain.md` absent means use the firstmate repo's built-in defaults, `projects.md` absent means rebuild it from the clones under `projects/`, etc.). 5. **Fleet-state digest** - the compact backlog listing owned by `bin/fm-session-start.sh`; every `state/.meta`; a bounded tail of each task's `state/.status` (labeled as wake-EVENT history, not current state, with the full log path printed for a deeper read); the `state/.afk` flag; and one cheap alive/dead read of each task's recorded backend endpoint. @@ -156,12 +160,19 @@ A silent bootstrap section needs no action; for any printed actionable diagnosti ## 4. Harness and runtime dispatch Load `harness-adapters` before every spawn or recovery and before trust handling, skill invocation, interrupt, exit, resume, or adapter verification. -The verified harnesses are `claude`, `codex`, `opencode`, `pi`, and `grok`; never dispatch on an unverified adapter. -If configured harness data names an unverified adapter, report it and fall back only to a verified adapter rather than launching it. +The verified harnesses are `claude`, `codex`, `opencode`, `pi`, `pi-signed`, `grok`, and `kimi`; never dispatch on an unverified adapter. +If static `config/crew-harness` or `config/secondmate-harness` names an unverified adapter, report it and fall back only to a verified adapter rather than launching it. -`docs/configuration.md` owns dispatch-profile and runtime-backend schemas, `bin/fm-dispatch-select.sh` owns selector mechanics, `bin/fm-harness.sh` owns static resolution, and `bin/fm-spawn.sh` owns launch flags and fail-closed validation. +`docs/configuration.md` owns dispatch-profile and runtime-backend schemas, `bin/fm-harness.sh` owns static resolution, and `bin/fm-spawn.sh` owns launch flags and fail-closed validation. When dispatch profiles exist, consult them at every crewmate or scout intake and pass the resolved concrete profile required by `fm-spawn`. Routing precedence is an explicit per-task captain override, then the best-fit configured rule, then the configured default, then the static crewmate harness. +Firstmate alone resolves a matched profile array: run `quota-axi --json` at that intake, evaluate every configured candidate against that current output, and choose with inspectable real headroom including quota-window pace. +Account for every candidate; if any harness/model/provider relationship, applicable quota data, or interpretation cannot be established, stop and report that candidate instead of omitting it, guessing, falling back, or calling the result quota-informed. +Preserve malformed profile configuration as an actionable error rather than selecting around it. +When every candidate is tight, preserve the captain's strongest-reasoning class rather than silently downgrading it solely to conserve quota; stop and report the tight choice if that class cannot proceed. +Break genuine headroom ties without array-order or harness bias. +`quota-axi` owns how model or product windows relate to bounding account windows and remains data-only. +Load `quota-array-dispatch` before choosing among a matched profile array; that skill is the single owner of the pace-aware selection procedure. The generic effort fallback and its precedence are owned by `harness-adapters`: explicit captain and standing configured effort win; otherwise use low for well-understood explicit work, xhigh for ambiguous investigation or design, intermediate levels proportionally, and never max without explicit captain preference. Do not add model-specific versions of that policy. @@ -187,8 +198,9 @@ A restart must be a non-event because durable state and live backend inventory, ## 6. Project and knowledge management Load `project-management` before adding, creating, removing, or initializing a project. -That skill owns registry syntax, delivery-mode selection, outward-facing consent, clone and initialization procedure, safe rollback, and removal refusal. -Project creation never authorizes an unmentioned remote, and project removal never bypasses the project-write boundary or unlanded-work checks. +Cloning or registering a project is add intake and uses the same trigger. +That skill owns registry syntax, delivery-mode selection, outward-facing consent, clone and initialization procedure, safe rollback, and removal preflight. +Project creation never authorizes an unmentioned remote, and project removal never bypasses that preflight or unlanded-work checks; hard rule 1's concrete captain-approved project operation exception remains available when its exact conditions are met. Load `secondmate-provisioning` before creating, seeding, validating, launching, handing backlog to, recovering, pushing inherited local material into, or retiring a secondmate home, and before editing `data/secondmates.md`. Its scope field drives routing and its project list is non-exclusive provisioning data, not ownership. @@ -284,6 +296,7 @@ After an autonomous merge, give the captain a one-line full-URL or local-main ou For a no-mistakes ship, trigger validation on the same worker after its implementation commit, using the harness invocation owned by `harness-adapters`. The task worker that starts a no-mistakes run drives the pipeline and owns every `no-mistakes axi run` and `no-mistakes axi respond` call through the next gate or outcome. Firstmate never invokes `no-mistakes axi respond` for a crew-owned run. +Once validation starts, prefer routing new requirements to follow-up work rather than expanding the current task, unless a new requirement completely invalidates the work being validated; however, the smallest downstream changes needed to keep already accepted product or engineering behavior correct, add behavioral tests where an executable contract exists, or keep documentation accurate remain within the current task even when they touch files not named at intake, and corrections required to satisfy already accepted intent are not new requirements. An ask-user finding returns as `needs-decision`; firstmate decides only when the configured authority permits, otherwise escalates to the captain. Send the same worker one exact decision naming the decision key, step, action, affected finding IDs, instructions where needed, and exact response command. @@ -374,6 +387,7 @@ Load `stuck-crewmate-recovery` after a stale wake, looping or confused pane, ans ## 9. Escalation and captain etiquette Load `i-have-adhd` before every captain-facing response; the skill owns presentation shape while this section owns outcome translation and internal-vocabulary rewriting. +When the captain says they are in mobile mode or identifies Moshi as the active surface, load `mobile-mode`; it owns the mobile presentation and review handoff delta until the captain returns to desktop or normal mode. **Talk in outcomes, not mechanics.** Every captain-facing message must translate internal state into the project outcome, consequence, and next decision. @@ -463,13 +477,15 @@ It performs guarded fast-forward updates of firstmate and registered secondmate These skills are not captain-invocable; load them only at their precise triggers. -- `bootstrap-diagnostics` - load whenever the session-start digest's bootstrap section prints an actionable diagnostic line (`MISSING:`, `MISSING_MANUAL:`, `BACKEND_INVALID:`, `NEEDS_GH_AUTH`, `TANGLE:`, `CREW_DISPATCH: invalid`, `FLEET_SYNC:`, `PR_CHECK_MIGRATION:`, `SECONDMATE_SYNC:`, `SECONDMATE_LIVENESS:`, `NUDGE_SECONDMATES:`, or `FMX:`); silence and `BOOTSTRAP_INFO:` need no load. +- `bootstrap-diagnostics` - load whenever the session-start digest's bootstrap section prints an actionable diagnostic line (`MISSING:`, `MISSING_MANUAL:`, `BACKEND_INVALID:`, `NEEDS_GH_AUTH`, `TANGLE:`, `STARTUP_MEMORY_BUDGET:`, `CREW_DISPATCH: invalid`, `FLEET_SYNC:`, `PR_CHECK_MIGRATION:`, `SECONDMATE_SYNC:`, `SECONDMATE_LIVENESS:`, `NUDGE_SECONDMATES:`, or `FMX:`); silence and `BOOTSTRAP_INFO:` need no load. - `diagnostic-reasoning` - load before scoping a reported bug and before acting on a diagnostic report. - `ask-user-authority` - load before deciding any ask-user finding, regardless of the project's `yolo` posture. +- `quota-array-dispatch` - load before choosing among a matched crew-dispatch profile array from current quota-axi output. - `harness-adapters` - load before spawning or recovering a crewmate or secondmate, handling a trust dialog, sending a harness-specific skill invocation, interrupting or exiting an agent, resuming an exited agent, or verifying a new harness adapter. - `firstmate-orca` - load before switching to Orca, spawning or supervising Orca-backed work, smoke-testing Orca backend behavior, debugging Orca task state, or reconciling Orca-backed task metadata. - `project-management` - load before adding, creating, removing, or initializing a project. -- `secrets-management` - load before project intake or initialization and before work that handles credentials or adds secret access to CI or deployment. + - `project-management` - load before adding, creating, removing, initializing, cloning, or registering a project; cloning or registering is add intake and uses the same trigger. + - `secrets-management` - load before project intake or initialization and before work that handles credentials or adds secret access to CI or deployment. - `stuck-crewmate-recovery` - load when the session-start digest reports an ordinary direct report's endpoint dead or its metadata has no window, or after a stale wake, looping pane, repeated confusion, an answered-by-brief question, an unresponsive crewmate, or a failed steer. - `secondmate-provisioning` - load before creating, seeding, validating, launching, handing backlog to, recovering, pushing inherited local material into, or retiring a secondmate home, and before editing `data/secondmates.md`. - `decision-hold-lifecycle` - load before treating an investigation or visual review as complete, before ending a visual review that exposed a decision, and when recording or routing the captain's answer. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 2f10ddf8af0..6a443e1c462 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -48,7 +48,8 @@ See the [no-mistakes quick start](https://kunchenguid.github.io/no-mistakes/star `bin/fm-lint.sh` must pass: it is the single owner of the lint definition (the shellcheck file set, config, and pinned shellcheck version), and both CI and the no-mistakes pre-push gate run it, so local and CI can never diverge. It pins one exact shellcheck version and refuses to run under any other; print it with `bin/fm-lint.sh --required-version` and install that build locally. - Changes to harness adapters (detection in `bin/fm-harness.sh`, launch and hook mechanics in `bin/fm-spawn.sh`, busy signatures in `bin/fm-watch.sh` and `bin/fm-tmux-lib.sh`, cleanup in `bin/fm-teardown.sh`, and facts in `.agents/skills/harness-adapters/SKILL.md`) must be verified empirically against the real harness, never written from documentation alone. -- Changes to runtime session backends (`bin/fm-backend.sh`, `bin/backends/`, and the scripts that dispatch through them) need empirical adapter notes in the relevant backend guide: `docs/tmux-backend.md`, `docs/herdr-backend.md`, `docs/zellij-backend.md`, `docs/orca-backend.md`, `docs/cmux-backend.md`, or `docs/codex-app-backend.md` for blocked Codex App transport work. +- Changes to runtime session backends (`bin/fm-backend.sh`, `bin/backends/`, and the scripts that dispatch through them) keep current setup and limits in the relevant backend guide and active empirical evidence in [`docs/verification/runtime-backends.md`](docs/verification/runtime-backends.md). +- [`docs/documentation-audiences.md`](docs/documentation-audiences.md) and its machine-consumed inventory own prose classification; run `bin/fm-doc-audience-check.sh` after documentation changes. - In Markdown, put each full sentence on its own line. - `README.md` stays a concise overview plus pointers: it never carries a wall of inline detail. Route detail to the most specific `docs/` file (architecture, configuration, or a backend guide) and link to it instead. @@ -70,7 +71,7 @@ That is firstmate-specific; do not commit `.no-mistakes/evidence/` here even whe Check and test the toolbelt before pushing: ```sh -for script in bin/*.sh bin/backends/*.sh; do bash -n "$script"; done # syntax-check the toolbelt +while IFS= read -r script; do /bin/bash -n "$script" || exit; done < <(bin/fm-lint.sh --list-files) # syntax-check the canonical shell surface bin/fm-lint.sh # lint the toolbelt and behavior tests; the single owner CI and the no-mistakes gate both run bin/fm-test-run.sh tests/.test.sh # one script (primary local focus path, timed) bin/fm-test-run.sh --family pure-contract-unit # ordinary family-scoped local path (serial, timed) @@ -79,7 +80,7 @@ bin/fm-test-run.sh --proven-isolated --jobs 4 # explicit local parallel of the bin/fm-test-run.sh --lane portable-serial # portable serial remainder (watcher/AFK/tmux/stateful) bin/fm-test-run.sh --check-coverage # prove portable shards + serial + Herdr equal the full inventory bin/fm-test-run.sh --all # deliberate complete regression (optional local full walk; not no-mistakes Test) -bin/fm-test-isolation-proof.sh --list # proven parallel candidate set (Phase 2 owner) +bin/fm-test-isolation-proof.sh --list # proven parallel candidate set (concurrent isolation proof owner) bin/fm-test-isolation-proof.sh --jobs 4 --json /tmp/fm-isolation-proof.json # re-run concurrent isolation proof only [ "$(readlink CLAUDE.md)" = "AGENTS.md" ] [ "$(readlink .claude/skills)" = "../.agents/skills" ] @@ -88,15 +89,15 @@ tmp=$(mktemp -d) && printf 'done: smoke\n' > "$tmp/smoke.status" && FM_STATE_OVE `bin/fm-test-run.sh` is the single owner of behavior-suite selection, portable CI lane composition, optional local `--jobs` for the proven-isolated set only, per-script timing markers, family totals, the coverage guard, and the optional JSON timing artifact. Its header and `--help` own the flags, family labels, lanes, and changed-file map; this section only documents the entry points. -`bin/fm-test-isolation-proof.sh` remains the single owner of the Phase 2 concurrent isolation proof and the exact proven candidate set; see `docs/fm-test-isolation-proof.md`. +`bin/fm-test-isolation-proof.sh` remains the single owner of the concurrent isolation proof and the exact proven candidate set; see `docs/fm-test-isolation-proof.md`. Portable shard balance evidence lives in `docs/fm-test-portable-shards.md`. Local no-mistakes Test stays intent-targeted and must not wire `commands.test` to `--all` or a `tests/*.test.sh` walk. Family selection is the ordinary local path; `--all` is deliberate full regression only. -CI owns broad regression across required portable parallel shards, the portable serial lane, the Herdr lane, lint, invariants, the coverage guard, and macOS snapshot compatibility in [`.github/workflows/ci.yml`](.github/workflows/ci.yml). +CI owns broad regression across required portable parallel shards, the portable serial lane, the Herdr lane, lint, invariants, the coverage guard, and stock macOS Bash compatibility in [`.github/workflows/ci.yml`](.github/workflows/ci.yml). Use `bin/fm-test-run.sh --help` for lane names, `--jobs` rules, and required gate-skip flags when reproducing a lane locally. Discover tests by listing `tests/*.test.sh`: each is a self-contained bash script named `.test.sh`, and its header comment describes what it covers, so pass one to `bin/fm-test-run.sh` to focus on a subject with canonical timing output. Tests that need a real optional backend or an explicit opt-in (real herdr/zellij/cmux smoke tests, the live Pi regression) skip themselves and print the tool or environment gate needed to enable them, so the portable suite remains safe on machines without those tools. -The [Herdr backend guide](docs/herdr-backend.md) owns the lane's safety and isolation rationale, including why live harness credential tests remain opt-in. +The [Herdr backend guide](docs/herdr-backend.md#destructive-lab-safety) owns the lane's isolation boundary, while [runtime backend verification](docs/verification/runtime-backends.md#herdr) owns active empirical evidence; live harness credential tests remain opt-in. ## Questions diff --git a/README.md b/README.md index 7eb82be38c3..526109c792a 100644 --- a/README.md +++ b/README.md @@ -49,7 +49,7 @@ Launching a supported harness inside it instantiates your first mate - and makes - **Optional secondmates** - opt in to persistent second mates that run from isolated firstmate homes with their own `FM_HOME`, state, projects, and session lock, supervising project clones or a project-less firstmate-repo domain, kept on the primary firstmate version by guarded local fast-forwards and checked for live agent processes at session start. - **Event-driven, zero-token supervision** - a bash watcher sleeps on the fleet and wakes the first mate only when something needs you; verified primary harnesses also get a turn-end backstop that blocks or follows up on a blind stop when work is under way and supervision is not live. - **Optional X mode** - opt in with one local `.env` token so firstmate can answer your public `@myfirstmate` mentions, act on normal reversible mention requests through the same lifecycle as chat requests, acknowledge spawned work, and post up to three public-safe completion follow-ups within seven days for genuine milestones and the final outcome without changing non-X behavior; dry-run preview records would-be replies and dismissals locally before go-live. -- **Guarded by construction** - the first mate is read-only over your projects except for the guarded paths authorized by [hard rule 1](AGENTS.md#1-identity-and-prime-directives), with fleet sync's safe branch pruning remaining part of the fleet-sync exception; crewmates make every project change behind the configured merge authority. +- **Strict project boundary** - the first mate is read-only over your projects except for the narrow guarded and captain-approved operations authorized by [hard rule 1](AGENTS.md#1-identity-and-prime-directives), including fleet sync's guarded safe branch pruning; crewmates make every other project change behind the configured merge authority. - **Doppler by default** - the conditional [secrets-management policy](.agents/skills/secrets-management/SKILL.md) prefers secretless provider identity, otherwise scopes Doppler by project and environment, and validates declarations and rollout data through `bin/fm-secrets-check.sh`. - **Restart-proof** - all state lives on disk and in the active session backend (tmux by hard default, herdr or cmux when selected or auto-detected, zellij/orca when explicitly selected); kill the session anytime and the next one reconciles, including confirmed-dead secondmate agents, and carries on. @@ -59,16 +59,17 @@ Full detail on every feature lives in [docs/architecture.md](docs/architecture.m ### Requirements -- A verified agent harness: Claude Code, Grok, Pi, Codex, or OpenCode. +- A verified primary agent harness: Claude Code, Grok, Pi, `pi-signed`, Codex, or OpenCode. - Git and the GitHub CLI, authenticated through `gh auth login`. -- tmux, for the reference session backend. +- The CLI and dependencies for your selected runtime backend; tmux is the reference default. -The first mate detects and offers to install everything else. +The first mate detects and offers to install supported missing tools after you approve. +Backend-specific setup is linked in [Documentation](#documentation). ### Recommended harnesses -**Claude Code, Grok, and Pi are equal co-primary recommendations** for running the primary firstmate session. -Claude Code and Grok use background-notify wake cycles; Pi uses its tracked primary watcher extension. +**Claude Code, Grok, and Pi are equal co-primary recommendations** for running the primary firstmate session, with `pi-signed` supported as Pi's distinct signed-wrapper identity. +Claude Code uses a tracked Stop hook for tokenless watcher re-arm and rewake, Grok uses background-notify wake cycles, and Pi uses its tracked primary watcher extension. All three have verified turn-end guard paths when launched with their documented setup. Pick whichever one matches your subscription and workflow. @@ -100,20 +101,16 @@ grok --trust ```sh pi +# or, when the signed wrapper is installed +FM_PI_HARNESS=pi-signed pi-signed ``` For Grok, `--trust` is needed once per clone so project hooks and the turn-end guard load; `/hooks-trust` inside Grok works too. For Pi, approve the project trust prompt once per clone on first launch so the tracked `.pi/extensions/*.ts` files auto-load. -`/calm` is a conversation-focused transcript toggle whose last choice persists for the effective Firstmate home across Pi session starts and resumes. -While active, it keeps Pi's built-in `Working...` activity visible and uses Pi's presentation surface to hide collapsed thinking blocks, canonically classified Firstmate operational user rows, all seven built-in tool shells, the Firstmate watcher tool shell, and compatible presentation entries stored by earlier Calm versions. -Calm adds no persistent status row, and controllable hidden rows are removed without reserving vertical space. -Canonically classified Firstmate operational input remains an ordinary user-role message with its exact origin, ordering, model authority, and session persistence unchanged while its Calm presentation occupies zero rows. -The session-start nudge remains on its existing non-displayed custom-message path. -Toggling off restores ordinary rendering, and `Ctrl+O` expansion behavior stays unchanged. -Tool execution, input delivery, model context, session storage, diagnostics, and `/export` and `/share` operation remain unchanged. -Exports and shares remain complete session artifacts, including current operational user messages and any legacy hidden custom messages retained in serialized session data and Pi 0.81.1's sidebar tree. -Pi 0.81.1 still exposes no global transcript filter, so expanded reasoning, built-in tool images, user-bash rows, skill and summary rows, status notices, and arbitrary custom-tool or extension rows remain supported-API boundaries. -The version-scoped feasibility evidence and complete render taxonomy are recorded in [docs/calm-mode-feasibility.md](docs/calm-mode-feasibility.md). +Pi's `/calm` toggle hides supported transcript chrome, including canonically classified Firstmate operational user rows, while retaining native working activity and all model context and session data. +The hidden operational inputs remain ordinary user-role messages with unchanged delivery, ordering, authority, persistence, and exports. +The preference persists for the effective Firstmate home, and toggling it off restores ordinary rendering. +[Calm's current behavior and supported limits](docs/calm.md) are separate from its [version-scoped maintainer evidence](docs/calm-mode-feasibility.md). ### Talk to it @@ -121,8 +118,7 @@ The version-scoped feasibility evidence and complete render taxonomy are recorde > ahoy! look at my github project xyz, then fix the flaky login test and add dark mode # firstmate checks its toolchain (asking your consent before installing anything), -# clones the project under projects/, and spawns two crewmates in the active backend -# fm-fix-login-k3 and fm-dark-mode-p7. +# clones the project under projects/ and spawns two isolated workers in the active backend. # Minutes later: PR ready for review, captain: https://github.com/you/xyz/pull/42 @@ -176,10 +172,17 @@ Claude and grok use the slash form shown here; codex uses the same names with `$ | ------------------ | -------------------------------------------------------------------------------------------------------------------------------------------- | | `/afk` | Enter away-mode supervision: the sub-supervisor self-handles routine notifications in bash, escalates captain-relevant events and bounded declared-external-wait rechecks as batched digests, and actively alerts if delivery gets stuck while you step away | | `/ahoy` | Recap visible session events since the prior real captain message plus visibly unanswered captain decisions, falling back to Bearings when invoked as the session's first real captain message | -| `/bearings` | Generate a standalone current-status report from bounded local fleet and registered-secondmate state, with live PR enrichment only when requested, written to a dated file in `data/` and surfaced concisely in chat; read-mostly, mutates no task state | +| `/bearings` | Generate a concise four-section chat digest from bounded local fleet and registered-secondmate state; use `/bearings file` to also replace today's dated report in `data/`, and add `include PRs` when live PR enrichment is wanted | | `/updatefirstmate` | Self-update the running firstmate and its secondmates to the latest from origin with fast-forward-only pulls, then re-read instructions and nudge secondmates | | `/stow` | Sweep the session for uncaptured durable knowledge, route each finding to its disk home per AGENTS.md, file undone next steps to the backlog, and report what is now safe to reset | +Bearings invocation examples: + +- `/bearings` returns the fresh four-section digest in chat only. +- `/bearings include PRs` keeps chat-only mode and opts into live PR enrichment. +- `/bearings file` replaces today's `data/status-report-.md` from scratch and links it from the four-section chat digest. +- `/bearings file include PRs` combines the dated report with live PR enrichment. + Agent-only reference skills live under `.agents/skills/` and are loaded by firstmate at the trigger points named in [`AGENTS.md`](AGENTS.md). ### Two-tier skill layout @@ -194,8 +197,10 @@ Firstmate's skills live in two separate places with different audiences: ## Documentation -- [docs/architecture.md](docs/architecture.md) - how the crew, supervision, worktrees, secondmates, and project modes work. +- [docs/architecture.md](docs/architecture.md) - maintainer architecture for the crew, supervision, worktrees, secondmates, and project modes. - [docs/configuration.md](docs/configuration.md) - environment variables, `FM_HOME`, runtime backend selection, optional X mode, the files you set, and harness support. +- [docs/calm.md](docs/calm.md) - current Pi `/calm` behavior and supported presentation limits. +- [docs/moshi-mobile-review.md](docs/moshi-mobile-review.md) - host-local Moshi Pro Preview, Diff, Chat View, and private mobile-review fallbacks. - [docs/wedge-alarm.md](docs/wedge-alarm.md) - configure the active alert for an away-mode escalation delivery that gets stuck. - [docs/tmux-backend.md](docs/tmux-backend.md) - setup guide for the tmux reference backend: prerequisites, attaching, and watching crew windows. - [docs/herdr-backend.md](docs/herdr-backend.md) - setup guide for the experimental herdr backend, plus its verification notes and known gaps. @@ -206,8 +211,11 @@ Firstmate's skills live in two separate places with different audiences: - [docs/gitlab-merge-watch.md](docs/gitlab-merge-watch.md) - how the merge watch follows a GitLab merge request on any instance, and the evidence behind it. - [docs/promotion-ladder.md](docs/promotion-ladder.md) - the promotion contract for projects with deployable applications. - [docs/turnend-guard.md](docs/turnend-guard.md) - the primary session's structural "no turn ends blind" backstop: verified per-harness hook mechanisms, scoping, loop safety, and fail-open tradeoffs. -- [docs/supervision-protocols/](docs/supervision-protocols/) - rendered primary-harness watcher protocols for Claude, Codex, OpenCode, Pi, Grok, and unknown harness fallback. +- [docs/verification/runtime-backends.md](docs/verification/runtime-backends.md) - active maintainer verification for runtime backend guarantees. +- [docs/verification/supervision.md](docs/verification/supervision.md) - active maintainer verification for session-start, guard, continuity, and wedge integrations. +- [docs/supervision-protocols/](docs/supervision-protocols/) - rendered primary-harness watcher protocols for Claude, Codex, OpenCode, Pi and `pi-signed`, Grok, and unknown harness fallback. - [docs/scripts.md](docs/scripts.md) - the `bin/` toolbelt reference. +- [docs/documentation-audiences.md](docs/documentation-audiences.md) - documentation audiences and the machine-checked placement boundary. - [`AGENTS.md`](AGENTS.md) - the distro's always-loaded operating contract and routing index for conditional procedures. - [CONTRIBUTING.md](CONTRIBUTING.md) - how to contribute, including the dev/test commands. diff --git a/bin/backends/cmux.sh b/bin/backends/cmux.sh index 69dc0b53bde..12dc7629eb6 100644 --- a/bin/backends/cmux.sh +++ b/bin/backends/cmux.sh @@ -581,8 +581,8 @@ fm_backend_cmux_composer_state() { # [expected-label] -> empty|pending # has since moved its own confirmation to a native agent-state read instead # (docs/herdr-backend.md "Native agent-state submit confirmation"); cmux has # no analogous native primitive, so this composer-row approach remains -# cmux's own confirmation strategy. Echoes empty|pending|unknown|send-failed, the -# SAME vocabulary every existing backend already speaks. +# cmux's own confirmation strategy. Echoes empty|pending|unknown|send-failed, a +# subset of the proof-carrying submit vocabulary. fm_backend_cmux_send_text_submit() { # [expected-label] local target=$1 text=$2 retries=$3 sleep_s=$4 settle=$5 expected_label=${6:-} i=0 state fm_backend_cmux_parse_target "$target" || { printf 'unknown'; return 0; } diff --git a/bin/backends/herdr.sh b/bin/backends/herdr.sh index 432311d6710..237d2348c5a 100644 --- a/bin/backends/herdr.sh +++ b/bin/backends/herdr.sh @@ -1831,7 +1831,15 @@ FM_BACKEND_HERDR_IDLE_RE=${FM_BACKEND_HERDR_IDLE_RE:-'^Type a message\.\.\.$'} # Known bare (unbordered) prompt glyphs a composer row may start with: ❯ # (claude) and › (codex) only. Generic shell-style glyphs > $ % # are still # recognized after a bordered composer row has already been structurally found. -FM_BACKEND_HERDR_BARE_PROMPT_RE=${FM_BACKEND_HERDR_BARE_PROMPT_RE:-'^[❯›]'} +# Deliberately an alternation, not a `[...]` bracket expression: under a C/POSIX +# locale (LC_CTYPE=C, the fleet default), grep's bracket expressions match +# individual BYTES rather than whole multibyte characters, so `[❯›]` silently +# decomposes into the shared leading UTF-8 byte (0xE2) and spuriously matches +# ANY multibyte glyph in that range - including box-drawing corners like ╰, +# misclassifying a bordered composer's bottom border row as the bare shape. +# An alternation's branches are matched as whole literal byte sequences and +# stay correct regardless of locale. +FM_BACKEND_HERDR_BARE_PROMPT_RE=${FM_BACKEND_HERDR_BARE_PROMPT_RE:-'^(❯|›)'} # Pi allows a multi-line composer between its horizontal separators. Bound the # structural candidate so two unrelated transcript rules with an arbitrarily # large region between them can never be promoted into a composer. @@ -1995,7 +2003,7 @@ EOF fi # Delegate the empty/pending/unknown decision to the shared owner. The bare # shape only ever starts with an AGENT glyph (FM_BACKEND_HERDR_BARE_PROMPT_RE - # is '^[❯›]'), so a bare shell prompt never reaches here - it stays 'unknown' + # is '^(❯|›)'), so a bare shell prompt never reaches here - it stays 'unknown' # via the no-composer-row path above, exactly as before. fm_composer_classify_content "$bordered" "$stripped" "$FM_BACKEND_HERDR_IDLE_RE" } @@ -2057,10 +2065,10 @@ EOF # re-invokes this function from scratch with the same text after seeing # an error, which is a human/escalation decision, not an automatic # retry). -# Echoes empty|pending|unknown|send-failed, the SAME vocabulary fm-send.sh -# already branches on for tmux ("empty" means "confirmed submitted" for every -# backend; how each backend confirms it is an internal decision - herdr's is -# no longer literally "the composer read empty"). +# Echoes empty|pending|unknown|send-failed, a subset of the proof-carrying +# submit vocabulary. Empty means confirmed submitted for every backend; how +# each backend confirms it is an internal decision, and herdr's is no longer +# literally "the composer read empty". fm_backend_herdr_send_text_submit() { # local target=$1 text=$2 retries=$3 sleep_s=$4 settle=$5 i=0 verdict baseline confirm_sleep fm_backend_herdr_parse_target "$target" || { printf 'unknown'; return 0; } diff --git a/bin/backends/tmux.sh b/bin/backends/tmux.sh index ab93b88556a..f8da21bf0de 100644 --- a/bin/backends/tmux.sh +++ b/bin/backends/tmux.sh @@ -117,10 +117,22 @@ fm_backend_tmux_send_literal() { # tmux send-keys -t "$1" -l "$2" } -# fm_backend_tmux_kill: remove the task's window, best-effort. Mirrors -# fm-teardown.sh's `tmux kill-window -t "$T" 2>/dev/null || true`. +# fm_backend_tmux_kill: remove one explicitly named task window, best-effort. +# Empty, omitted, and malformed targets return nonzero before invoking tmux so +# tmux can never interpret an empty target as the caller's current window. fm_backend_tmux_kill() { # - tmux kill-window -t "$1" 2>/dev/null || true + local target=${1:-} session window + case "$target" in + *:*) + session=${target%%:*} + window=${target#*:} + ;; + *) return 1 ;; + esac + case "$session:$window" in + :*|*:|*:*:*) return 1 ;; + esac + tmux kill-window -t "=$session:=$window" 2>/dev/null || true } # fm_backend_tmux_current_command: 's live foreground process name - @@ -181,7 +193,7 @@ fm_backend_tmux_agent_state() { # } comm=${comm#-} case "$comm" in - *claude*|*codex*|*opencode*|*grok*) printf 'alive' ;; + *claude*|*codex*|*opencode*|*grok*|*kimi*|pi|pi-signed|pi-launcher|Pi) printf 'alive' ;; zsh|bash|sh|dash|ash|ksh|mksh|tcsh|csh|fish) printf 'dead' ;; '') printf 'unreadable' ;; *) printf 'ambiguous' ;; diff --git a/bin/backends/zellij.sh b/bin/backends/zellij.sh index 162ea5477b6..20d53a3c2de 100644 --- a/bin/backends/zellij.sh +++ b/bin/backends/zellij.sh @@ -501,8 +501,8 @@ fm_backend_zellij_capture() { # [expected-label] # also the load-bearing defense against the # unconditional-exit-0 CLI quirk documented in the file header: a truly dead # target never shows a change, so it correctly reports pending/unknown rather -# than a false "sent". Echoes empty|pending|unknown|send-failed, the SAME -# vocabulary fm-send.sh already branches on for tmux and herdr. +# than a false "sent". Echoes empty|pending|unknown|send-failed, a subset of the +# proof-carrying submit vocabulary. fm_backend_zellij_send_text_submit() { # [expected-label] local target=$1 text=$2 retries=$3 sleep_s=$4 settle=$5 expected_label=${6:-} typed after i=0 fm_backend_zellij_send_literal "$target" "$text" "$expected_label" || { printf 'send-failed'; return 0; } diff --git a/bin/fm-afk-launch.sh b/bin/fm-afk-launch.sh index 57b7f6590db..3bc1a1cbac4 100755 --- a/bin/fm-afk-launch.sh +++ b/bin/fm-afk-launch.sh @@ -48,6 +48,28 @@ set -u FM_AFK_LAUNCH_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$FM_AFK_LAUNCH_DIR/.." && pwd)}" FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" +case "$FM_HOME" in + /*) ;; + *) + FM_AFK_LAUNCH_HOME_INPUT=$FM_HOME + FM_HOME=$(CDPATH='' cd -- "$FM_AFK_LAUNCH_HOME_INPUT" 2>/dev/null && pwd -P) || { + echo "error: FM_HOME directory cannot be resolved: $FM_AFK_LAUNCH_HOME_INPUT" >&2 + exit 1 + } + ;; +esac +if [ -n "${FM_STATE_OVERRIDE:-}" ]; then + case "$FM_STATE_OVERRIDE" in + /*) ;; + *) + FM_AFK_LAUNCH_STATE_INPUT=$FM_STATE_OVERRIDE + FM_STATE_OVERRIDE=$(CDPATH='' cd -- "$FM_AFK_LAUNCH_STATE_INPUT" 2>/dev/null && pwd -P) || { + echo "error: FM_STATE_OVERRIDE directory cannot be resolved: $FM_AFK_LAUNCH_STATE_INPUT" >&2 + exit 1 + } + ;; + esac +fi FM_AFK_LAUNCH_STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" FM_AFK_LAUNCH_RECORD="$FM_AFK_LAUNCH_STATE/.afk-daemon-terminal" FM_AFK_LAUNCH_LOCK="$FM_AFK_LAUNCH_STATE/.afk-launch.lock" diff --git a/bin/fm-backend.sh b/bin/fm-backend.sh index 7c2790bd43d..e505b99f757 100644 --- a/bin/fm-backend.sh +++ b/bin/fm-backend.sh @@ -360,6 +360,177 @@ fm_backend_target_of_meta() { # [ -n "$window" ] && printf '%s' "$window" } +# fm_backend_validate_task_endpoint: validate a task cleanup record entirely +# from its durable metadata before any runtime command or cleanup mutation. +# The validation binds the exact task id, selected backend, target, project, +# and worktree. New non-tmux records carry endpoint_task_id because their +# opaque runtime ids do not encode the task label. Legacy tmux records remain +# valid only when their window name itself is exactly fm-. +# On success, sets FM_BACKEND_VALIDATED_BACKEND and +# FM_BACKEND_VALIDATED_TARGET. On failure, prints one refusal and returns 1. +fm_backend_meta_exact_value() { # + local meta=$1 key=$2 count value + count=$(grep -c "^$key=" "$meta" 2>/dev/null || true) + [ "$count" -eq 1 ] || return 1 + value=$(grep "^$key=" "$meta" | cut -d= -f2-) + [ -n "$value" ] || return 1 + printf '%s' "$value" +} + +fm_backend_endpoint_atom_valid() { # + case "$1" in + ''|*[!A-Za-z0-9._@%+-]*) return 1 ;; + esac +} + +fm_backend_validate_task_endpoint() { # + local meta=$1 id=$2 backend_count backend window worktree project binding_count binding + local session pane recorded_session workspace tab terminal worktree_id surface + FM_BACKEND_VALIDATED_BACKEND= + FM_BACKEND_VALIDATED_TARGET= + [ -f "$meta" ] && [ ! -L "$meta" ] || { + echo "REFUSED: task $id has no regular endpoint metadata at $meta; preserving task state." >&2 + return 1 + } + case "$id" in ''|*[!A-Za-z0-9._-]*) + echo "REFUSED: task endpoint identity has an invalid task id; preserving task state." >&2 + return 1 + esac + window=$(fm_backend_meta_exact_value "$meta" window) || { + echo "REFUSED: task $id has a missing, empty, or ambiguous window endpoint; preserving task state." >&2 + return 1 + } + worktree=$(fm_backend_meta_exact_value "$meta" worktree) || { + echo "REFUSED: task $id has a missing, empty, or ambiguous worktree identity; preserving task state." >&2 + return 1 + } + project=$(fm_backend_meta_exact_value "$meta" project) || { + echo "REFUSED: task $id has a missing, empty, or ambiguous project identity; preserving task state." >&2 + return 1 + } + case "$worktree$project$window" in *$'\n'*|*$'\r'*|*$'\t'*) + echo "REFUSED: task $id has malformed endpoint metadata; preserving task state." >&2 + return 1 + esac + backend_count=$(grep -c '^backend=' "$meta" 2>/dev/null || true) + case "$backend_count" in + 0) backend=tmux ;; + 1) backend=$(fm_backend_meta_exact_value "$meta" backend) || backend= ;; + *) backend= ;; + esac + if [ -z "$backend" ] || ! fm_backend_is_known "$backend"; then + echo "REFUSED: task $id has a missing, ambiguous, or unknown backend identity; preserving task state." >&2 + return 1 + fi + binding_count=$(grep -c '^endpoint_task_id=' "$meta" 2>/dev/null || true) + case "$binding_count" in + 0) binding= ;; + 1) + binding=$(fm_backend_meta_exact_value "$meta" endpoint_task_id) || { + echo "REFUSED: task $id has an empty endpoint task binding; preserving task state." >&2 + return 1 + } + ;; + *) + echo "REFUSED: task $id has an ambiguous endpoint task binding; preserving task state." >&2 + return 1 + ;; + esac + if [ -n "$binding" ] && [ "$binding" != "$id" ]; then + echo "REFUSED: endpoint metadata belongs to task $binding, not $id; preserving task state." >&2 + return 1 + fi + + case "$backend" in + tmux) + session=${window%%:*} + pane=${window#*:} + if [ "$pane" = "$window" ] || [ "$pane" != "fm-$id" ] \ + || [ -z "$session" ]; then + echo "REFUSED: tmux endpoint '$window' is malformed or does not belong to task $id; preserving task state." >&2 + return 1 + fi + ;; + herdr) + [ "$binding" = "$id" ] || { + echo "REFUSED: legacy Herdr endpoint metadata for task $id lacks an exact task binding; preserving task state." >&2 + return 1 + } + recorded_session=$(fm_backend_meta_exact_value "$meta" herdr_session) || recorded_session= + workspace=$(fm_backend_meta_exact_value "$meta" herdr_workspace_id) || workspace= + tab=$(fm_backend_meta_exact_value "$meta" herdr_tab_id) || tab= + pane=$(fm_backend_meta_exact_value "$meta" herdr_pane_id) || pane= + if [ -z "$recorded_session" ] || [ -z "$workspace" ] || [ -z "$tab" ] || [ -z "$pane" ] \ + || [ "$window" != "$recorded_session:$pane" ] \ + || ! fm_backend_endpoint_atom_valid "$recorded_session" \ + || ! fm_backend_endpoint_atom_valid "$workspace" \ + || ! fm_backend_endpoint_atom_valid "${tab//:/_}" \ + || ! fm_backend_endpoint_atom_valid "${pane//:/_}"; then + echo "REFUSED: Herdr endpoint metadata for task $id is malformed or inconsistent; preserving task state." >&2 + return 1 + fi + ;; + zellij) + [ "$binding" = "$id" ] || { + echo "REFUSED: legacy Zellij endpoint metadata for task $id lacks an exact task binding; preserving task state." >&2 + return 1 + } + recorded_session=$(fm_backend_meta_exact_value "$meta" zellij_session) || recorded_session= + tab=$(fm_backend_meta_exact_value "$meta" zellij_tab_id) || tab= + pane=$(fm_backend_meta_exact_value "$meta" zellij_pane_id) || pane= + case "$tab:$pane" in *[!0-9:]*) tab= ;; esac + if [ -z "$recorded_session" ] || [ -z "$tab" ] || [ -z "$pane" ] \ + || [ "$window" != "$recorded_session:$pane" ] \ + || ! fm_backend_endpoint_atom_valid "$recorded_session"; then + echo "REFUSED: Zellij endpoint metadata for task $id is malformed or inconsistent; preserving task state." >&2 + return 1 + fi + ;; + orca) + [ "$binding" = "$id" ] || { + echo "REFUSED: legacy Orca endpoint metadata for task $id lacks an exact task binding; preserving task state." >&2 + return 1 + } + terminal=$(fm_backend_meta_exact_value "$meta" terminal) || terminal= + worktree_id=$(fm_backend_meta_exact_value "$meta" orca_worktree_id) || worktree_id= + [ -n "$terminal" ] || { + echo "REFUSED: missing terminal in $meta; cannot close Orca endpoint; preserving task state." >&2 + return 1 + } + [ -n "$worktree_id" ] || { + echo "REFUSED: missing orca_worktree_id in $meta; cannot remove Orca worktree; preserving task state." >&2 + return 1 + } + if [ "$window" != "fm-$id" ] \ + || ! fm_backend_endpoint_atom_valid "$terminal" \ + || ! fm_backend_endpoint_atom_valid "$worktree_id"; then + echo "REFUSED: Orca endpoint metadata for task $id is malformed or inconsistent; preserving task state." >&2 + return 1 + fi + window=$terminal + ;; + cmux) + [ "$binding" = "$id" ] || { + echo "REFUSED: legacy cmux endpoint metadata for task $id lacks an exact task binding; preserving task state." >&2 + return 1 + } + workspace=$(fm_backend_meta_exact_value "$meta" cmux_workspace_id) || workspace= + surface=$(fm_backend_meta_exact_value "$meta" cmux_surface_id) || surface= + if [ -z "$workspace" ] || [ -z "$surface" ] || [ "$window" != "$workspace:$surface" ] \ + || ! fm_backend_endpoint_atom_valid "$workspace" \ + || ! fm_backend_endpoint_atom_valid "$surface"; then + echo "REFUSED: cmux endpoint metadata for task $id is malformed or inconsistent; preserving task state." >&2 + return 1 + fi + ;; + esac + # shellcheck disable=SC2034 # Output globals are consumed by sourcing callers. + FM_BACKEND_VALIDATED_BACKEND=$backend + # shellcheck disable=SC2034 # Output globals are consumed by sourcing callers. + FM_BACKEND_VALIDATED_TARGET=$window + return 0 +} + fm_backend_meta_for_window() { # local target=$1 state=$2 meta window terminal for meta in "$state"/*.meta; do @@ -551,8 +722,8 @@ fm_backend_send_key() { # [expected-label] } # fm_backend_send_text_submit: type text once, then submit and verify, -# retrying only the submission (never retyping). Echoes the verdict -# (empty|pending|unknown|send-failed for submit-verifying adapters). +# retrying only the submission (never retyping). Echoes the backend's +# proof-carrying verdict; callers require exact empty for confirmed delivery. fm_backend_send_text_submit() { # [expected-label] local backend=$1 shift @@ -573,6 +744,7 @@ fm_backend_send_text_submit() { # local backend=$1 shift + [ -n "${1:-}" ] || { echo "error: refusing empty backend kill target" >&2; return 1; } fm_backend_source "$backend" || return 1 case "$backend" in tmux) fm_backend_tmux_kill "$@" ;; @@ -608,9 +780,9 @@ fm_backend_worktree_path() { # # native agent-state (herdr-addendum "busy state" row - the first backend # where this gets real semantics beyond pane-regex). Backends with no such # primitive (tmux) report unknown. Callers own the fallback policy: fm-watch.sh -# uses unknown as the cue for its pane-hash + FM_BUSY_REGEX detection, while -# fm-crew-state.sh also corroborates native idle verdicts before treating a -# no-run crew as not busy. +# uses unknown as the cue for harness-scoped pane-tail detection, while +# fm-crew-state.sh also corroborates native idle verdicts with the recorded +# harness's signature before treating a no-run crew as not busy. fm_backend_busy_state() { # local backend=$1 shift @@ -622,8 +794,8 @@ fm_backend_busy_state() { # } # fm_backend_composer_state: classify the composer/input row of as -# empty|pending|unknown for callers that need a pre-submit pending-input guard -# or an adapter's conservative submit fallback. It is exposed generically so a +# empty|pending|pending-unproven|unknown for callers that need a pre-submit +# input guard or an adapter's conservative submit fallback. It is exposed so a # caller other than the send path (the away-mode daemon's supervisor-pane # pending-input guard, bin/fm-supervise-daemon.sh) can ask the same question # without duplicating per-backend composer-reading logic. tmux and herdr both @@ -633,7 +805,7 @@ fm_backend_busy_state() { # # submit path uses an internal content-diff approach with no separately named # classifier, so it reports unknown here - callers fall back to their own # policy, exactly as an unknown fm_backend_busy_state already does. -fm_backend_composer_state() { # -> empty|pending|unknown +fm_backend_composer_state() { # -> empty|pending|pending-unproven|unknown local backend=$1 shift fm_backend_source "$backend" || { printf 'unknown'; return 0; } diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index 0202ca40932..16102adfa45 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -8,6 +8,7 @@ # Lines: "MISSING: (install: )", # "MISSING_MANUAL: (instructions: )", "NEEDS_GH_AUTH", # "BACKEND_INVALID: (known: )", +# "STARTUP_MEMORY_BUDGET: invalid config/startup-memory-budget - ", # "CREW_DISPATCH: invalid config/crew-dispatch.json - ", # "FLEET_SYNC: : skipped|recovered|STUCK: ", # "PR_CHECK_MIGRATION: ", @@ -52,10 +53,14 @@ # lavish-axi). tasks-axi is also version and feature gated (0.1.1+ # with update --archive-body and mv [...]); an installed but # incompatible build reports MISSING like no-mistakes. A compatible -# tasks-axi default backend is silent. quota-axi is required because -# every crew-dispatch profile array calls it automatically; -# fm-dispatch-select.sh still uses OS-backed random selection across -# valid candidates when quota data is unavailable. +# tasks-axi default backend is silent. quota-axi is required for the +# agent-owned dispatch-profile array procedure in AGENTS.md section 4 +# and .agents/skills/quota-array-dispatch/SKILL.md. +# On a primary home, the locked mutable path materializes the visible +# default config/startup-memory-budget=7500 when absent. It never +# guesses at malformed or unsafe existing files, and secondmate homes +# await the primary-authoritative inherited value instead of creating +# their own. # X mode is OPTIONAL and inert unless FM_HOME/.env has a non-empty # FMX_PAIRING_TOKEN. When opted in, bootstrap requires curl+jq, writes # the relay poll shim and 30s cadence config, and prints an FMX line. @@ -98,6 +103,8 @@ DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" . "$SCRIPT_DIR/fm-ff-lib.sh" # shellcheck source=bin/fm-config-inherit-lib.sh disable=SC1091 . "$SCRIPT_DIR/fm-config-inherit-lib.sh" +# shellcheck source=bin/fm-startup-memory-budget-lib.sh disable=SC1091 +. "$SCRIPT_DIR/fm-startup-memory-budget-lib.sh" # shellcheck source=bin/fm-x-lib.sh disable=SC1091 . "$SCRIPT_DIR/fm-x-lib.sh" # shellcheck source=bin/fm-backend.sh disable=SC1091 @@ -438,7 +445,7 @@ secondmate_liveness_sweep() { [ -n "$target" ] || target="$window" agent_state=$(fm_backend_agent_state "$backend" "$target" 2>/dev/null) || agent_state=unreadable case "$harness" in - claude|codex|opencode|pi|grok) ;; + claude|codex|opencode|pi|pi-signed|grok|kimi) ;; *) case "$agent_state" in dead|missing) agent_state=unverified-harness ;; esac ;; @@ -621,7 +628,7 @@ x_mode_remove_artifact() { # applying a cadence transition to a running watcher is the caller's job via # the emitted harness-aware supervision repair instruction. x_mode_setup() { - local env_file token shim cadence shim_body cadence_body tool missing + local env_file token shim cadence shim_body cadence_body tool missing shim_home env_file="$FM_HOME/.env" shim="$STATE/x-watch.check.sh" cadence="$CONFIG/x-mode.env" @@ -684,9 +691,16 @@ x_mode_setup() { mkdir -p "$STATE" "$CONFIG" 2>/dev/null || { fmx_arm_failed; return 0; } - shim_body=$(fmx_poll_shim_content "$FM_HOME" "$FM_ROOT") + case "$FM_HOME" in + /*) shim_home=$FM_HOME ;; + *) + shim_home=$(CDPATH='' cd -- "$FM_HOME" 2>/dev/null && pwd -P) \ + || { fmx_arm_failed; return 0; } + ;; + esac + shim_body=$(fmx_poll_shim_content "$shim_home" "$FM_ROOT") x_mode_write_if_changed "$shim" "$shim_body" 700 || { fmx_arm_failed; return 0; } - fmx_poll_shim_valid "$shim" "$FM_HOME" "$FM_ROOT" \ + fmx_poll_shim_valid "$shim" "$shim_home" "$FM_ROOT" \ || { fmx_arm_failed; return 0; } cadence_body=$(cat <<'EOF' @@ -715,15 +729,15 @@ crew_dispatch_validate() { return 0 fi err=$(jq -r ' - def verified($h): ["claude","codex","opencode","pi","grok"] | index($h); + def verified($h): ["claude","codex","opencode","pi","pi-signed","grok","kimi"] | index($h); def effort_ok($h; $e): if $e == null then true elif ($e | type) != "string" then false elif $h == "claude" then (["low","medium","high","xhigh","max"] | index($e)) elif $h == "codex" then (["low","medium","high","xhigh"] | index($e)) elif $h == "grok" then (["low","medium","high"] | index($e)) - elif $h == "pi" then (["low","medium","high","xhigh","max"] | index($e)) - elif $h == "opencode" then false + elif $h == "pi" or $h == "pi-signed" then (["low","medium","high","xhigh","max"] | index($e)) + elif $h == "opencode" or $h == "kimi" then false else true end; def profiles($value): @@ -799,6 +813,18 @@ crew_dispatch_validate() { fi } +startup_memory_budget_setup() { + # Primary bootstrap owns default publication. A secondmate is deliberately + # passive here because its setting must converge from the primary through the + # inherited-local-material contract rather than becoming a local authority. + if [ -e "$FM_HOME/.fm-secondmate-home" ] || [ -L "$FM_HOME/.fm-secondmate-home" ]; then + return 0 + fi + if ! fm_startup_memory_budget_materialize "$CONFIG"; then + echo "STARTUP_MEMORY_BUDGET: invalid config/$FM_STARTUP_MEMORY_BUDGET_FILE - $FM_STARTUP_MEMORY_BUDGET_ERROR" + fi +} + if [ "${1:-}" = "install" ]; then shift [ $# -gt 0 ] || { echo "usage: fm-bootstrap.sh install ..." >&2; exit 1; } @@ -821,6 +847,7 @@ fi # runnable. Detect-only sessions never touch state. if [ "${FM_BOOTSTRAP_DETECT_ONLY:-0}" != 1 ]; then "$SCRIPT_DIR/fm-pr-check-migrate.sh" || true + startup_memory_budget_setup fi if [ "$BACKEND_VALID" -eq 0 ]; then diff --git a/bin/fm-brief.sh b/bin/fm-brief.sh index 00ea34ddabe..9c98723b013 100755 --- a/bin/fm-brief.sh +++ b/bin/fm-brief.sh @@ -66,10 +66,31 @@ esac # shellcheck source=bin/fm-classify-lib.sh . "$SCRIPT_DIR/fm-classify-lib.sh" PAUSED_VERB=${FM_CLASSIFY_PAUSED_VERB:-$FM_CLASSIFY_PAUSED_VERB_DEFAULT} + +resolve_directory_input() { + local name=$1 path=$2 resolved + case "$path" in + /*) printf '%s\n' "$path"; return 0 ;; + esac + resolved=$(CDPATH='' cd -- "$path" 2>/dev/null && pwd -P) || { + echo "error: $name directory cannot be resolved: $path" >&2 + return 1 + } + printf '%s\n' "$resolved" +} + FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" -FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" -DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" -STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" +FM_HOME=$(resolve_directory_input FM_HOME "${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}") || exit 1 +if [ -n "${FM_DATA_OVERRIDE:-}" ]; then + DATA=$(resolve_directory_input FM_DATA_OVERRIDE "$FM_DATA_OVERRIDE") || exit 1 +else + DATA="$FM_HOME/data" +fi +if [ -n "${FM_STATE_OVERRIDE:-}" ]; then + STATE=$(resolve_directory_input FM_STATE_OVERRIDE "$FM_STATE_OVERRIDE") || exit 1 +else + STATE="$FM_HOME/state" +fi KIND=ship HERDR_LAB=0 NO_PROJECTS=0 @@ -217,13 +238,13 @@ HERDR_SECTION=$(printf '%s\n' \ 'Never bypass the helper, even for a read-only lifecycle probe or cleanup after failure.' \ 'The captain fleet uses the running `default` session.') else -HERDR_SECTION=$(cat <<'EOF' +IFS= read -r -d '' HERDR_SECTION <<'EOF' || true # Herdr lifecycle declaration - NOT ENABLED **HARD SAFETY GATE:** this scaffold cannot inspect the task text that replaces `{TASK}` later. If the task will start, stop, delete, restart, profile, or otherwise drive Herdr lifecycle behavior, stop and regenerate the brief with `--herdr-lab` before dispatch. Do not add Herdr lifecycle commands to this unguarded brief by hand. EOF -) +HERDR_SECTION=${HERDR_SECTION%$'\n'} fi if [ "$KIND" = scout ]; then @@ -285,19 +306,18 @@ case "$MODE" in direct-PR) SETUP2="" RULE1='1. Never push to the default branch (push only your `fm/'"$ID"'` branch). Never merge a PR.' - DOD=$(cat < "$BRIEF" </dev/null 2>&1 || true + +# --- scope: genuine primary checkout only ----------------------------------- +fm_primary_scope_matches "$FM_ROOT" "$STATE" || exit 0 + +# --- identity: only the lock-owning session's hooks may arm ------------------ +# A prior session may have died after leaving its numeric harness pid in .lock. +# Use the shared liveness predicate to recognize only that stale-owner case. +# Defer the mutating claim until after the unchanged AFK and need gates, so an +# idle or away home remains byte-for-byte inert. Missing or malformed locks are +# uncertainty rather than stale-owner evidence and remain inert. +RECOVER_SESSION_LOCK=0 +if ! fm_session_lock_owned_by_self "$STATE"; then + LOCK_PID=$(cat "$STATE/.lock" 2>/dev/null || true) + case "$LOCK_PID" in + ''|*[!0-9]*) exit 0 ;; + esac + fm_harness_pid_alive "$LOCK_PID" && exit 0 + RECOVER_SESSION_LOCK=1 +fi + +# --- AFK: the away daemon owns the watcher and triage; never rewake ---------- +[ -e "$STATE/.afk" ] && exit 0 + +# --- need: in-flight work or an X-mode relay poll ---------------------------- +need_supervision() { + fm_supervision_needed "$STATE" "$GRACE" +} +need_supervision || exit 0 + +# --- stale session-lock recovery --------------------------------------------- +# Delegate the claim to fm-lock.sh so its live-owner refusal and write semantics +# remain the single acquisition owner, then re-verify current-session identity +# before touching any auto-arm state. +if [ "$RECOVER_SESSION_LOCK" -eq 1 ]; then + "$SCRIPT_DIR/fm-lock.sh" >/dev/null 2>&1 || exit 0 + fm_session_lock_owned_by_self "$STATE" || exit 0 +fi + +# --- single-flight owner claim ------------------------------------------------ +# Claude runs one background process per firing with no dedupe. Exactly one +# owner foregrounds the arm and translates its close; every other firing exits +# 0 so one watcher cycle maps to at most one exit-2 rewake. +fm_lock_try_acquire "$OWNER_LOCK" || exit 0 +trap 'fm_lock_release "$OWNER_LOCK"' EXIT + +write_epoch() { # + local outcome=$1 seq tmp + seq=$(sed -n 's/^epoch=\([0-9][0-9]*\) .*/\1/p' "$EPOCH" 2>/dev/null || true) + case "$seq" in + ''|*[!0-9]*) seq=0 ;; + esac + seq=$((seq + 1)) + tmp="$EPOCH.tmp.$$" + printf 'epoch=%s owner_pid=%s outcome=%s updated_at=%s\n' \ + "$seq" "${BASHPID:-$$}" "$outcome" "$(date +%s)" > "$tmp" 2>/dev/null \ + && mv -f "$tmp" "$EPOCH" 2>/dev/null + rm -f "$tmp" 2>/dev/null || true +} + +write_epoch arming + +# X mode cadence: source the generated config so an X instance polls at its +# 30s cadence (fm-bootstrap.sh x_mode_setup contract). +# shellcheck source=/dev/null +[ -f "$CONFIG/x-mode.env" ] && . "$CONFIG/x-mode.env" + +# --- foreground the real arm wrapper ------------------------------------------ +# NO shell &: this hook process tree is the harness-owned lifecycle. The arm +# forks the watcher as its own tracked child exactly as it does for the +# model-driven background-task path, and propagates the wake reason on close. +OUT=$(mktemp "$STATE/.claude-autoarm-output.XXXXXX") || OUT= +if [ -n "$OUT" ]; then + "$SCRIPT_DIR/fm-watch-arm.sh" >"$OUT" 2>&1 + RC=$? +else + "$SCRIPT_DIR/fm-watch-arm.sh" >/dev/null 2>&1 + RC=$? +fi + +# --- classify and translate --------------------------------------------------- +# AFK may have appeared mid-cycle: the daemon owns triage now, so suppress the +# rewake even for an actionable close. +if [ -e "$STATE/.afk" ]; then + write_epoch afk + [ -z "$OUT" ] || rm -f "$OUT" 2>/dev/null || true + exit 0 +fi + +ACTIONABLE=0 +FAILED=0 +if [ -n "$OUT" ]; then + grep -Eq '^(signal:|stale:|check:|heartbeat($|:))' "$OUT" 2>/dev/null && ACTIONABLE=1 + grep -q '^watcher: FAILED' "$OUT" 2>/dev/null && FAILED=1 +fi +[ "$RC" -ne 0 ] && FAILED=1 + +if [ "$ACTIONABLE" -eq 0 ] && [ "$FAILED" -eq 0 ]; then + write_epoch clean + [ -z "$OUT" ] || rm -f "$OUT" 2>/dev/null || true + exit 0 +fi + +# The need may have vanished mid-cycle (fleet torn down, X opted out): nothing +# left to supervise, so close quietly instead of waking the model. +if ! need_supervision; then + write_epoch clean + [ -z "$OUT" ] || rm -f "$OUT" 2>/dev/null || true + exit 0 +fi + +write_epoch rewake +if [ "$FAILED" -eq 1 ]; then + { + printf 'firstmate watcher cycle FAILED - supervision is down while this home still needs it.\n' + [ -n "$OUT" ] && grep -E '^(watcher:|signal:|stale:|check:|heartbeat)' "$OUT" 2>/dev/null | head -8 + printf 'Run bin/fm-wake-drain.sh first. Then repair supervision with bin/fm-watch-arm.sh as its own Claude Code background task (never shell &). If the failure repeats, treat it as a blocker and report it instead of ending blind.\n' + } >&2 +else + { + printf 'firstmate watcher wake - one supervision event needs a handling turn now.\n' + [ -n "$OUT" ] && grep -E '^(signal:|stale:|check:|heartbeat)' "$OUT" 2>/dev/null | head -8 + printf 'Run bin/fm-wake-drain.sh first and handle the wake. This Stop hook owns watcher continuity: when the handling turn ends, the next needed cycle arms automatically - do NOT run bin/fm-watch-arm.sh after an ordinary wake.\n' + } >&2 +fi +[ -z "$OUT" ] || rm -f "$OUT" 2>/dev/null || true +exit 2 diff --git a/bin/fm-composer-lib.sh b/bin/fm-composer-lib.sh index 437b8c68977..6e2509ec2c1 100644 --- a/bin/fm-composer-lib.sh +++ b/bin/fm-composer-lib.sh @@ -39,11 +39,12 @@ # bold-wrapped) and no adapter covered grok's truecolor placeholder at all. # # Each adapter still owns its own CAPTURE and structural row-finding, because -# those use genuinely different primitives (tmux's cursor-row read, herdr's ANSI -# tail scan, orca/cmux's plain read-screen). Once an adapter has a candidate -# composer row it hands the RAW styled row to fm_composer_strip_ghost for the -# real-typed-content extraction, strips the box borders, trims, and hands the -# result plus a flag to fm_composer_classify_content for the shared +# those use genuinely different primitives (tmux's visible-pane box scan, +# herdr's ANSI tail scan, orca/cmux's plain read-screen). Once an adapter has a +# candidate composer row it hands the RAW styled row to +# fm_composer_strip_ghost for the real-typed-content extraction, strips the box +# borders, trims, and hands the result plus a flag to +# fm_composer_classify_content for the shared # empty|pending|unknown verdict. orca/cmux read a plain (unstyled) screen so # they have no ghost styling to strip and rely on the idle-placeholder match # below. Re-sourcing is a cheap idempotent redefinition, so this file needs no diff --git a/bin/fm-config-inherit-lib.sh b/bin/fm-config-inherit-lib.sh index 95abba2439a..bffbd5234d7 100644 --- a/bin/fm-config-inherit-lib.sh +++ b/bin/fm-config-inherit-lib.sh @@ -5,10 +5,13 @@ # (e.g. primary config/crew-dispatch.json makes a secondmate use the same dispatch # profile rules, primary config/crew-harness=codex makes a secondmate's crewmates # spawn on codex too, primary config/backlog-backend=manual makes that home -# hand-edit backlog files too, and primary config/herdr-presentation-spaces -# enables the same default-off Herdr presentation projection). It also pushes -# the one primary-authoritative shared captain-preference file, -# data/captain-shared.md, into each secondmate home's data/ as a read-only copy. +# hand-edit backlog files too, primary config/backend pins that home's local +# runtime-backend default for future spawns, primary config/startup-memory-budget +# bounds that home's startup-memory curation, and primary +# config/herdr-presentation-spaces enables the same default-off Herdr presentation +# projection). It also pushes the one primary-authoritative shared +# captain-preference file, data/captain-shared.md, into each secondmate home's +# data/ as a read-only copy. # # Usage: . bin/fm-config-inherit-lib.sh (no FM_* setup required) # @@ -30,6 +33,9 @@ # is deliberately NOT in the list: it is the primary's own setting for launching # secondmates, and a secondmate never spawns secondmates, so it must not flow # downstream. +# +# shellcheck source=bin/fm-startup-memory-budget-lib.sh +. "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/fm-startup-memory-budget-lib.sh" # The one shared data file in this inheritance contract. There is deliberately # no shared learnings file. @@ -40,7 +46,7 @@ FM_SHARED_CAPTAIN_MODE="444" # The declared inheritable set (space-separated, config-dir-relative item paths). # Extend here to inherit more of the primary's local config; override via the # environment only in tests. Items must not contain whitespace. -FM_INHERITABLE_CONFIG="${FM_INHERITABLE_CONFIG:-crew-dispatch.json crew-harness backlog-backend herdr-presentation-spaces}" +FM_INHERITABLE_CONFIG="${FM_INHERITABLE_CONFIG:-crew-dispatch.json crew-harness backlog-backend backend herdr-presentation-spaces startup-memory-budget}" fm_inherit_file_mode() { if [ "$(uname)" = Darwin ]; then @@ -399,6 +405,47 @@ propagate_inheritable_config() { esac src="$src_config/$item" dest="$dest_config/$item" + # This one scalar config is consumed as a local safety boundary, so reject + # every unsafe or malformed source/destination artifact before the generic + # byte-copy behavior below can treat it as ordinary inherited material. + if [ "$item" = "$FM_STARTUP_MEMORY_BUDGET_FILE" ]; then + if [ -e "$src_config" ] || [ -L "$src_config" ]; then + if ! fm_startup_memory_budget_config_dir_safe "$src_config"; then + reason="unsafe primary config directory: $FM_STARTUP_MEMORY_BUDGET_ERROR" + warn_inheritable_config_error "$item" "$src_config" "$reason" + record_inheritable_config_result "$item" error "$reason" + rc=1 + continue + fi + fi + if [ -e "$dest_config" ] || [ -L "$dest_config" ]; then + if ! fm_startup_memory_budget_config_dir_safe "$dest_config"; then + reason="unsafe destination config directory: $FM_STARTUP_MEMORY_BUDGET_ERROR" + warn_inheritable_config_error "$item" "$dest_config" "$reason" + record_inheritable_config_result "$item" error "$reason" + rc=1 + continue + fi + fi + if [ -e "$src" ] || [ -L "$src" ]; then + if ! fm_startup_memory_budget_file_valid "$src"; then + reason="unsafe or invalid primary source: $FM_STARTUP_MEMORY_BUDGET_ERROR" + warn_inheritable_config_error "$item" "$src" "$reason" + record_inheritable_config_result "$item" error "$reason" + rc=1 + continue + fi + fi + if [ -e "$dest" ] || [ -L "$dest" ]; then + if ! fm_startup_memory_budget_file_valid "$dest"; then + reason="unsafe or invalid destination: $FM_STARTUP_MEMORY_BUDGET_ERROR" + warn_inheritable_config_error "$item" "$dest" "$reason" + record_inheritable_config_result "$item" error "$reason" + rc=1 + continue + fi + fi + fi if [ -f "$src" ]; then if ! destination_allows_inherited_item "$dest_config" "$item"; then reason=$(inheritable_config_skip_reason) diff --git a/bin/fm-config-push.sh b/bin/fm-config-push.sh index b4056744bc7..b760666dd1b 100755 --- a/bin/fm-config-push.sh +++ b/bin/fm-config-push.sh @@ -3,8 +3,8 @@ # Usage: fm-config-push.sh [--help] # # Mid-session convergence for inherited local material such as -# config/crew-dispatch.json edits or data/captain-shared.md updates. This -# discovers live secondmate homes from state/*.meta, backfills +# config/crew-dispatch.json, config/backend, or data/captain-shared.md updates. +# This discovers live secondmate homes from state/*.meta, backfills # home= from data/secondmates.md for older meta records, and reuses the same # propagation machinery as bootstrap, but deliberately does not # fast-forward tracked files. diff --git a/bin/fm-continuity-command-policy.mjs b/bin/fm-continuity-command-policy.mjs deleted file mode 100755 index 34e635a778b..00000000000 --- a/bin/fm-continuity-command-policy.mjs +++ /dev/null @@ -1,145 +0,0 @@ -#!/usr/bin/env node -// Narrow shell classifier for the Claude watcher-continuity PreToolUse gate. -// -// The shared Lexer, program splitter, and command-position resolver remain owned -// by fm-arm-command-policy.mjs. This policy only identifies executed firstmate -// fleet scripts and divides them into recovery commands (wake drain, watcher -// arm, and fail-closed teardown) versus every other bin/fm-*.sh command. Unparseable or opaque dynamic -// commands fail open so this gate can never become a blanket shell block. - -import path from "node:path"; -import { fileURLToPath } from "node:url"; -import { Lexer, commandPosition, splitProgram } from "./fm-arm-command-policy.mjs"; - -const RECOVERY_SCRIPTS = new Set(["fm-wake-drain.sh", "fm-watch-arm.sh", "fm-teardown.sh"]); - -function parseArguments(argv) { - const result = { command: "", root: "" }; - for (let index = 0; index < argv.length; index += 1) { - const name = argv[index]; - if (name !== "--command" && name !== "--root") throw new Error(`unknown argument: ${name}`); - if (index + 1 >= argv.length) throw new Error(`${name} requires a value`); - result[name.slice(2)] = argv[index + 1]; - index += 1; - } - return result; -} - -function basename(value) { - return value.split("/").filter(Boolean).at(-1) || value; -} - -function fleetScript(value, root) { - const normalized = path.normalize(value); - const name = basename(normalized); - if (!/^fm-[A-Za-z0-9._-]+\.sh$/.test(name)) return ""; - const relative = `bin/${name}`; - if (normalized === relative || normalized === path.join(root, relative) || normalized.endsWith(`/${relative}`)) return name; - return ""; -} - -function literalShellPayload(position) { - if (!position.command || !["sh", "bash", "zsh"].includes(basename(position.command.value))) return null; - const words = position.words; - let noExecute = false; - for (let index = position.index + 1; index < words.length; index += 1) { - const word = words[index]; - if (/^-[A-Za-z]*n[A-Za-z]*$/.test(word.value)) noExecute = true; - if (/^-[A-Za-z]*c[A-Za-z]*$/.test(word.value)) { - const payload = words[index + 1]; - if (!payload || !payload.literal || payload.subs.length > 0) return null; - return { kind: noExecute ? "none" : "command", value: payload.value }; - } - if (/^[-+]O$/.test(word.value)) { - index += 1; - continue; - } - if (word.value === "--" || /^[-+]/.test(word.value)) continue; - return { kind: noExecute ? "none" : "script", value: word.value, index }; - } - return { kind: noExecute ? "none" : "stdin", value: "" }; -} - -function literalEvalPayload(position) { - if (!position.command || basename(position.command.value) !== "eval") return ""; - const payloads = position.words.slice(position.index + 1); - if (payloads.length === 0 || payloads.some((payload) => !payload.literal || payload.subs.length > 0)) return ""; - return payloads.map((payload) => payload.value).join(" "); -} - -function fleetScriptRecord(name, position, scriptIndex) { - const argumentsAfterScript = position.words.slice(scriptIndex + 1); - const unsafeTeardown = name === "fm-teardown.sh" - && argumentsAfterScript.some((argument) => argument.value === "--force" || !argument.literal); - return { name, unsafeTeardown }; -} - -function collectExecutedFleetScripts(command, root, depth = 0) { - if (depth > 12) return []; - const lexed = new Lexer(command).tokenize(); - if (lexed.error) return []; - const scripts = []; - const program = splitProgram(lexed.tokens); - - for (const tokens of program.nodes) { - const position = commandPosition(tokens); - const direct = fleetScript(position.command?.value || "", root); - if (direct) scripts.push(fleetScriptRecord(direct, position, position.index)); - - for (const token of tokens) { - if (token.type === "group") scripts.push(...collectExecutedFleetScripts(token.content, root, depth + 1)); - if (token.type !== "word") continue; - for (const substitution of token.subs) { - scripts.push(...collectExecutedFleetScripts(substitution.content, root, depth + 1)); - } - } - - const shell = literalShellPayload(position); - if (shell?.kind === "command") scripts.push(...collectExecutedFleetScripts(shell.value, root, depth + 1)); - if (shell?.kind === "script") { - const script = fleetScript(shell.value, root); - if (script) scripts.push(fleetScriptRecord(script, position, shell.index)); - } - if (shell?.kind === "stdin") { - for (const token of tokens) { - if (token.type === "redir" && token.fd === 0 && typeof token.heredoc === "string") { - scripts.push(...collectExecutedFleetScripts(token.heredoc, root, depth + 1)); - } - } - } - - const sourcedIndex = position.index + 1; - const sourced = position.command && [".", "source"].includes(position.command.value) - ? fleetScript(position.words[sourcedIndex]?.value || "", root) - : ""; - if (sourced) scripts.push(fleetScriptRecord(sourced, position, sourcedIndex)); - const evaluated = literalEvalPayload(position); - if (evaluated) scripts.push(...collectExecutedFleetScripts(evaluated, root, depth + 1)); - } - - return scripts; -} - -export function classifyContinuityCommand(command, root) { - const scripts = collectExecutedFleetScripts(command, root); - const blocked = scripts.find(({ name, unsafeTeardown }) => !RECOVERY_SCRIPTS.has(name) || unsafeTeardown); - if (!blocked) return { decision: "allow", script: "" }; - const code = blocked.unsafeTeardown ? "unsafe-teardown" : "other-fleet"; - return { decision: "deny", script: blocked.name, code }; -} - -function main() { - const args = parseArguments(process.argv.slice(2)); - if (!args.command || !args.root) return; - const result = classifyContinuityCommand(args.command, args.root); - if (result.decision === "deny") process.stdout.write(`deny\t${result.script}\t${result.code}\n`); -} - -const invokedPath = process.argv[1] ? path.resolve(process.argv[1]) : ""; -if (invokedPath === fileURLToPath(import.meta.url)) { - try { - main(); - } catch { - process.exitCode = 0; - } -} diff --git a/bin/fm-continuity-pretool-check.sh b/bin/fm-continuity-pretool-check.sh deleted file mode 100755 index aa28f2cef2e..00000000000 --- a/bin/fm-continuity-pretool-check.sh +++ /dev/null @@ -1,113 +0,0 @@ -#!/usr/bin/env bash -# Claude primary watcher-continuity PreToolUse gate. -# -# This hook is deliberately narrow. It denies only an executed bin/fm-*.sh fleet -# command other than bin/fm-wake-drain.sh, bin/fm-watch-arm.sh, or the -# independently fail-closed bin/fm-teardown.sh, and only when the active primary -# home has task metadata in flight but no identity-matched live watcher holds the -# home lock. Ordinary shell commands, recovery commands, healthy supervision, -# fleet-idle homes, and child worktrees are always allowed. -# -# The existing turn-end guard remains the unchanged final backstop. This gate -# closes the long-turn gap before another fleet mutation, but does not replace or -# weaken the Stop hook. -# -# Input is Claude PreToolUse JSON on stdin. Tests may pass --command directly. -# Malformed transport, missing jq/Node, a missing classifier, or classifier -# failure all fail open. A deny writes Claude's hook decision to stderr only and -# exits 2. -set -u - -COMMAND= -COMMAND_SET=0 - -usage() { - cat <<'EOF' -Usage: fm-continuity-pretool-check.sh [--command ] - -Reads Claude PreToolUse JSON from stdin unless --command is supplied. -Exits 0 to allow. Exits 2 with a Claude deny object on stderr only when an -unhealthy primary tries to execute a non-recovery firstmate fleet script. -EOF -} - -while [ "$#" -gt 0 ]; do - case "$1" in - --command) - [ "$#" -gt 1 ] || { echo "error: --command requires a value" >&2; exit 2; } - COMMAND=$2 - COMMAND_SET=1 - shift 2 - ;; - --command=*) - COMMAND=${1#--command=} - COMMAND_SET=1 - shift - ;; - -h|--help) - usage - exit 0 - ;; - *) - echo "error: unknown argument: $1" >&2 - usage >&2 - exit 2 - ;; - esac -done - -if [ "$COMMAND_SET" -eq 0 ]; then - PAYLOAD=$(cat 2>/dev/null || true) - [ -n "$PAYLOAD" ] || exit 0 - command -v jq >/dev/null 2>&1 || exit 0 - COMMAND=$(printf '%s' "$PAYLOAD" | jq -r '.tool_input.command // empty' 2>/dev/null) || exit 0 -fi -[ -n "$COMMAND" ] || exit 0 - -SCRIPT_DIR=$(CDPATH='' cd -- "$(dirname -- "${BASH_SOURCE[0]}")" 2>/dev/null && pwd -P) || exit 0 -FM_ROOT=${FM_ROOT_OVERRIDE:-$(CDPATH='' cd -- "$SCRIPT_DIR/.." 2>/dev/null && pwd -P)} -FM_HOME=${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}} -STATE=${FM_STATE_OVERRIDE:-$FM_HOME/state} -WATCH="$SCRIPT_DIR/fm-watch.sh" -POLICY="$SCRIPT_DIR/fm-continuity-command-policy.mjs" - -# shellcheck source=bin/fm-supervision-lib.sh -. "$SCRIPT_DIR/fm-supervision-lib.sh" -# shellcheck source=bin/fm-primary-scope-lib.sh -. "$SCRIPT_DIR/fm-primary-scope-lib.sh" -# shellcheck source=bin/fm-wake-lib.sh -. "$SCRIPT_DIR/fm-wake-lib.sh" - -fm_primary_scope_matches "$FM_ROOT" "$STATE" || exit 0 -fm_supervision_status "$STATE" "${FM_GUARD_GRACE:-300}" -[ "$FM_SUP_IN_FLIGHT" -gt 0 ] || exit 0 -LOCK_PID=$(cat "$STATE/.watch.lock/pid" 2>/dev/null || true) -if fm_pid_alive "$LOCK_PID" && fm_watcher_lock_matches_pid "$STATE" "$WATCH" "$LOCK_PID" "$FM_HOME"; then - exit 0 -fi - -command -v node >/dev/null 2>&1 || exit 0 -[ -f "$POLICY" ] || exit 0 -CLASSIFICATION=$(node "$POLICY" --command "$COMMAND" --root "$FM_ROOT" 2>/dev/null) || exit 0 -case "$CLASSIFICATION" in - deny*) ;; - *) exit 0 ;; -esac - -TAB=$(printf '\t') -REST=${CLASSIFICATION#*"$TAB"} -[ -n "$REST" ] && [ "$REST" != "$CLASSIFICATION" ] || exit 0 -BLOCKED_SCRIPT=${REST%%"$TAB"*} -REASON_CODE=${REST#*"$TAB"} -[ "$REASON_CODE" != "$REST" ] || REASON_CODE="" -case "$REASON_CODE" in - unsafe-teardown) - REASON="[watcher-continuity] tasks are in flight and no live watcher holds this home lock; during recovery only the ordinary literal bin/fm-teardown.sh is allowed, so drop --force and any shell-expanded arguments and retry the literal invocation (blocked: $BLOCKED_SCRIPT)" - ;; - *) - REASON="[watcher-continuity] tasks are in flight and no live watcher holds this home lock; drain wakes with bin/fm-wake-drain.sh, use fail-closed bin/fm-teardown.sh for completed tasks when needed, then re-arm with bin/fm-watch-arm.sh as a tracked Claude background task before running other fleet commands (blocked: $BLOCKED_SCRIPT)" - ;; -esac -ESCAPED=$(printf '%s' "$REASON" | sed -e 's/\\/\\\\/g' -e 's/"/\\"/g' | tr '\n' ' ') -printf '{"hookSpecificOutput":{"hookEventName":"PreToolUse","permissionDecision":"deny"},"systemMessage":"%s"}\n' "$ESCAPED" >&2 -exit 2 diff --git a/bin/fm-crew-state.sh b/bin/fm-crew-state.sh index 3281a99b1b2..32dff236687 100755 --- a/bin/fm-crew-state.sh +++ b/bin/fm-crew-state.sh @@ -96,6 +96,7 @@ meta_value() { # WT=$(meta_value worktree) KIND=$(meta_value kind) +HARNESS=$(meta_value harness) [ -n "$KIND" ] || KIND=ship # A torn-down (or never-created) worktree has no current state to read. @@ -150,8 +151,8 @@ pane_readable() { # } # crew_pane_is_busy: the busy-signature fallback, backend-aware the same way - # fm_backend_busy_state's native semantic state (herdr's agent.get) when -# available, else the shared tmux pane-regex reader (fm_pane_is_busy, -# bin/fm-tmux-lib.sh) unchanged for tmux/unknown. +# available, else the shared harness-scoped pane-regex reader +# (fm_pane_is_busy, bin/fm-tmux-lib.sh). # # `busy` alone is trusted outright. Both `idle` and unknown/unparseable fall # through to the shared tail-regex corroboration, NOT just unknown: herdr's @@ -162,9 +163,9 @@ pane_readable() { # # `no-mistakes axi run` without --yes, which blocks synchronously until a gate # or outcome - AGENTS.md section 7) is not generating for that whole span, so # agent.get can read idle/blocked (bin/backends/herdr.sh maps both to `idle`) -# while the pane's own rendered text still shows the harness's busy banner -# (BUSY_REGEX, e.g. "esc to interrupt") for the entire tool call, exactly like -# tmux's regex-only reader would correctly report. Trusting herdr's `idle` +# while the pane's own rendered text still shows that recorded harness's busy +# signature for the entire tool call, exactly like tmux's regex-only reader +# would correctly report. Trusting herdr's `idle` # outright (skipping that corroboration) is what let a still-working crew read # as not-busy here, and - combined with a no-mistakes run-step lookup that also # missed attribution (see nm_runs_status_for_branch) - as not provably working in @@ -174,7 +175,7 @@ pane_readable() { # # corroboration does not mask that case: it stays correctly not-busy. crew_pane_is_busy() { # case "$TASK_BACKEND" in - tmux) fm_pane_is_busy "$1" ;; + tmux) fm_pane_is_busy "$1" "$HARNESS" ;; *) local bs tail40 bs=$(fm_backend_busy_state "$TASK_BACKEND" "$1" 2>/dev/null) @@ -182,8 +183,8 @@ crew_pane_is_busy() { # busy) return 0 ;; *) tail40=$(fm_backend_capture "$TASK_BACKEND" "$1" 40 "$EXPECTED_LABEL" 2>/dev/null) || return 1 - printf '%s' "$tail40" | grep -v '^[[:space:]]*$' | tail -6 \ - | grep -qiE "${FM_BUSY_REGEX:-$FM_TMUX_BUSY_REGEX_DEFAULT}" + printf '%s' "$tail40" | grep -v '^[[:space:]]*$' | tail -12 \ + | fm_busy_lines_match "$HARNESS" ;; esac ;; diff --git a/bin/fm-dispatch-select.sh b/bin/fm-dispatch-select.sh deleted file mode 100755 index 5caf59ffa5a..00000000000 --- a/bin/fm-dispatch-select.sh +++ /dev/null @@ -1,337 +0,0 @@ -#!/usr/bin/env bash -# Resolve one already-matched crew-dispatch rule or default to a concrete profile. -# Usage: -# fm-dispatch-select.sh [--select ] [--quota-json ] [] -# -# Input may be a full rule object with `use` and optional `select`, a single -# profile object, or a non-empty array of profile objects. -# Output is one compact JSON profile object on stdout. -# Selection diagnostics go to stderr and never alter the profile JSON. -# -# This header is the single owner of quota-aware selection mechanics: -# - A profile object resolves to itself for backward compatibility. -# - Every profile array is quota-aware, whether or not it carries the legacy -# explicit `select: "quota-balanced"` strategy. -# - It runs the installed quota-axi --json (or the --quota-json fixture). -# - Candidates map to the quota provider and product their model consumes: -# direct Claude -> Claude, direct Codex -> Codex, direct Grok -> Grok Build, -# and Pi/OpenCode models prefixed anthropic/, openai-codex/, or xai/ -> -# Claude, Codex, or the xAI API product respectively. -# - A candidate's score is the minimum percentRemaining among its relevant -# general and matching model windows, or its exact Grok product window. -# Grok's aggregate credits window is used only when product windows are not -# exposed, so Grok Build and xAI API remain distinct. -# - Unscorable candidates never beat candidates with usable quota data. -# - Stale-but-cached numbers remain usable, but a fresh candidate wins unless -# the best stale score is at least the stale-clear margin higher (default -# 20 points). Equal winning scores use a random tie-break. -# - If quota-axi is unavailable, fails, returns unusable data, or no candidate -# can be scored, selection falls back uniformly across every valid candidate -# using rejection sampling over a 32-bit value from /dev/urandom. -# - Runtime quota trouble never turns malformed profile JSON into a fallback; -# invalid input exits 2 with an actionable validation error. -# -# FM_DISPATCH_QUOTA_AXI overrides the quota command. -# FM_DISPATCH_STALE_CLEAR_MARGIN overrides the default 20 point stale margin. -# FM_DISPATCH_RANDOM_SOURCE overrides /dev/urandom for deterministic tests only. -set -u - -STALE_CLEAR_MARGIN=${FM_DISPATCH_STALE_CLEAR_MARGIN:-20} -SELECT_OVERRIDE= -QUOTA_JSON_FILE= -ARGS=() - -usage() { - awk ' - NR == 1 { next } - /^#/ { sub(/^# ?/, ""); print; next } - { exit } - ' "$0" >&2 -} - -log() { - printf 'fm-dispatch-select: %s\n' "$*" >&2 -} - -while [ "$#" -gt 0 ]; do - case "$1" in - --select) - [ "$#" -gt 1 ] || { echo "error: --select requires a value" >&2; exit 2; } - SELECT_OVERRIDE=$2 - shift 2 - ;; - --select=*) - SELECT_OVERRIDE=${1#--select=} - shift - ;; - --quota-json) - [ "$#" -gt 1 ] || { echo "error: --quota-json requires a file" >&2; exit 2; } - QUOTA_JSON_FILE=$2 - shift 2 - ;; - --quota-json=*) - QUOTA_JSON_FILE=${1#--quota-json=} - shift - ;; - -h|--help) - usage - exit 0 - ;; - --) - shift - while [ "$#" -gt 0 ]; do - ARGS+=("$1") - shift - done - ;; - -*) - echo "error: unknown option $1" >&2 - exit 2 - ;; - *) - ARGS+=("$1") - shift - ;; - esac -done - -[ "${#ARGS[@]}" -le 1 ] || { echo "error: expected at most one JSON argument" >&2; exit 2; } -command -v jq >/dev/null 2>&1 || { echo "error: jq is required" >&2; exit 2; } -command -v od >/dev/null 2>&1 || { echo "error: od is required for OS-backed random selection" >&2; exit 2; } - -if [ "${#ARGS[@]}" -eq 1 ]; then - SPEC_JSON=${ARGS[0]} -else - SPEC_JSON=$(cat) -fi - -profiles_json=$(printf '%s\n' "$SPEC_JSON" | jq -ec ' - (if type == "object" and has("use") then .use else . end) - | if type == "array" then . - elif type == "object" then [.] - else error("dispatch input must be a rule, profile, or profile array") - end -' 2>/dev/null) || { echo "error: dispatch input must be a rule, profile, or profile array" >&2; exit 2; } - -validation_error=$(printf '%s\n' "$profiles_json" | jq -r ' - def verified($h): ["claude", "codex", "opencode", "pi", "grok"] | index($h); - def effort_ok($h; $e): - if $h == "claude" then ["low", "medium", "high", "xhigh", "max"] | index($e) - elif $h == "codex" then ["low", "medium", "high", "xhigh"] | index($e) - elif $h == "grok" then ["low", "medium", "high"] | index($e) - elif $h == "pi" then ["low", "medium", "high", "xhigh", "max"] | index($e) - elif $h == "opencode" then false - else false - end; - if length == 0 then "dispatch profile array must not be empty" - elif any(.[]; type != "object") then "each dispatch profile must be an object" - elif any(.[]; ((.harness? | type) != "string") or (.harness | length) == 0) then "each dispatch profile needs a non-empty harness" - elif any(.[]; has("model") and (((.model | type) != "string") or (.model | length) == 0)) then "dispatch profile model must be a non-empty string when present" - elif any(.[]; has("effort") and (((.effort | type) != "string") or (.effort | length) == 0)) then "dispatch profile effort must be a non-empty string when present" - elif any(.[]; .harness as $h | verified($h) | not) then "dispatch profile contains an unverified harness" - elif any(.[]; has("effort") and (. as $profile | effort_ok($profile.harness; $profile.effort) | not)) then "dispatch profile contains an unsupported harness/effort pair" - else empty - end -') -[ -z "$validation_error" ] || { echo "error: $validation_error" >&2; exit 2; } - -clean_profile_at() { - local index=$1 - printf '%s\n' "$profiles_json" | jq -c --argjson index "$index" ' - def clean($p): - {harness: $p.harness} - + (if ($p.model? | type) == "string" then {model: $p.model} else {} end) - + (if ($p.effort? | type) == "string" then {effort: $p.effort} else {} end); - clean(.[$index]) - ' -} - -random_index() { - local count=$1 source raw ceiling attempts - source=${FM_DISPATCH_RANDOM_SOURCE:-/dev/urandom} - [ "$count" -gt 0 ] || return 1 - [ -r "$source" ] || return 1 - ceiling=$((4294967296 - (4294967296 % count))) - attempts=0 - while [ "$attempts" -lt 32 ]; do - raw=$(LC_ALL=C od -An -N4 -tu4 "$source" 2>/dev/null | tr -d '[:space:]') - case "$raw" in - ''|*[!0-9]*) attempts=$((attempts + 1)); continue ;; - esac - if [ "$raw" -lt "$ceiling" ]; then - printf '%s\n' "$((raw % count))" - return 0 - fi - attempts=$((attempts + 1)) - done - return 1 -} - -random_profile() { - local reason count index - reason=$1 - count=$(printf '%s\n' "$profiles_json" | jq 'length') - if ! index=$(random_index "$count"); then - echo "error: OS-backed random source is unavailable" >&2 - exit 1 - fi - log "$reason" - log "selection basis: random fallback" - clean_profile_at "$index" -} - -select_strategy=$SELECT_OVERRIDE -if [ -z "$select_strategy" ]; then - select_strategy=$(printf '%s\n' "$SPEC_JSON" | jq -r ' - if type == "object" and has("use") and (.select? | type) == "string" then .select else "" end - ' 2>/dev/null || true) -fi -if [ -n "$select_strategy" ] && [ "$select_strategy" != quota-balanced ]; then - echo "error: unknown select strategy '$select_strategy'" >&2 - exit 2 -fi - -is_array=$(printf '%s\n' "$SPEC_JSON" | jq -r ' - if type == "object" and has("use") then (.use | type) == "array" else type == "array" end -') -if [ "$is_array" != true ] && [ -z "$select_strategy" ]; then - log "selection basis: single profile" - clean_profile_at 0 - exit 0 -fi - -if [ -n "$QUOTA_JSON_FILE" ]; then - if ! quota_json=$(cat "$QUOTA_JSON_FILE" 2>/dev/null); then - random_profile "cannot read quota JSON" - exit 0 - fi -else - quota_cmd=${FM_DISPATCH_QUOTA_AXI:-quota-axi} - if ! command -v "$quota_cmd" >/dev/null 2>&1; then - random_profile "quota-axi missing" - exit 0 - fi - quota_json=$("$quota_cmd" --json 2>/dev/null) - quota_status=$? - if [ "$quota_status" -ne 0 ]; then - random_profile "quota-axi exited $quota_status" - exit 0 - fi -fi - -if ! printf '%s\n' "$quota_json" | jq -e 'type == "object" and (.providers | type) == "array"' >/dev/null 2>&1; then - random_profile "quota-axi returned unparseable JSON" - exit 0 -fi - -selection=$(printf '%s\n' "$quota_json" | jq -ec \ - --argjson profiles "$profiles_json" \ - --argjson margin "$STALE_CLEAR_MARGIN" ' - def clean_text: - ascii_downcase | gsub("[^a-z0-9]"; ""); - def model_name($model): - ($model | split("/") | last | split(":") | first); - def route($profile): - ($profile.harness // "") as $h - | ($profile.model // "") as $model - | if $h == "claude" then {provider: "claude", model: $model} - elif $h == "codex" then {provider: "codex", model: $model} - elif $h == "grok" then {provider: "grok", product: "grok_build", model: $model} - elif (($h == "pi" or $h == "opencode") and ($model | startswith("anthropic/"))) then - {provider: "claude", model: (model_name($model))} - elif (($h == "pi" or $h == "opencode") and ($model | startswith("openai-codex/"))) then - {provider: "codex", model: (model_name($model))} - elif (($h == "pi" or $h == "opencode") and ($model | startswith("xai/"))) then - {provider: "grok", product: "api", model: (model_name($model))} - else null - end; - def provider_for($id): [.providers[]? | select(.provider == $id)][0]; - def model_window_matches($window; $model): - if (($window.kind? // "") != "model") or ($model | length) == 0 then false - else - (($window.id? // "") + " " + ($window.label? // "") | clean_text) as $scope - | ($model | clean_text) as $wanted - | (($scope | contains($wanted)) or ($wanted | contains($scope)) - or (["fable", "opus", "haiku", "sonnet", "spark"] - | map(. as $family | ($scope | contains($family)) and ($wanted | contains($family))) - | any)) - end; - def usable_percent($window): - (($window.percentRemaining? | type) == "number") - and ($window.percentRemaining >= 0) - and ($window.percentRemaining <= 100); - def general_window_matches($window; $provider): - if $provider == "claude" then ["five_hour", "seven_day"] | index($window.id? // "") != null - elif $provider == "codex" then ["five_hour", "weekly"] | index($window.id? // "") != null - else false - end; - def relevant_windows($provider; $route): - ($provider.windows // []) as $all_windows - | ($all_windows | map(select(usable_percent(.)))) as $windows - | if $route.provider == "grok" then - ($windows | map(select(.id == ("product:" + $route.product)))) as $product_windows - | if ($product_windows | length) > 0 then $product_windows - elif ($all_windows | map(select((.id? // "") | startswith("product:"))) | length) == 0 then - ($windows | map(select(.id == "credits"))) - else [] - end - else - $windows | map(select( - general_window_matches(.; $route.provider) - or model_window_matches(.; $route.model) - )) - end; - def candidate_metric($profile; $index): - . as $root - | route($profile) as $route - | if $route == null then empty - else ($root | provider_for($route.provider)) as $provider - | if ($provider == null) or (["fresh", "stale"] | index($provider.state.status? // "") | not) then empty - else relevant_windows($provider; $route) as $windows - | if ($windows | length) == 0 then empty - else { - index: $index, - score: ($windows | map(.percentRemaining) | min), - fresh: (($provider.state.status? // "") == "fresh") - } - end - end - end; - def best_score($items): if ($items | length) == 0 then null else ($items | map(.score) | max) end; - . as $quota_root - | ([$profiles | to_entries[] | . as $entry - | ($quota_root | candidate_metric($entry.value; $entry.key))]) as $candidates - | if ($candidates | length) == 0 then {fallback: true} - else - ($candidates | map(select(.fresh))) as $fresh - | ($candidates | map(select(.fresh | not))) as $stale - | best_score($fresh) as $fresh_best - | best_score($stale) as $stale_best - | (if $fresh_best != null and $stale_best != null then - if $stale_best >= ($fresh_best + $margin) then {items: $stale, score: $stale_best} - else {items: $fresh, score: $fresh_best} - end - elif $fresh_best != null then {items: $fresh, score: $fresh_best} - else {items: $stale, score: $stale_best} - end) as $winning - | {fallback: false, indices: [$winning.items[] | select(.score == $winning.score) | .index]} - end -' 2>/dev/null) || { - random_profile "quota-axi data could not be evaluated" - exit 0 -} - -if [ "$(printf '%s\n' "$selection" | jq -r '.fallback')" = true ]; then - random_profile "no usable quota windows for candidates" - exit 0 -fi - -winner_indices=$(printf '%s\n' "$selection" | jq -c '.indices') -winner_count=$(printf '%s\n' "$winner_indices" | jq 'length') -if ! winner_offset=$(random_index "$winner_count"); then - echo "error: OS-backed random source is unavailable" >&2 - exit 1 -fi -winner_index=$(printf '%s\n' "$winner_indices" | jq -r --argjson offset "$winner_offset" '.[$offset]') -log "selection basis: quota-selected" -clean_profile_at "$winner_index" diff --git a/bin/fm-doc-audience-check.sh b/bin/fm-doc-audience-check.sh new file mode 100755 index 00000000000..3551a7f0434 --- /dev/null +++ b/bin/fm-doc-audience-check.sh @@ -0,0 +1,269 @@ +#!/usr/bin/env bash +# fm-doc-audience-check.sh - validate the tracked documentation audience inventory. +# +# Usage: +# bin/fm-doc-audience-check.sh +# bin/fm-doc-audience-check.sh --root [--inventory ] +# +# The inventory owns classification and setup routing. +# This check validates structure only and does not keyword-lint prose. +set -eu + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +cd "$ROOT" +exec python3 - "$@" <<'PY' +from __future__ import annotations + +import argparse +import json +import os +import re +import subprocess +import sys +from collections import Counter +from pathlib import Path +from urllib.parse import unquote, urlsplit + +MARKDOWN_LINK_RE = re.compile(r"!?\[[^\]]*\]\(([^)]+)\)") +HTML_LINK_RE = re.compile(r"\b(?:href|src)=[\"']([^\"']+)[\"']", re.IGNORECASE) +REQUIRED_TRACKED_PATTERNS = ["*.md", "*.mdx", "*.rst", "*.txt", "docs/examples/*"] + + +class CheckError(Exception): + """One deterministic audience-check failure.""" + + +def fail(message: str) -> None: + raise CheckError(message) + + +def git_tracked(root: Path, patterns: list[str]) -> list[str]: + proc = subprocess.run( + ["git", "-C", str(root), "ls-files", "-z", "--", *patterns], + check=False, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + ) + if proc.returncode != 0: + detail = proc.stderr.decode("utf-8", "replace").strip() + fail(f"git ls-files failed: {detail or 'unknown error'}") + return sorted(p for p in proc.stdout.decode("utf-8").split("\0") if p) + + +def load_inventory(path: Path) -> dict: + try: + data = json.loads(path.read_text(encoding="utf-8")) + except FileNotFoundError: + fail(f"inventory is missing: {path}") + except (OSError, json.JSONDecodeError) as exc: + fail(f"inventory is unreadable: {exc}") + if not isinstance(data, dict): + fail("inventory root must be an object") + if data.get("version") != 1: + fail("inventory version must be 1") + return data + + +def list_of_strings(value: object, label: str) -> list[str]: + if not isinstance(value, list) or not value or not all(isinstance(v, str) and v for v in value): + fail(f"{label} must be a non-empty string array") + return value + + +def normalized_link_value(raw: str) -> str: + value = raw.strip() + if value.startswith("<") and value.endswith(">"): + value = value[1:-1].strip() + if " " in value: + value = value.split()[0] + return value + + +def resolve_local_target(root: Path, source: Path, raw: str) -> Path | None: + split = urlsplit(normalized_link_value(raw)) + if split.scheme or split.netloc: + return None + if not split.path: + return source.resolve(strict=False) if split.fragment else None + decoded = unquote(split.path) + if decoded.startswith("/"): + fail(f"absolute local link in {source.relative_to(root)}: {raw}") + target = (source.parent / decoded).resolve(strict=False) + try: + target.relative_to(root.resolve()) + except ValueError: + fail(f"local link escapes repository in {source.relative_to(root)}: {raw}") + return target + + +def markdown_local_links(root: Path, source: Path) -> list[tuple[str, Path]]: + try: + text = source.read_text(encoding="utf-8") + except (OSError, UnicodeDecodeError) as exc: + fail(f"cannot read prose surface {source.relative_to(root)}: {exc}") + raw_links = MARKDOWN_LINK_RE.findall(text) + HTML_LINK_RE.findall(text) + result: list[tuple[str, Path]] = [] + for raw in raw_links: + target = resolve_local_target(root, source, raw) + if target is not None: + result.append((raw, target)) + return result + + +def github_heading_slug(value: str) -> str: + value = re.sub(r"<[^>]+>", "", value) + value = value.replace("`", "").strip().lower() + value = re.sub(r"[^\w\- ]", "", value, flags=re.UNICODE) + return re.sub(r"\s", "-", value) + + +def markdown_anchors(path: Path) -> set[str]: + anchors: set[str] = set() + counts: Counter[str] = Counter() + try: + lines = path.read_text(encoding="utf-8").splitlines() + except (OSError, UnicodeDecodeError) as exc: + fail(f"cannot read link target {path}: {exc}") + for line in lines: + match = re.match(r"^#{1,6}\s+(.+?)\s*#*\s*$", line) + if match: + base = github_heading_slug(match.group(1)) + if base: + count = counts[base] + anchors.add(base if count == 0 else f"{base}-{count}") + counts[base] += 1 + for explicit in re.findall(r"<(?:a|span)\s+(?:name|id)=[\"']([^\"']+)[\"']", line, re.IGNORECASE): + anchors.add(explicit) + return anchors + + +def validate(root: Path, inventory_path: Path) -> tuple[int, int]: + data = load_inventory(inventory_path) + scope = data.get("scope") + if not isinstance(scope, dict): + fail("scope must be an object") + patterns = list_of_strings(scope.get("trackedPatterns"), "scope.trackedPatterns") + if patterns != REQUIRED_TRACKED_PATTERNS: + fail("scope.trackedPatterns must match the fixed maintained-prose scope") + audiences = set(list_of_strings(data.get("allowedAudiences"), "allowedAudiences")) + setup_audiences = set(list_of_strings(data.get("setupAudiences"), "setupAudiences")) + if not setup_audiences <= audiences: + fail("setupAudiences contains an audience outside allowedAudiences") + + surfaces = data.get("surfaces") + if not isinstance(surfaces, list): + fail("surfaces must be an array") + paths: list[str] = [] + classifications: dict[str, str] = {} + for index, entry in enumerate(surfaces): + if not isinstance(entry, dict): + fail(f"surfaces[{index}] must be an object") + path = entry.get("path") + audience = entry.get("audience") + if not isinstance(path, str) or not path: + fail(f"surfaces[{index}].path must be a non-empty string") + if audience not in audiences: + fail(f"{path}: unsupported audience {audience!r}") + paths.append(path) + classifications[path] = audience + + duplicates = sorted(path for path, count in Counter(paths).items() if count != 1) + if duplicates: + fail("surfaces classified more than once: " + ", ".join(duplicates)) + + tracked = set(git_tracked(root, patterns)) + classified = set(paths) + missing = sorted(tracked - classified) + extra = sorted(classified - tracked) + if missing or extra: + details = [] + if missing: + details.append("unclassified: " + ", ".join(missing)) + if extra: + details.append("not tracked/in scope: " + ", ".join(extra)) + fail("; ".join(details)) + + readme_path = root / "README.md" + readme_targets = { + os.path.relpath(target, root).replace(os.sep, "/") + for _, target in markdown_local_links(root, readme_path) + } + setup_targets = list_of_strings(data.get("readmeSetupTargets"), "readmeSetupTargets") + for target in setup_targets: + if target not in readme_targets: + fail(f"README setup target is not linked from README.md: {target}") + if classifications.get(target) not in setup_audiences: + fail( + f"README setup target {target} has disallowed audience " + f"{classifications.get(target)!r}" + ) + + pointers = data.get("requiredOwnerPointers") + if not isinstance(pointers, list) or not pointers: + fail("requiredOwnerPointers must be a non-empty array") + for index, pointer in enumerate(pointers): + if not isinstance(pointer, dict): + fail(f"requiredOwnerPointers[{index}] must be an object") + source = pointer.get("source") + target = pointer.get("target") + if not isinstance(source, str) or not isinstance(target, str) or not source or not target: + fail(f"requiredOwnerPointers[{index}] needs non-empty source and target") + source_path = root / source + target_path = root / target + if not source_path.exists(): + fail(f"owner-pointer source is missing: {source}") + if not target_path.exists(): + fail(f"owner-pointer target is missing: {target}") + try: + source_text = source_path.read_text(encoding="utf-8") + except (OSError, UnicodeDecodeError) as exc: + fail(f"owner-pointer source is unreadable {source}: {exc}") + linked_targets: set[str] = set() + if source_path.suffix.lower() in {".md", ".mdx"}: + linked_targets = { + os.path.relpath(linked, root).replace(os.sep, "/") + for _, linked in markdown_local_links(root, source_path) + } + if target not in source_text and target not in linked_targets: + fail(f"required owner pointer missing: {source} -> {target}") + + checked_links = 0 + anchor_cache: dict[Path, set[str]] = {} + for path in sorted(tracked): + if Path(path).suffix.lower() not in {".md", ".mdx"}: + continue + source = root / path + for raw, target in markdown_local_links(root, source): + checked_links += 1 + if not target.exists(): + fail(f"unresolved local link in {path}: {raw}") + fragment = unquote(urlsplit(normalized_link_value(raw)).fragment) + if fragment and target.is_file() and target.suffix.lower() in {".md", ".mdx"}: + anchors = anchor_cache.setdefault(target, markdown_anchors(target)) + if fragment not in anchors: + fail(f"unresolved local anchor in {path}: {raw}") + + return len(tracked), checked_links + + +def main() -> int: + parser = argparse.ArgumentParser(description="Validate Firstmate documentation audiences and local links.") + parser.add_argument("--root", type=Path, default=Path.cwd()) + parser.add_argument("--inventory", type=Path) + args = parser.parse_args() + root = args.root.resolve() + inventory_path = args.inventory or (root / "docs/documentation-audiences.json") + if not inventory_path.is_absolute(): + inventory_path = root / inventory_path + try: + surfaces, links = validate(root, inventory_path) + except CheckError as exc: + print(f"fm-doc-audience-check: {exc}", file=sys.stderr) + return 1 + print(f"fm-doc-audience-check: ok surfaces={surfaces} local_links={links}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) +PY diff --git a/bin/fm-guard.sh b/bin/fm-guard.sh index 029681c0b43..e36b7f46b0a 100755 --- a/bin/fm-guard.sh +++ b/bin/fm-guard.sh @@ -1,7 +1,7 @@ #!/usr/bin/env bash # Watcher liveness and worktree-tangle guard, called by supervision scripts, by # fm-wake-drain.sh after it empties queued wakes, and by fm-session-start.sh in -# read-only advisory mode when another session holds the fleet lock. +# read-only advisory mode whenever session-lock ownership was not verified. # First, always warn if the firstmate primary checkout (FM_ROOT) is on a named # non-default branch, because that means firstmate-on-itself work landed in the # primary instead of an isolated worktree. @@ -130,7 +130,7 @@ if [ -n "$tangle_branch" ]; then printf '● A crewmate likely branched/committed in the primary instead of its own worktree.\n' printf "● The work is SAFE on the '%s' ref.\n" "$tangle_branch" if [ "$READ_ONLY" -eq 1 ]; then - printf '● This read-only session must leave restore work to the session holding the fleet lock.\n' + printf '● This read-only session must leave restore work to a session with verified fleet-lock ownership.\n' else printf "● Restore the primary to '%s':\n" "$tangle_default" printf '● git -C %s checkout %s\n' "$FM_ROOT" "$tangle_default" @@ -212,7 +212,7 @@ fi # Dedup of the watcher-down banner never suppresses this warning. if "$queue_pending"; then if [ "$READ_ONLY" -eq 1 ]; then - echo "WARNING: queued wakes pending - left untouched for the session holding the fleet lock." >&2 + echo "WARNING: queued wakes pending - left untouched because this session lacks verified fleet-lock ownership." >&2 else echo "WARNING: queued wakes pending - drain them with bin/fm-wake-drain.sh before anything else." >&2 fi diff --git a/bin/fm-harness.sh b/bin/fm-harness.sh index 3267e922073..824b95804de 100755 --- a/bin/fm-harness.sh +++ b/bin/fm-harness.sh @@ -1,6 +1,6 @@ #!/usr/bin/env bash # Detect the agent harness this process tree runs on. -# Usage: fm-harness.sh print own harness: claude|codex|opencode|pi|grok|unknown +# Usage: fm-harness.sh print own harness: claude|codex|opencode|pi|pi-signed|grok|kimi|unknown # fm-harness.sh crew print the effective CREWMATE harness # (config/crew-harness; "default" resolves to own) # fm-harness.sh secondmate print the harness the PRIMARY uses to launch @@ -29,8 +29,17 @@ CONFIG="${FM_CONFIG_OVERRIDE:-$FM_HOME/config}" detect_own() { # Layer 1: environment markers for verified harnesses. + # Keep marker detection before ancestry detection as an explicit precedence rule. + # Only claude, pi, and grok set verified markers of their own; codex, opencode, + # and kimi are markerless, so a foreign marker retained in a terminal + # multiplexer's stored environment can silently misidentify one of them before + # ancestry is consulted. This is a precedence hazard, not evidence that + # CLAUDECODE inheritance into a kimi child was observed; it was not observed. [ "${CLAUDECODE:-}" = "1" ] && { echo claude; return; } - [ "${PI_CODING_AGENT:-}" = "true" ] && { echo pi; return; } + if [ "${PI_CODING_AGENT:-}" = "true" ]; then + if [ "${FM_PI_HARNESS:-}" = pi-signed ]; then echo pi-signed; else echo pi; fi + return + fi # grok sets GROK_AGENT=1 for its child/tool processes (verified, grok 0.2.73). # It does NOT set CLAUDECODE despite being Claude-Code-compatible, so this marker # is unambiguous when firstmate runs natively on grok. @@ -39,11 +48,13 @@ detect_own() { local pid=$$ comm args for _ in 1 2 3 4 5 6 7 8; do comm=$(ps -o comm= -p "$pid" 2>/dev/null) || break - case "$(basename "$comm")" in + case "$(basename -- "$comm")" in *claude*) echo claude; return ;; *codex*) echo codex; return ;; *opencode*) echo opencode; return ;; *grok*) echo grok; return ;; + kimi) echo kimi; return ;; + pi-signed) echo pi; return ;; pi) echo pi; return ;; node*|python*) # Bare interpreter: match the harness name in its script path. diff --git a/bin/fm-herdr-session-cleanup.sh b/bin/fm-herdr-session-cleanup.sh new file mode 100755 index 00000000000..05b6db9f947 --- /dev/null +++ b/bin/fm-herdr-session-cleanup.sh @@ -0,0 +1,383 @@ +#!/usr/bin/env bash +# Retire stale restored-shell Herdr presentation children at locked session start. +# +# Usage: fm-herdr-session-cleanup.sh +# +# The caller must already own this Firstmate home's session lock. This script is +# home-local and considers only the current named Herdr session and ordinary +# state/*.herdr-presentation journals in the effective FM_HOME. Each candidate +# is additionally serialized by the existing state/.spawn-.lock and the +# shared named-session Herdr presentation lock, in that order. +# +# A visible title is discovery only. Cleanup requires the exact current +# "└ · p:<22-char-token>" grammar, one token occurrence across +# the named-session snapshot, exactly one matching home-local journal, one tab, +# one pane, absent task metadata, no registered agent, and a process proof that +# the pane contains only one idle recognized shell with no child process. A +# version 2 journal must also bind the exact workspace, tab, and pane. +# Topology is first checked from one locked API snapshot, then every mutation +# prerequisite is immediately rechecked before the existing exact-pane +# focus-preserving close helper is called. +# The script never closes a workspace. It removes only the matching journal, +# and only after the exact pane is confirmed gone. Every error warns and returns +# success so session startup continues conservatively. +set -u + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" +FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" +STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" + +# shellcheck source=bin/fm-wake-lib.sh +. "$SCRIPT_DIR/fm-wake-lib.sh" +# shellcheck source=bin/fm-backend.sh +. "$SCRIPT_DIR/fm-backend.sh" +fm_backend_source herdr +# shellcheck source=bin/fm-pr-lib.sh +. "$SCRIPT_DIR/fm-pr-lib.sh" + +fm_herdr_cleanup_warn() { + printf 'warning: herdr session-start projection cleanup: %s\n' "$*" >&2 +} + +fm_herdr_cleanup_title_token() { # + local title=$1 prefix token rest + case "$title" in + '└ '*' · p:'*) ;; + *) return 1 ;; + esac + token=${title##*' · p:'} + prefix=${title%" · p:$token"} + [ "$prefix" != "$title" ] && [ -n "${prefix#'└ '}" ] || return 1 + [ "${#token}" -eq 22 ] || return 1 + case "$token" in *[!A-Za-z0-9_-]*) return 1 ;; esac + rest=${title#*p:} + [ "$rest" != "$title" ] || return 1 + case "$rest" in *p:*) return 1 ;; esac + printf '%s' "$token" +} + +fm_herdr_cleanup_home_identity() { + [ -d "$FM_HOME" ] && [ ! -L "$FM_HOME" ] || return 1 + (cd "$FM_HOME" 2>/dev/null && pwd -P) +} + +fm_herdr_cleanup_process_argv0() { # + printf '%s' "$1" | jq -er ' + .result.process_info.foreground_processes[0] as $process + | ($process.argv0 // $process.argv[0]) + | select(type == "string" and length > 0) + ' 2>/dev/null +} + +fm_herdr_cleanup_journal_matches() { # <session> <home-real> + local title=$1 session=$2 home_real=$3 journal id expected journal_home + [ -d "$STATE" ] && [ ! -L "$STATE" ] || return 1 + for journal in "$STATE"/*"$FM_BACKEND_HERDR_PRESENTATION_JOURNAL_SUFFIX"; do + [ -f "$journal" ] && [ ! -L "$journal" ] || continue + id=$(basename "$journal" "$FM_BACKEND_HERDR_PRESENTATION_JOURNAL_SUFFIX") + fm_task_id_creation_valid "$id" || continue + fm_backend_herdr_projection_journal_snapshot "$journal" "$id" || continue + if [ "$FM_BACKEND_HERDR_JOURNAL_VERSION" = 2 ]; then + journal_home=$(fm_backend_herdr_projection_home_identity \ + "$FM_BACKEND_HERDR_JOURNAL_HOME" 2>/dev/null) || continue + [ "$journal_home" = "$home_real" ] \ + && [ "$FM_BACKEND_HERDR_JOURNAL_SESSION" = "$session" ] || continue + fi + expected=$(fm_backend_herdr_projection_workspace_label \ + "$id" "$FM_BACKEND_HERDR_JOURNAL_PROJECTION_ID") + [ "$expected" = "$title" ] || continue + printf '%s\t%s\t%s\n' "$journal" "$id" "$FM_BACKEND_HERDR_JOURNAL_PROJECTION_ID" + done +} + +fm_herdr_cleanup_unique_match() { # <title> <session> <home-real> + local title=$1 session=$2 home_real=$3 matches count record + FM_HERDR_CLEANUP_JOURNAL= + FM_HERDR_CLEANUP_ID= + FM_HERDR_CLEANUP_TOKEN= + FM_HERDR_CLEANUP_VERSION= + FM_HERDR_CLEANUP_BOUND_WORKSPACE= + FM_HERDR_CLEANUP_BOUND_TAB= + FM_HERDR_CLEANUP_BOUND_PANE= + matches=$(fm_herdr_cleanup_journal_matches "$title" "$session" "$home_real") || return 1 + count=$(printf '%s\n' "$matches" | awk 'NF { n++ } END { print n+0 }') + [ "$count" -eq 1 ] || return 1 + record=$(printf '%s\n' "$matches" | awk 'NF { print; exit }') + FM_HERDR_CLEANUP_JOURNAL=${record%%$'\t'*} + record=${record#*$'\t'} + FM_HERDR_CLEANUP_ID=${record%%$'\t'*} + FM_HERDR_CLEANUP_TOKEN=${record#*$'\t'} + [ -n "$FM_HERDR_CLEANUP_JOURNAL" ] \ + && [ -n "$FM_HERDR_CLEANUP_ID" ] \ + && [ -n "$FM_HERDR_CLEANUP_TOKEN" ] || return 1 + fm_backend_herdr_projection_journal_snapshot \ + "$FM_HERDR_CLEANUP_JOURNAL" "$FM_HERDR_CLEANUP_ID" || return 1 + [ "$FM_BACKEND_HERDR_JOURNAL_PROJECTION_ID" = "$FM_HERDR_CLEANUP_TOKEN" ] || return 1 + FM_HERDR_CLEANUP_VERSION=$FM_BACKEND_HERDR_JOURNAL_VERSION + if [ "$FM_HERDR_CLEANUP_VERSION" = 2 ]; then + FM_HERDR_CLEANUP_BOUND_WORKSPACE=$FM_BACKEND_HERDR_JOURNAL_WORKSPACE_ID + FM_HERDR_CLEANUP_BOUND_TAB=$FM_BACKEND_HERDR_JOURNAL_TAB_ID + FM_HERDR_CLEANUP_BOUND_PANE=$FM_BACKEND_HERDR_JOURNAL_PANE_ID + fi +} + +fm_herdr_cleanup_process_is_idle_shell() { # <session> <pane-id> + local session=$1 pane=$2 info shell_pid foreground_pgid count + local process_pid name argv0 shell_name rows stat ps_bin + info=$(fm_backend_herdr_cli "$session" pane process-info --pane "$pane" 2>/dev/null) || return 1 + printf '%s' "$info" | jq -e --arg pane "$pane" ' + .result.type == "pane_process_info" + and .result.process_info.pane_id == $pane + ' >/dev/null 2>&1 || return 1 + shell_pid=$(printf '%s' "$info" | jq -er \ + '.result.process_info.shell_pid | select(type == "number" and . > 1) | floor' 2>/dev/null) || return 1 + foreground_pgid=$(printf '%s' "$info" | jq -er \ + '.result.process_info.foreground_process_group_id | select(type == "number" and . > 1) | floor' 2>/dev/null) || return 1 + [ "$foreground_pgid" = "$shell_pid" ] || return 1 + count=$(printf '%s' "$info" | jq -er \ + '.result.process_info.foreground_processes | select(type == "array") | length' 2>/dev/null) || return 1 + [ "$count" -eq 1 ] || return 1 + process_pid=$(printf '%s' "$info" | jq -er \ + '.result.process_info.foreground_processes[0].pid | select(type == "number") | floor' 2>/dev/null) || return 1 + [ "$process_pid" = "$shell_pid" ] || return 1 + name=$(printf '%s' "$info" | jq -er \ + '.result.process_info.foreground_processes[0].name | select(type == "string" and length > 0)' 2>/dev/null) || return 1 + argv0=$(fm_herdr_cleanup_process_argv0 "$info") || return 1 + shell_name=${name##*/} + argv0=${argv0#-} + argv0=${argv0##*/} + [ "$argv0" = "$shell_name" ] || return 1 + case "$shell_name" in sh|bash|zsh|dash|ksh|fish) ;; *) return 1 ;; esac + + ps_bin=${FM_HERDR_PS_BIN:-ps} + command -v "$ps_bin" >/dev/null 2>&1 || return 1 + rows=$("$ps_bin" -axo pid=,ppid= 2>/dev/null) || return 1 + printf '%s\n' "$rows" | awk -v shell="$shell_pid" ' + $1 == shell { found++ } + $2 == shell { child++ } + END { exit(found == 1 && child == 0 ? 0 : 1) } + ' || return 1 + stat=$("$ps_bin" -p "$shell_pid" -o stat= 2>/dev/null | tr -d '[:space:]') || return 1 + case "$stat" in S*|I*) ;; *) return 1 ;; esac +} + +fm_herdr_cleanup_snapshot_candidate() { # <snapshot> <workspace> <title> <token> <bound-workspace> <bound-tab> <bound-pane> + local snapshot=$1 workspace=$2 title=$3 token=$4 + local bound_workspace=$5 bound_tab=$6 bound_pane=$7 record + FM_HERDR_CLEANUP_TAB= + FM_HERDR_CLEANUP_PANE= + record=$(printf '%s' "$snapshot" | jq -er \ + --arg workspace "$workspace" --arg title "$title" --arg token "$token" \ + --arg bound_workspace "$bound_workspace" --arg bound_tab "$bound_tab" \ + --arg bound_pane "$bound_pane" ' + .result.snapshot as $s + | [$s.workspaces[]? | select(.workspace_id == $workspace)] as $workspaces + | [$s.tabs[]? | select(.workspace_id == $workspace)] as $tabs + | [$s.panes[]? | select(.workspace_id == $workspace)] as $panes + | ([ $s.workspaces[]?.label? // "" | + ((split("p:" + $token) | length) - 1) ] | add // 0) as $token_count + | select($workspaces | length == 1) + | select($workspaces[0].label == $title) + | select($workspaces[0].tab_count == 1 and $workspaces[0].pane_count == 1) + | select($tabs | length == 1) + | select($panes | length == 1) + | select($panes[0].tab_id == $tabs[0].tab_id) + | select($bound_workspace == "" or $workspace == $bound_workspace) + | select($bound_tab == "" or $tabs[0].tab_id == $bound_tab) + | select($bound_pane == "" or $panes[0].pane_id == $bound_pane) + | select($token_count == 1) + | select(($s.focused_workspace_id | type) == "string") + | select(($s.focused_tab_id | type) == "string") + | select(($s.focused_pane_id | type) == "string") + | select($s.focused_tab_id != $tabs[0].tab_id) + | [$tabs[0].tab_id, $panes[0].pane_id] | @tsv + ' 2>/dev/null) || return 1 + [ -n "$record" ] && [ "${record#*$'\t'}" != "$record" ] || return 1 + FM_HERDR_CLEANUP_TAB=${record%%$'\t'*} + FM_HERDR_CLEANUP_PANE=${record#*$'\t'} + [ -n "$FM_HERDR_CLEANUP_TAB" ] && [ -n "$FM_HERDR_CLEANUP_PANE" ] +} + +fm_herdr_cleanup_revalidate() { # <session> <workspace> <tab> <pane> <title> <token> <home-real> <journal> <task-id> <version> <bound-workspace> <bound-tab> <bound-pane> + local session=$1 workspace=$2 tab=$3 pane=$4 title=$5 token=$6 home_real=$7 + local journal=$8 id=$9 version=${10} bound_workspace=${11} bound_tab=${12} bound_pane=${13} + local workspaces workspace_info tabs panes focus + [ ! -e "$STATE/$id.meta" ] && [ ! -L "$STATE/$id.meta" ] || return 1 + fm_herdr_cleanup_unique_match "$title" "$session" "$home_real" || return 1 + [ "$FM_HERDR_CLEANUP_JOURNAL" = "$journal" ] \ + && [ "$FM_HERDR_CLEANUP_ID" = "$id" ] \ + && [ "$FM_HERDR_CLEANUP_TOKEN" = "$token" ] \ + && [ "$FM_HERDR_CLEANUP_VERSION" = "$version" ] \ + && [ "$FM_HERDR_CLEANUP_BOUND_WORKSPACE" = "$bound_workspace" ] \ + && [ "$FM_HERDR_CLEANUP_BOUND_TAB" = "$bound_tab" ] \ + && [ "$FM_HERDR_CLEANUP_BOUND_PANE" = "$bound_pane" ] || return 1 + + workspaces=$(fm_backend_herdr_cli "$session" workspace list 2>/dev/null) || return 1 + printf '%s' "$workspaces" | jq -e --arg workspace "$workspace" --arg title "$title" --arg token "$token" ' + ([.result.workspaces[]? | select(.workspace_id == $workspace and .label == $title)] | length) == 1 + and ([.result.workspaces[]?.label? // "" | + ((split("p:" + $token) | length) - 1)] | add // 0) == 1 + ' >/dev/null 2>&1 || return 1 + workspace_info=$(fm_backend_herdr_cli "$session" workspace get "$workspace" 2>/dev/null) || return 1 + printf '%s' "$workspace_info" | jq -e --arg workspace "$workspace" --arg title "$title" ' + .result.workspace.workspace_id == $workspace + and .result.workspace.label == $title + and .result.workspace.tab_count == 1 + and .result.workspace.pane_count == 1 + ' >/dev/null 2>&1 || return 1 + tabs=$(fm_backend_herdr_cli "$session" tab list --workspace "$workspace" 2>/dev/null) || return 1 + printf '%s' "$tabs" | jq -e --arg workspace "$workspace" --arg tab "$tab" ' + (.result.tabs | type) == "array" + and (.result.tabs | length) == 1 + and .result.tabs[0].workspace_id == $workspace + and .result.tabs[0].tab_id == $tab + ' >/dev/null 2>&1 || return 1 + panes=$(fm_backend_herdr_cli "$session" pane list --workspace "$workspace" 2>/dev/null) || return 1 + printf '%s' "$panes" | jq -e --arg workspace "$workspace" --arg tab "$tab" --arg pane "$pane" ' + (.result.panes | type) == "array" + and (.result.panes | length) == 1 + and .result.panes[0].workspace_id == $workspace + and .result.panes[0].tab_id == $tab + and .result.panes[0].pane_id == $pane + ' >/dev/null 2>&1 || return 1 + [ "$(fm_backend_herdr_pane_agent_state "$session" "$pane")" = no-agent ] || return 1 + fm_herdr_cleanup_process_is_idle_shell "$session" "$pane" || return 1 + focus=$(fm_backend_herdr_projection_focus_snapshot "$session") || return 1 + [ "${focus#*$'\t'}" != "$tab" ] +} + +fm_herdr_cleanup_one() { # <session> <workspace> <title> <home-real> + local session=$1 workspace=$2 title=$3 home_real=$4 token journal id task_lock + local version bound_workspace bound_tab bound_pane presentation_lock snapshot + local tab pane state close_status=0 + token=$(fm_herdr_cleanup_title_token "$title") || return 0 + if ! fm_herdr_cleanup_unique_match "$title" "$session" "$home_real"; then + return 0 + fi + journal=$FM_HERDR_CLEANUP_JOURNAL + id=$FM_HERDR_CLEANUP_ID + version=$FM_HERDR_CLEANUP_VERSION + bound_workspace=$FM_HERDR_CLEANUP_BOUND_WORKSPACE + bound_tab=$FM_HERDR_CLEANUP_BOUND_TAB + bound_pane=$FM_HERDR_CLEANUP_BOUND_PANE + [ "$FM_HERDR_CLEANUP_TOKEN" = "$token" ] || return 0 + task_lock="$STATE/.spawn-$id.lock" + if ! fm_lock_try_acquire "$task_lock"; then + fm_herdr_cleanup_warn "$id skipped because its task lock is busy" + return 0 + fi + presentation_lock=$(fm_backend_herdr_presentation_session_lock_path "$session" 2>/dev/null) || { + fm_lock_release "$task_lock" || true + fm_herdr_cleanup_warn "$id skipped because the shared presentation lock is unavailable" + return 0 + } + if ! fm_lock_try_acquire "$presentation_lock"; then + fm_lock_release "$task_lock" || true + fm_herdr_cleanup_warn "$id skipped because the shared presentation lock is busy" + return 0 + fi + + if [ -e "$STATE/$id.meta" ] || [ -L "$STATE/$id.meta" ]; then + fm_lock_release "$presentation_lock" || true + fm_lock_release "$task_lock" || true + return 0 + fi + snapshot=$(fm_backend_herdr_cli "$session" api snapshot 2>/dev/null) || snapshot= + if [ -z "$snapshot" ] \ + || ! fm_herdr_cleanup_snapshot_candidate \ + "$snapshot" "$workspace" "$title" "$token" \ + "$bound_workspace" "$bound_tab" "$bound_pane"; then + fm_herdr_cleanup_warn "$id preserved because its locked candidate snapshot was ambiguous" + fm_lock_release "$presentation_lock" || true + fm_lock_release "$task_lock" || true + return 0 + fi + tab=$FM_HERDR_CLEANUP_TAB + pane=$FM_HERDR_CLEANUP_PANE + if [ "$(fm_backend_herdr_pane_agent_state "$session" "$pane")" != no-agent ] \ + || ! fm_herdr_cleanup_process_is_idle_shell "$session" "$pane"; then + fm_herdr_cleanup_warn "$id preserved because its pane is not a provably idle childless shell" + fm_lock_release "$presentation_lock" || true + fm_lock_release "$task_lock" || true + return 0 + fi + if ! fm_herdr_cleanup_revalidate \ + "$session" "$workspace" "$tab" "$pane" "$title" "$token" "$home_real" \ + "$journal" "$id" "$version" "$bound_workspace" "$bound_tab" "$bound_pane"; then + fm_herdr_cleanup_warn "$id preserved because immediate revalidation changed or was unreadable" + fm_lock_release "$presentation_lock" || true + fm_lock_release "$task_lock" || true + return 0 + fi + + fm_backend_herdr_projection_close_pane_focus_preserving \ + "$session" "$pane" no-agent || close_status=$? + state=$(fm_backend_herdr_pane_agent_state "$session" "$pane") + if [ "$state" = dead ]; then + if [ -f "$journal" ] && [ ! -L "$journal" ] \ + && fm_herdr_cleanup_unique_match "$title" "$session" "$home_real" \ + && [ "$FM_HERDR_CLEANUP_JOURNAL" = "$journal" ] \ + && [ "$FM_HERDR_CLEANUP_ID" = "$id" ] \ + && [ "$FM_HERDR_CLEANUP_VERSION" = "$version" ] \ + && [ "$FM_HERDR_CLEANUP_BOUND_WORKSPACE" = "$bound_workspace" ] \ + && [ "$FM_HERDR_CLEANUP_BOUND_TAB" = "$bound_tab" ] \ + && [ "$FM_HERDR_CLEANUP_BOUND_PANE" = "$bound_pane" ] \ + && [ ! -e "$STATE/$id.meta" ] && [ ! -L "$STATE/$id.meta" ]; then + rm -f -- "$journal" || fm_herdr_cleanup_warn "$id pane closed but its journal could not be retired" + else + fm_herdr_cleanup_warn "$id pane closed but its journal changed and was preserved" + fi + elif [ "$close_status" -ne 0 ]; then + fm_herdr_cleanup_warn "$id preserved because exact focus-safe pane closure was refused or unconfirmed" + else + fm_herdr_cleanup_warn "$id preserved because exact pane closure could not be confirmed" + fi + fm_lock_release "$presentation_lock" || true + fm_lock_release "$task_lock" || true + return 0 +} + +fm_herdr_session_cleanup() { + local session home_real list candidates workspace title journal found=0 + [ -d "$STATE" ] && [ ! -L "$STATE" ] || return 0 + for journal in "$STATE"/*"$FM_BACKEND_HERDR_PRESENTATION_JOURNAL_SUFFIX"; do + if [ -f "$journal" ] && [ ! -L "$journal" ]; then + found=1 + break + fi + done + [ "$found" -eq 1 ] || return 0 + command -v herdr >/dev/null 2>&1 \ + && command -v jq >/dev/null 2>&1 || return 0 + home_real=$(fm_herdr_cleanup_home_identity) || { + fm_herdr_cleanup_warn 'home identity is unreadable; preserving every candidate' + return 0 + } + session=$(fm_backend_herdr_session) + list=$(fm_backend_herdr_cli "$session" workspace list 2>/dev/null) || { + fm_herdr_cleanup_warn "session '$session' workspace discovery failed; preserving every candidate" + return 0 + } + candidates=$(printf '%s' "$list" | jq -er ' + .result.workspaces + | select(type == "array") + | .[] + | select((.workspace_id | type) == "string" and (.workspace_id | length) > 0) + | select((.label | type) == "string" and (.label | length) > 0) + | [.workspace_id, .label] | @tsv + ' 2>/dev/null) || { + fm_herdr_cleanup_warn "session '$session' workspace discovery was unreadable; preserving every candidate" + return 0 + } + while IFS=$'\t' read -r workspace title; do + [ -n "$workspace" ] && [ -n "$title" ] || continue + fm_herdr_cleanup_one "$session" "$workspace" "$title" "$home_real" + done <<< "$candidates" + return 0 +} + +if [ "${FM_HERDR_SESSION_CLEANUP_SOURCE_ONLY:-0}" != 1 ]; then + fm_herdr_session_cleanup + exit 0 +fi diff --git a/bin/fm-kimi-turnend-hook.sh b/bin/fm-kimi-turnend-hook.sh new file mode 100755 index 00000000000..34fee869d30 --- /dev/null +++ b/bin/fm-kimi-turnend-hook.sh @@ -0,0 +1,276 @@ +#!/usr/bin/env bash +# Install or remove Firstmate's guarded Kimi crew turn-end hook. +# +# This command is the sole owner of the text-level edit to +# $HOME/.kimi-code/config.toml. It validates the existing TOML but never +# serializes it: install adds or replaces one marker-delimited Firstmate region, +# and remove excises only that region. Missing, malformed, symlinked, partially +# marked, or otherwise surprising config is refused without a config write. +# +# The installed Stop hook always exits 0 and stays silent. It reads cwd from the +# hook payload, checks for a .fm-kimi-turnend pointer before registry work, and +# touches a task turn-end marker only when the pointer names a Firstmate-created +# token in $HOME/.kimi-code/fm-turn-end.d/. +# +# Usage: +# fm-kimi-turnend-hook.sh install +# fm-kimi-turnend-hook.sh remove +set -u + +case "${1:-}" in + install|remove) ACTION=$1 ;; + -h|--help) + sed -n '2,18{s/^# \{0,1\}//;p;}' "$0" + exit 0 + ;; + *) + printf 'usage: %s install|remove\n' "${0##*/}" >&2 + exit 2 + ;; +esac + +if [ -z "${HOME:-}" ]; then + printf 'fm-kimi-turnend-hook: refused: HOME is unset.\n' >&2 + exit 1 +fi +if ! command -v python3 >/dev/null 2>&1; then + printf 'fm-kimi-turnend-hook: refused: python3 with tomllib is required to validate config.toml.\n' >&2 + exit 1 +fi +if [ "$ACTION" = install ] && ! command -v jq >/dev/null 2>&1; then + printf 'fm-kimi-turnend-hook: refused: jq is required by the installed Kimi turn-end hook.\n' >&2 + exit 1 +fi + +python3 - "$ACTION" "$HOME/.kimi-code" <<'PY' +import os +import re +import shutil +import stat +import sys +import tempfile + +try: + import tomllib +except ImportError: + print( + "fm-kimi-turnend-hook: refused: python3 with tomllib is required to validate config.toml.", + file=sys.stderr, + ) + raise SystemExit(1) + +ACTION = sys.argv[1] +CONFIG_DIR = sys.argv[2] +CONFIG = os.path.join(CONFIG_DIR, "config.toml") +HOOK = os.path.join(CONFIG_DIR, "fm-turn-end.sh") +REGISTRY = os.path.join(CONFIG_DIR, "fm-turn-end.d") +BEGIN = b"# BEGIN FIRSTMATE KIMI TURN-END HOOK" +BEGIN_OWNS_NEWLINE = BEGIN + b" (OWNS PRECEDING NEWLINE)" +END = b"# END FIRSTMATE KIMI TURN-END HOOK" +IDENTIFIER = b"FIRSTMATE KIMI TURN-END HOOK" +HOOK_NAME = b"fm-turn-end.sh" +TOKEN_NAME = re.compile(r"fm\.[A-Za-z0-9]{12}\Z") + +HOOK_BYTES = b'''#!/usr/bin/env bash +# Firstmate Kimi turn-end hook. Managed by fm-kimi-turnend-hook.sh. +# This hook is deliberately passive: every path is silent and exits zero. +set +e +exec >/dev/null 2>&1 +payload= +IFS= read -r payload || [ -n "$payload" ] || exit 0 +command -v jq >/dev/null 2>&1 || exit 0 +workspace=$(jq -er 'select(.hook_event_name == "Stop") | .cwd | strings | select(length > 0)' <<< "$payload" 2>/dev/null) || exit 0 +pointer="$workspace/.fm-kimi-turnend" +[ -f "$pointer" ] || exit 0 +first= +IFS= read -r -n 256 first < "$pointer" 2>/dev/null || [ -n "$first" ] || exit 0 +case "$first" in token=*) token=${first#token=} ;; *) exit 0 ;; esac +case "$token" in fm.????????????) : ;; *) exit 0 ;; esac +case "$token" in *[!A-Za-z0-9._-]*) exit 0 ;; esac +auth_dir=${HOME:-}/.kimi-code/fm-turn-end.d +[ -n "${HOME:-}" ] || exit 0 +target=$(cat "$auth_dir/$token" 2>/dev/null) || exit 0 +case "$target" in /*.turn-ended) : ;; *) exit 0 ;; esac +touch -- "$target" 2>/dev/null || true +exit 0 +''' + + +def refuse(reason: str) -> None: + print(f"fm-kimi-turnend-hook: refused: {reason}", file=sys.stderr) + raise SystemExit(1) + + +def regular_not_symlink(path: str, label: str) -> os.stat_result: + try: + info = os.lstat(path) + except FileNotFoundError: + refuse(f"{label} is missing at {path}.") + if stat.S_ISLNK(info.st_mode) or not stat.S_ISREG(info.st_mode): + refuse(f"{label} is not a regular non-symlink file at {path}.") + return info + + +def parse_toml(data: bytes, label: str): + try: + text = data.decode("utf-8") + except UnicodeDecodeError as error: + refuse(f"{label} is not UTF-8: {error}.") + try: + parsed = tomllib.loads(text) + except tomllib.TOMLDecodeError as error: + refuse(f"{label} is malformed TOML: {error}.") + hooks = parsed.get("hooks") + if hooks is not None and not isinstance(hooks, list): + refuse(f"{label} has an unexpected non-array 'hooks' value.") + return parsed + + +def locate_region(data: bytes): + normal_count = data.count(BEGIN + b"\n") + owned_count = data.count(BEGIN_OWNS_NEWLINE + b"\n") + end_count = data.count(END) + identifier_count = data.count(IDENTIFIER) + if normal_count + owned_count == 0 and end_count == 0 and identifier_count == 0: + return None + if normal_count + owned_count != 1 or end_count != 1 or identifier_count != 2: + refuse("config.toml has partial, duplicated, or altered Firstmate region markers.") + marker = BEGIN_OWNS_NEWLINE if owned_count else BEGIN + marker_at = data.find(marker) + if marker_at != 0 and data[marker_at - 1 : marker_at] != b"\n": + refuse("the Firstmate begin marker is not at a line boundary.") + start = marker_at + if owned_count: + if marker_at == 0 or data[marker_at - 1 : marker_at] != b"\n": + refuse("the Firstmate region claims a preceding newline that is absent.") + start -= 1 + end_at = data.find(END, marker_at + len(marker)) + if end_at < 0: + refuse("the Firstmate end marker is missing.") + after = end_at + len(END) + if after < len(data): + if data[after : after + 1] != b"\n": + refuse("the Firstmate end marker is not a complete line.") + after += 1 + return start, after, marker + + +def block(marker: bytes) -> bytes: + return b"\n".join( + ( + marker, + b"[[hooks]]", + b'event = "Stop"', + b'matcher = "^$"', + b'command = "bash \\"$HOME/.kimi-code/fm-turn-end.sh\\" >/dev/null 2>&1 || true"', + b"timeout = 1", + END, + b"", + ) + ) + + +def without_region(data: bytes, region) -> bytes: + prefix = data[: region[0]] + suffix = data[region[1] :] + # Retain a newline so removal cannot join the captain's preceding line to content appended after installation. + separator = b"\n" if region[2] == BEGIN_OWNS_NEWLINE and suffix else b"" + return prefix + separator + suffix + + +def atomic_write(path: str, data: bytes, mode: int) -> None: + fd, temporary = tempfile.mkstemp(prefix=f".{os.path.basename(path)}.", dir=os.path.dirname(path)) + try: + os.fchmod(fd, mode) + with os.fdopen(fd, "wb") as stream: + fd = -1 + stream.write(data) + stream.flush() + os.fsync(stream.fileno()) + os.replace(temporary, path) + except Exception: + if fd >= 0: + os.close(fd) + try: + os.unlink(temporary) + except FileNotFoundError: + pass + raise + + +def validate_firstmate_files_for_remove() -> None: + if os.path.lexists(HOOK): + info = regular_not_symlink(HOOK, "Firstmate hook script") + with open(HOOK, "rb") as stream: + if stream.read() != HOOK_BYTES: + refuse(f"Firstmate hook script has unexpected content at {HOOK}.") + if stat.S_IMODE(info.st_mode) & 0o077: + refuse(f"Firstmate hook script has unexpectedly broad permissions at {HOOK}.") + if os.path.lexists(REGISTRY): + info = os.lstat(REGISTRY) + if stat.S_ISLNK(info.st_mode) or not stat.S_ISDIR(info.st_mode): + refuse(f"Firstmate registry is not a regular directory at {REGISTRY}.") + for name in os.listdir(REGISTRY): + path = os.path.join(REGISTRY, name) + child = os.lstat(path) + if not TOKEN_NAME.fullmatch(name) or stat.S_ISLNK(child.st_mode) or not stat.S_ISREG(child.st_mode): + refuse(f"Firstmate registry contains an unexpected entry at {path}.") + + +try: + if not os.path.isdir(CONFIG_DIR) or os.path.islink(CONFIG_DIR): + refuse(f"Kimi config directory is missing or unexpected at {CONFIG_DIR}.") + config_info = regular_not_symlink(CONFIG, "Kimi config") + with open(CONFIG, "rb") as stream: + original = stream.read() + parse_toml(original, "config.toml") + region = locate_region(original) + outside = original if region is None else without_region(original, region) + if HOOK_NAME in outside: + refuse("config.toml references fm-turn-end.sh outside the Firstmate-owned region.") + + if ACTION == "install": + if os.path.lexists(REGISTRY): + info = os.lstat(REGISTRY) + if stat.S_ISLNK(info.st_mode) or not stat.S_ISDIR(info.st_mode): + refuse(f"Firstmate registry is not a regular directory at {REGISTRY}.") + if os.path.lexists(HOOK): + regular_not_symlink(HOOK, "Firstmate hook script") + with open(HOOK, "rb") as stream: + existing_hook = stream.read() + if existing_hook != HOOK_BYTES and not existing_hook.startswith( + b"#!/usr/bin/env bash\n# Firstmate Kimi turn-end hook." + ): + refuse(f"Firstmate hook path has unexpected content at {HOOK}.") + if region is None: + marker = BEGIN if original.endswith(b"\n") else BEGIN_OWNS_NEWLINE + addition = block(marker) + candidate = original + (b"" if original.endswith(b"\n") else b"\n") + addition + else: + candidate = original[: region[0]] + ( + (b"\n" if region[2] == BEGIN_OWNS_NEWLINE else b"") + block(region[2]) + ) + original[region[1] :] + parse_toml(candidate, "updated config.toml") + os.makedirs(REGISTRY, mode=0o700, exist_ok=True) + os.chmod(REGISTRY, 0o700) + installed_hook = None + if os.path.exists(HOOK): + with open(HOOK, "rb") as stream: + installed_hook = stream.read() + if installed_hook != HOOK_BYTES or stat.S_IMODE(os.stat(HOOK).st_mode) != 0o700: + atomic_write(HOOK, HOOK_BYTES, 0o700) + if candidate != original: + atomic_write(CONFIG, candidate, stat.S_IMODE(config_info.st_mode)) + else: + validate_firstmate_files_for_remove() + candidate = outside + parse_toml(candidate, "config.toml after Firstmate region removal") + if candidate != original: + atomic_write(CONFIG, candidate, stat.S_IMODE(config_info.st_mode)) + if os.path.lexists(HOOK): + os.unlink(HOOK) + if os.path.lexists(REGISTRY): + shutil.rmtree(REGISTRY) +except OSError as error: + refuse(f"filesystem operation failed: {error}.") +PY diff --git a/bin/fm-lint.sh b/bin/fm-lint.sh index caddb631804..d1d761dd271 100755 --- a/bin/fm-lint.sh +++ b/bin/fm-lint.sh @@ -22,6 +22,7 @@ # fm-lint.sh --jobs <1|2> [path]... override bounded worker count # fm-lint.sh --telemetry <path> ... write a quiet metrics snapshot # fm-lint.sh --required-version print the ShellCheck pin +# fm-lint.sh --list-files print the canonical file set # fm-lint.sh --help print this usage set -u @@ -83,11 +84,12 @@ if [ "${1:-}" = "--required-version" ]; then fi fm_lint_usage() { - sed -n '2,25{s/^# \{0,1\}//;p;}' "$SELF" + sed -n '2,26{s/^# \{0,1\}//;p;}' "$SELF" } JOBS=${FM_LINT_JOBS:-2} TELEMETRY=${FM_LINT_TELEMETRY:-} +LIST_FILES=0 while [ "$#" -gt 0 ]; do case "$1" in --jobs) @@ -108,6 +110,10 @@ while [ "$#" -gt 0 ]; do TELEMETRY=${1#*=} shift ;; + --list-files) + LIST_FILES=1 + shift + ;; --help|-h) fm_lint_usage exit 0 @@ -125,6 +131,22 @@ case "$JOBS" in *) printf 'fm-lint.sh: jobs must be 1 or 2, got %s.\n' "$JOBS" >&2; exit 2 ;; esac +if [ "$#" -gt 0 ]; then + ROOTS=("$@") +else + ROOTS=(bin/*.sh bin/backends/*.sh tests/*.sh) +fi +ROOT_COUNT=${#ROOTS[@]} + +if [ "$LIST_FILES" -eq 1 ]; then + [ "$#" -eq 0 ] || { + printf 'fm-lint.sh: --list-files does not accept explicit paths.\n' >&2 + exit 2 + } + printf '%s\n' "${ROOTS[@]}" + exit 0 +fi + if ! command -v shellcheck >/dev/null 2>&1; then printf 'fm-lint.sh: ShellCheck not found; install ShellCheck %s for CI parity.\n' \ "$REQUIRED_SHELLCHECK" >&2 @@ -144,15 +166,6 @@ if [ "$resolved" != "$REQUIRED_SHELLCHECK" ]; then exit 1 fi -if [ "$#" -gt 0 ]; then - ROOTS=("$@") -else - # Canonical file set: the one authoritative definition. Callers never repeat - # these globs, and every adapter and test shell remains an independent root. - ROOTS=(bin/*.sh bin/backends/*.sh tests/*.sh) -fi -ROOT_COUNT=${#ROOTS[@]} - if [ -n "$TELEMETRY" ]; then telemetry_parent=$(dirname "$TELEMETRY") [ -d "$telemetry_parent" ] || { diff --git a/bin/fm-lock.sh b/bin/fm-lock.sh index 33e4b0d279b..083675b2beb 100755 --- a/bin/fm-lock.sh +++ b/bin/fm-lock.sh @@ -3,7 +3,7 @@ # Writes the harness (agent) process PID found by walking the shell's ancestry, # which lives as long as the firstmate session - unlike the transient subshell # PID of any one tool call, which is dead moments after it is written. -# Usage: fm-lock.sh acquire; exit 1 if another live session holds it +# Usage: fm-lock.sh acquire; exit 1 unless ownership is verified # fm-lock.sh status print holder and liveness; always exits 0 set -u @@ -12,50 +12,76 @@ FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" LOCK="$STATE/.lock" -mkdir -p "$STATE" - -# Known harness command names; extend when a new adapter is verified. -HARNESS_RE='claude|codex|opencode|grok|^pi$' - -harness_pid() { - local pid=$$ comm args - for _ in 1 2 3 4 5 6 7 8; do - comm=$(ps -o comm= -p "$pid" 2>/dev/null) || return 1 - args=$(ps -o args= -p "$pid" 2>/dev/null) - if printf '%s' "$(basename "$comm")" | grep -qE "$HARNESS_RE"; then - echo "$pid"; return 0 - fi - # Bare interpreter (e.g. node): match the harness name in its script path. - case "$comm" in - *node*|*python*) printf '%s' "$args" | grep -qE "$HARNESS_RE" && { echo "$pid"; return 0; } ;; - esac - pid=$(ps -o ppid= -p "$pid" 2>/dev/null | tr -d ' ') - [ -n "$pid" ] && [ "$pid" -gt 1 ] || return 1 - done - return 1 +mkdir -p "$STATE" 2>/dev/null || { + echo "error: cannot create session-lock state directory $STATE; operate read-only until resolved" >&2 + exit 1 } -holder_alive() { # true if $1 is a live process that looks like a harness - local pid=$1 comm - kill -0 "$pid" 2>/dev/null || return 1 - comm=$(ps -o comm= -p "$pid" 2>/dev/null) || return 1 - printf '%s' "$(basename "$comm") $(ps -o args= -p "$pid" 2>/dev/null)" | grep -qE "$HARNESS_RE" -} +# Harness identity (FM_HARNESS_RE, ancestry walk, holder liveness) is owned by +# the shared session-lock lib so the Claude Stop auto-arm applies the exact +# same identity contract. +# shellcheck source=bin/fm-session-lock-lib.sh +. "$SCRIPT_DIR/fm-session-lock-lib.sh" if [ "${1:-}" = "status" ]; then if [ ! -f "$LOCK" ]; then echo "lock: free"; exit 0; fi - old=$(cat "$LOCK") - if holder_alive "$old"; then echo "lock: held by live harness pid $old"; else echo "lock: stale (pid $old dead or not a harness)"; fi + old=$(cat "$LOCK" 2>/dev/null) || { + echo "lock: unreadable" + exit 0 + } + if fm_harness_pid_alive "$old"; then echo "lock: held by live harness pid $old"; else echo "lock: stale (pid $old dead or not a harness)"; fi exit 0 fi -me=$(harness_pid) || { echo "error: cannot locate harness process in ancestry" >&2; exit 1; } -if [ -f "$LOCK" ]; then - old=$(cat "$LOCK") - if [ "$old" != "$me" ] && holder_alive "$old"; then +me=$(fm_harness_ancestry_pid) || { echo "error: cannot locate harness process in ancestry" >&2; exit 1; } +probe=$(mktemp "$STATE/.lock-write.XXXXXX" 2>/dev/null) || { + echo "error: cannot write session lock; operate read-only until resolved" >&2 + exit 1 +} +rm -f "$probe" 2>/dev/null || { + echo "error: cannot clean session-lock publication probe; operate read-only until resolved" >&2 + exit 1 +} +# shellcheck source=bin/fm-wake-lib.sh +. "$SCRIPT_DIR/fm-wake-lib.sh" +CLAIM_LOCK="$STATE/.lock.acquire" +CLAIM_LOCK_HELD=0 +release_claim_lock() { + if [ "$CLAIM_LOCK_HELD" -eq 1 ]; then + fm_lock_release "$CLAIM_LOCK" + CLAIM_LOCK_HELD=0 + fi +} +trap release_claim_lock EXIT +trap 'exit 1' HUP INT TERM +fm_lock_acquire_wait "$CLAIM_LOCK" +CLAIM_LOCK_HELD=1 + +if [ -e "$LOCK" ] || [ -L "$LOCK" ]; then + if [ ! -f "$LOCK" ] || [ -L "$LOCK" ]; then + echo "error: session lock is not a regular file; operate read-only until resolved" >&2 + exit 1 + fi + old=$(cat "$LOCK" 2>/dev/null) || { + echo "error: session lock is unreadable; operate read-only until resolved" >&2 + exit 1 + } + if [ "$old" != "$me" ] && fm_harness_pid_alive "$old"; then echo "error: another live firstmate session holds the lock (pid $old); operate read-only until resolved" >&2 exit 1 fi fi -echo "$me" > "$LOCK" +if ! { printf '%s\n' "$me" > "$LOCK"; } 2>/dev/null; then + echo "error: cannot write session lock; operate read-only until resolved" >&2 + exit 1 +fi +written=$(cat "$LOCK" 2>/dev/null) || { + echo "error: cannot verify session lock ownership; operate read-only until resolved" >&2 + exit 1 +} +if [ ! -f "$LOCK" ] || [ -L "$LOCK" ] || [ "$written" != "$me" ]; then + echo "error: session lock ownership verification failed; operate read-only until resolved" >&2 + exit 1 +fi +release_claim_lock echo "lock acquired: harness pid $me" diff --git a/bin/fm-pending-reply-lib.sh b/bin/fm-pending-reply-lib.sh index 9b8419b224d..5d04b65d673 100755 --- a/bin/fm-pending-reply-lib.sh +++ b/bin/fm-pending-reply-lib.sh @@ -573,8 +573,8 @@ fm_pending_reply_fallback_idle_eligible() { # <record-path> [ "$age" -ge "$grace" ] } -fm_pending_reply_backend_observation() { # <backend> <target> [expected-label] - local backend=$1 target=$2 expected_label=${3-} native tail40 +fm_pending_reply_backend_observation() { # <backend> <target> [expected-label] [harness] + local backend=$1 target=$2 expected_label=${3-} harness=${4-} native tail40 native=$(fm_backend_busy_state "$backend" "$target" 2>/dev/null || printf 'unknown') case "$native" in busy|idle) printf '%s' "$native"; return 0 ;; @@ -582,7 +582,7 @@ fm_pending_reply_backend_observation() { # <backend> <target> [expected-label] tail40=$(fm_backend_capture "$backend" "$target" 40 "$expected_label" 2>/dev/null) \ || { printf 'unknown'; return 0; } if printf '%s' "$tail40" | grep -v '^[[:space:]]*$' | tail -6 \ - | grep -qiE "${FM_BUSY_REGEX:-$FM_TMUX_BUSY_REGEX_DEFAULT}"; then + | fm_busy_lines_match "$harness"; then printf 'busy' else printf 'fallback-idle' @@ -925,7 +925,7 @@ fm_pending_reply_tick_one() { # <state-dir> <corr_id> <busy_state> [secondmate- # Never scrapes secondmate conversation; uses only parent status, backend busy # state, and optional secondmate-home wrong-home path checks. fm_pending_reply_tick() { # <state-dir> - local state=$1 dir rec corr task_id phase delivered meta backend target label busy sm_home + local state=$1 dir rec corr task_id phase delivered meta backend target label busy sm_home harness local observation observation_task found i local -a observation_tasks=() observation_values=() dir=$(fm_pending_reply_dir "$state") @@ -986,10 +986,12 @@ fm_pending_reply_tick() { # <state-dir> target= busy=unknown sm_home= + harness= if [ -f "$meta" ]; then backend=$(fm_backend_of_meta "$meta") target=$(fm_backend_target_of_meta "$meta") sm_home=$(fm_meta_get "$meta" home) + harness=$(fm_meta_get "$meta" harness) if [ -n "$target" ]; then label="fm-$task_id" observation= @@ -1002,7 +1004,7 @@ fm_pending_reply_tick() { # <state-dir> break done if [ "$found" = 0 ]; then - observation=$(fm_pending_reply_backend_observation "$backend" "$target" "$label") + observation=$(fm_pending_reply_backend_observation "$backend" "$target" "$label" "$harness") observation_tasks+=("$task_id") observation_values+=("$observation") fi diff --git a/bin/fm-secrets-check.mjs b/bin/fm-secrets-check.mjs index 0d20f8b59ed..a934de9c0ab 100755 --- a/bin/fm-secrets-check.mjs +++ b/bin/fm-secrets-check.mjs @@ -671,7 +671,7 @@ function validateTrackedWorkflows(workflowFiles, readTracked, manifest, projectL } const parsed = workflowJobs(content); const hasDopplerReference = - /dopplerhq\/secrets-fetch-action@|doppler-token:\s*|auth-method:\s*oidc\b/.test(content); + /dopplerhq\/secrets-fetch-action@|doppler-token:\s*|auth-method:\s*oidc\b|\bdoppler\s+run\b/.test(content); if (parsed.jobs.length === 0 || parsed.structuralErrors.length > 0) { fail( `${projectLabel}/${workflowFile}: workflow requires an unambiguous parsed jobs structure`, @@ -684,10 +684,12 @@ function validateTrackedWorkflows(workflowFiles, readTracked, manifest, projectL const triggerTrust = workflowTriggerTrust(content); parsed.jobs.forEach((job) => { const jobText = job.lines.join("\n"); - const usesDoppler = /uses:\s*dopplerhq\/secrets-fetch-action@/.test(jobText); + const usesDoppler = + /uses:\s*dopplerhq\/secrets-fetch-action@/.test(jobText) || /\bdoppler\s+run\b/.test(jobText); + const usesCliDoppler = /\bdoppler\s+run\b/.test(jobText); const usesServiceToken = /doppler-token:\s*\$\{\{\s*secrets\.[^}]+\}\}/.test(jobText); const usesOidc = /auth-method:\s*oidc\b/.test(jobText); - if (!usesDoppler || (!usesServiceToken && !usesOidc)) { + if (!usesDoppler || (!usesCliDoppler && !usesServiceToken && !usesOidc)) { return; } injectingJobs += 1; diff --git a/bin/fm-send.sh b/bin/fm-send.sh index e893cb8494c..dfae6f49e64 100755 --- a/bin/fm-send.sh +++ b/bin/fm-send.sh @@ -268,8 +268,8 @@ else esac retries=${FM_SEND_RETRIES:-3} sleep_s=${FM_SEND_SLEEP:-0.4} - # Type once, submit, verify. Lenient: only a positively-confirmed swallow - # (text still in the composer) is an error; an unreadable pane is assumed sent. + # Type once, submit, verify. Only exact empty confirms delivery; every other + # verdict preserves the loud refusal boundary. if ! verdict=$(fm_backend_send_text_submit "$TARGET_BACKEND" "$T" "$MESSAGE" "$retries" "$sleep_s" "$settle" "$EXPECTED_LABEL"); then if [ "$PENDING_REPLY_CREATED" = 1 ] && [ -n "$PENDING_REPLY_CORR" ]; then fm_pending_reply_discard_undelivered "$STATE" "$PENDING_REPLY_CORR" || true @@ -278,18 +278,20 @@ else exit 1 fi case "$verdict" in - pending) + empty) + ;; + send-failed) if [ "$PENDING_REPLY_CREATED" = 1 ] && [ -n "$PENDING_REPLY_CORR" ]; then fm_pending_reply_discard_undelivered "$STATE" "$PENDING_REPLY_CORR" || true fi - echo "error: text not submitted to $T (Enter swallowed; text left in composer; tried $RESOLUTION_TRIED)" >&2 + echo "error: text not sent to $T ($TARGET_BACKEND send failed; tried $RESOLUTION_TRIED)" >&2 exit 1 ;; - send-failed) + *) if [ "$PENDING_REPLY_CREATED" = 1 ] && [ -n "$PENDING_REPLY_CORR" ]; then fm_pending_reply_discard_undelivered "$STATE" "$PENDING_REPLY_CORR" || true fi - echo "error: text not sent to $T ($TARGET_BACKEND send failed; tried $RESOLUTION_TRIED)" >&2 + echo "error: text not submitted to $T (delivery unconfirmed; verdict=${verdict:-unknown}; tried $RESOLUTION_TRIED)" >&2 exit 1 ;; esac @@ -308,8 +310,8 @@ else exit 1 fi fi - # Submit landed (verdict was not pending/send-failed). Confirmation only proves - # the text was accepted; the harness still needs a beat to spin up the + # Submit landed with exact empty. Confirmation only proves the text was + # accepted; the harness still needs a beat to spin up the # turn before its busy footer shows. Pause so an immediate peek catches the # crewmate actually working instead of the stale idle pane. FM_SEND_SETTLE=0 # disables it. Scoped to this path only, never the shared submit core. diff --git a/bin/fm-session-lock-lib.sh b/bin/fm-session-lock-lib.sh new file mode 100644 index 00000000000..8343a8efd97 --- /dev/null +++ b/bin/fm-session-lock-lib.sh @@ -0,0 +1,96 @@ +#!/usr/bin/env bash +# Shared session-lock harness identity. +# +# ONE owner of the "which verified-harness process holds this home's session +# lock, and does the current process descend from that same harness?" decision. +# bin/fm-lock.sh uses it to acquire and inspect state/.lock; +# bin/fm-claude-stop-autoarm.sh uses it to prove a Stop hook fires inside the +# lock-owning primary session before it may arm or rewake. +# This file is sourced by scripts and has no side effects on source. + +# Known harness command names; extend when a new adapter is verified. +FM_HARNESS_RE='claude|codex|opencode|grok|kimi|^pi$|^pi-signed$' + +# Walk the current process ancestry (up to 16 hops) and print a harness pid. +# For every harness except Claude, the first match wins (innermost pid), which +# is where e.g. Pi's shared signed-wrapper ancestry actually holds the session: +# a "pi-signed" launcher can be the direct parent of the inner "pi" engine +# pid that owns the lock, and the wrapper pid above it is not that owner. +# Claude Code's bg-spare hook worker chain is the opposite shape: it nests +# several claude-named processes directly parent-child with no non-harness +# process between them, and the lock is held by the outermost pid of that +# run. So once a claude-named match is found, this keeps walking past it +# looking for a still-more-ancestral claude-named match, and stops the +# instant a non-match follows - never walking past that gap to an unrelated +# claude-named process further up the real process tree (e.g. the live +# session that launched a test as its own subprocess). The harness pid lives +# as long as the session, unlike the transient subshell pid of any one tool +# call. +fm_harness_ancestry_pid() { + local pid=$$ comm args best='' bc extending=0 hit=0 is_claude=0 + for _ in 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16; do + comm=$(ps -o comm= -p "$pid" 2>/dev/null) || break + args=$(ps -o args= -p "$pid" 2>/dev/null) + bc=$(basename -- "$comm") + hit=0; is_claude=0 + if printf '%s' "$bc" | grep -qE "$FM_HARNESS_RE"; then + hit=1 + case "$bc" in *claude*) is_claude=1 ;; esac + else + # Bare interpreter (e.g. node): match the harness name in its script path. + case "$comm" in + *node*|*python*) + if printf '%s' "$args" | grep -qE "$FM_HARNESS_RE"; then + hit=1 + case "$args" in *claude*) is_claude=1 ;; esac + fi + ;; + esac + fi + if [ "$hit" -eq 1 ]; then + best="$pid" + if [ "$is_claude" -eq 1 ]; then + extending=1 + else + break + fi + elif [ "$extending" -eq 1 ]; then + break + fi + pid=$(ps -o ppid= -p "$pid" 2>/dev/null | tr -d ' ') + [ -n "$pid" ] && [ "$pid" -gt 1 ] || break + done + [ -n "$best" ] && { echo "$best"; return 0; } + return 1 +} + +# True if $1 is a live process that looks like a verified harness. +fm_harness_pid_alive() { + local pid=$1 comm args + kill -0 "$pid" 2>/dev/null || return 1 + comm=$(ps -o comm= -p "$pid" 2>/dev/null) || return 1 + if printf '%s' "$(basename -- "$comm")" | grep -qE "$FM_HARNESS_RE"; then + return 0 + fi + case "$comm" in + *node*|*python*) + args=$(ps -o args= -p "$pid" 2>/dev/null) + printf '%s' "$args" | grep -qE "$FM_HARNESS_RE" + ;; + *) return 1 ;; + esac +} + +# True when state dir $1 holds a session lock whose pid is the harness ancestor +# of the current process: this script runs inside the session that owns the +# home's fleet lock. A missing lock, a lock held by another live harness, or an +# ancestry that cannot be resolved all fail closed. +fm_session_lock_owned_by_self() { + local state=$1 lock_pid my_pid + lock_pid=$(cat "$state/.lock" 2>/dev/null || true) + case "$lock_pid" in + ''|*[!0-9]*) return 1 ;; + esac + my_pid=$(fm_harness_ancestry_pid) || return 1 + [ "$my_pid" = "$lock_pid" ] +} diff --git a/bin/fm-session-start.sh b/bin/fm-session-start.sh index cdeb03f04a8..1abbace4bf1 100755 --- a/bin/fm-session-start.sh +++ b/bin/fm-session-start.sh @@ -27,10 +27,12 @@ # # 1. lock - acquire the per-home session lock FIRST, before any # mutating step runs. -# 2. bootstrap - detect-only diagnostics always run. The five -# MUTATING sweeps (legacy PR-check migration, secondmate -# fast-forward, secondmate liveness, X-mode artifact writes, fleet sync) run only -# when this session actually holds the lock. +# 2. bootstrap - home-local stale Herdr projection cleanup runs only +# when this session actually holds the lock. Detect-only +# diagnostics always run. Bootstrap's five MUTATING sweeps +# (legacy PR-check migration, secondmate fast-forward, +# secondmate liveness, X-mode artifact writes, fleet sync) +# also run only when locked. # 3. wake-drain - mutates the durable wake queue, so it also only runs # when locked. # 4. context digest - data/projects.md, data/secondmates.md, data/captain.md, @@ -61,9 +63,10 @@ # mode) for its read-only detect lines - missing tools, gh auth, the # worktree-tangle check, the harness override, crew-dispatch validation, # tasks-axi and quota-axi tool checks, and tasks-axi availability - none of -# which mutate shared state and all of which are safe to compute from a second -# session. -# Only the five mutating sweeps and the wake-queue drain are skipped. +# which mutate shared state and all of which are safe to compute without +# verified lock ownership. +# Only projection cleanup, the five bootstrap mutating sweeps, and the +# wake-queue drain are skipped. # The context and fleet-state digests # below are always read-only, so they run unconditionally in both modes. # @@ -251,10 +254,10 @@ if [ "$LOCK_RC" -ne 0 ]; then BAR='●━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━' { printf '%s\n' "$BAR" - printf '● READ-ONLY SESSION - ANOTHER LIVE FIRSTMATE SESSION HOLDS THE FLEET LOCK\n' + printf '● READ-ONLY SESSION - FLEET LOCK OWNERSHIP WAS NOT VERIFIED\n' printf '● %s\n' "$LOCK_OUT" - printf '● Skipping every mutating step: PR-check migration, secondmate sync,\n' - printf '● X-mode artifacts, fleet sync, and wake-queue drain. Detect-only bootstrap\n' + printf '● Skipping every mutating step: PR-check migration, stale Herdr child cleanup,\n' + printf '● secondmate sync, X-mode artifacts, fleet sync, and wake-queue drain. Detect-only bootstrap\n' printf '● diagnostics and the rest of this read-only-safe digest still ran below.\n' printf '● Operate read-only until this resolves - do not spawn, steer, merge, or\n' printf '● otherwise mutate fleet state from this session.\n' @@ -267,7 +270,10 @@ subsection "BOOTSTRAP" if [ "$READ_ONLY" -eq 1 ]; then BOOT_OUT=$(FM_BOOTSTRAP_DETECT_ONLY=1 "$SCRIPT_DIR/fm-bootstrap.sh" 2>&1) else - BOOT_OUT=$("$SCRIPT_DIR/fm-bootstrap.sh" 2>&1) + BOOT_OUT=$( + "$SCRIPT_DIR/fm-herdr-session-cleanup.sh" 2>&1 || true + "$SCRIPT_DIR/fm-bootstrap.sh" 2>&1 + ) fi if [ -n "$BOOT_OUT" ]; then printf '%s\n' "$BOOT_OUT" @@ -279,15 +285,15 @@ fi # Drained records are this turn's first work queue (AGENTS.md section 8); the # drain also runs fm-guard.sh internally on the locked path, so the # tangle/watcher-liveness alarms land right here too, ahead of the bulk digest -# below. The read-only path never touches the queue (another session -# may be actively draining it) but still runs fm-guard.sh directly with -# non-mutating advisory text, so the same alarms surface without repair -# commands. +# below. The read-only path never touches the queue because it lacks mutation +# authority, and another session may be actively draining it. It still runs +# fm-guard.sh directly with non-mutating advisory text, so the same alarms +# surface without repair commands. subsection "WAKE QUEUE" if [ "$READ_ONLY" -eq 1 ]; then QLEN=0 [ -s "$STATE/.wake-queue" ] && QLEN=$(grep -c . "$STATE/.wake-queue" 2>/dev/null || printf '0') - printf 'skipped (read-only session) - %s record(s) remain queued for the session holding the lock.\n' "$QLEN" + printf 'skipped (read-only session) - %s record(s) remain queued because this session lacks verified fleet-lock ownership.\n' "$QLEN" GUARD_OUT=$(FM_GUARD_READ_ONLY=1 "$SCRIPT_DIR/fm-guard.sh" 2>&1) [ -n "$GUARD_OUT" ] && printf '%s\n' "$GUARD_OUT" else @@ -305,17 +311,19 @@ AFK_PRESENT=0 X_MODE_PRESENT=0 [ -f "$CONFIG/x-mode.env" ] && X_MODE_PRESENT=1 -if [ "$PRIMARY_HARNESS" = pi ]; then +if [ "$PRIMARY_HARNESS" = pi ] || [ "$PRIMARY_HARNESS" = pi-signed ]; then PI_EXT="$FM_ROOT/.pi/extensions/fm-primary-pi-watch.ts" PI_TURNEND_EXT="$FM_ROOT/.pi/extensions/fm-primary-turnend-guard.ts" PI_WATCH_MARKER="$STATE/.pi-watch-extension-loaded" PI_TURNEND_MARKER="$STATE/.pi-turnend-extension-loaded" PI_LOCK="$STATE/.lock" + PI_RESTART_COMMAND=$PRIMARY_HARNESS + [ "$PRIMARY_HARNESS" != pi ] || PI_RESTART_COMMAND='plain pi' PI_WATCH_VERSION=$(hash_file "$PI_EXT" || printf '') PI_TURNEND_VERSION=$(hash_file "$PI_TURNEND_EXT" || printf '') if ! pi_extension_loaded "$PI_WATCH_MARKER" "$PI_WATCH_VERSION" "$PI_LOCK" \ || ! pi_extension_loaded "$PI_TURNEND_MARKER" "$PI_TURNEND_VERSION" "$PI_LOCK"; then - printf 'PI_WATCH_EXTENSION: not loaded - approve Pi project trust once per clone, then restart plain pi so %s and %s auto-load for turn-end guard and background wake coverage; use -e %s -e %s only if project hooks are not trusted\n' "$PI_TURNEND_EXT" "$PI_EXT" "$PI_TURNEND_EXT" "$PI_EXT" + printf 'PI_WATCH_EXTENSION: not loaded - approve Pi project trust once per clone, then restart %s so %s and %s auto-load for turn-end guard and background wake coverage; use -e %s -e %s only if project hooks are not trusted\n' "$PI_RESTART_COMMAND" "$PI_TURNEND_EXT" "$PI_EXT" "$PI_TURNEND_EXT" "$PI_EXT" fi fi "$SCRIPT_DIR/fm-supervision-instructions.sh" \ @@ -391,8 +399,8 @@ section "NEXT STEP" if [ "$READ_ONLY" -eq 1 ]; then cat <<'EOF' This session did not acquire the fleet lock. Stay read-only: do not arm, -drain, spawn, steer, merge, or repair fleet state from here. The session -holding the lock owns mutable follow-up. +drain, spawn, steer, merge, or repair fleet state from here. Only a session +with verified fleet-lock ownership may perform mutable follow-up. EOF elif [ "$AFK_PRESENT" -eq 1 ]; then diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index cbf9dd5aac4..3526572550c 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -59,10 +59,11 @@ # profile consultation. A --secondmate spawn is exempt and resolves the SECONDMATE # harness (config/secondmate-harness -> config/crew-harness -> own), so the # secondmate-vs-crewmate split is DURABLE across every respawn (recovery, -# /updatefirstmate, restart). A bare adapter name (claude|codex|opencode|pi|grok) +# /updatefirstmate, restart). A bare adapter name (claude|codex|opencode|pi|pi-signed|grok|kimi) # overrides it for this spawn (either kind). A non-flag string containing # whitespace is treated as a RAW launch command - the escape hatch for verifying -# new adapters. +# new adapters. pi-signed launches that exact executable name from PATH and +# refuses before endpoint creation when it is unavailable; it never falls back to pi. # config/secondmate-harness may also carry an optional model and effort as extra # whitespace-separated tokens ("<harness> [<model>] [<effort>]"). For a # --secondmate spawn, those tokens apply only when this spawn also resolves its @@ -101,7 +102,9 @@ # __PITURNEND__ absolute path to .pi/extensions/fm-primary-turnend-guard.ts in a pi secondmate home # __PIWATCH__ absolute path to .pi/extensions/fm-primary-pi-watch.ts in a pi secondmate home # __OPINPUT__ absolute path to the canonical operational-input encoder -# Per-harness turn-end hooks are installed automatically; some live outside the worktree. +# Verified per-harness turn-end hooks are installed automatically where enabled; some live outside the worktree. +# Kimi uses one surgically installed Firstmate region in $HOME/.kimi-code/config.toml, +# a firstmate-owned global hook and registry, and a gitignored per-task pointer. # grok uses a firstmate-owned global hook under ${GROK_HOME:-$HOME/.grok}/hooks # plus a gitignored .fm-grok-turnend worktree pointer and a state token. # On success prints: spawned <id> harness=<name> kind=<ship|scout|secondmate> mode=<mode> yolo=<on|off> window=<backend-target> worktree=<path> @@ -121,6 +124,26 @@ esac FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" + +resolve_directory_input() { + local name=$1 path=$2 resolved + case "$path" in + /*) printf '%s\n' "$path"; return 0 ;; + esac + resolved=$(CDPATH='' cd -- "$path" 2>/dev/null && pwd -P) || { + echo "error: $name directory cannot be resolved: $path" >&2 + return 1 + } + printf '%s\n' "$resolved" +} + +FM_HOME=$(resolve_directory_input FM_HOME "$FM_HOME") || exit 1 +if [ -n "${FM_STATE_OVERRIDE:-}" ]; then + FM_STATE_OVERRIDE=$(resolve_directory_input FM_STATE_OVERRIDE "$FM_STATE_OVERRIDE") || exit 1 +fi +if [ -n "${FM_DATA_OVERRIDE:-}" ]; then + FM_DATA_OVERRIDE=$(resolve_directory_input FM_DATA_OVERRIDE "$FM_DATA_OVERRIDE") || exit 1 +fi STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" PROJECTS="${FM_PROJECTS_OVERRIDE:-$FM_HOME/projects}" @@ -386,7 +409,7 @@ FIRSTMATE_HOME= if [ "$KIND" = secondmate ]; then case "${POS[1]:-}" in - ''|claude|codex|opencode|pi|grok) + ''|claude|codex|opencode|pi|pi-signed|grok|kimi) ARG3=${POS[1]:-} ;; *' '*) @@ -432,11 +455,11 @@ launch_template() { fi ;; opencode) printf '%s' 'OPENCODE_CONFIG_CONTENT='\''{"permission":{"*":"allow"}}'\'' opencode __MODELFLAG__--prompt "$(__OPINPUT__ encode launch-brief < __BRIEF__)"' ;; - pi) + pi|pi-signed) if [ "$kind" = secondmate ]; then - printf '%s' 'pi __MODELFLAG____EFFORTFLAG__-e __PITURNEND__ -e __PIWATCH__ "$(__OPINPUT__ encode launch-brief < __BRIEF__)"' + printf '%s%s' "$harness" ' __MODELFLAG____EFFORTFLAG__-e __PITURNEND__ -e __PIWATCH__ "$(__OPINPUT__ encode launch-brief < __BRIEF__)"' else - printf '%s' 'pi __MODELFLAG____EFFORTFLAG__-e __PIEXT__ "$(__OPINPUT__ encode launch-brief < __BRIEF__)"' + printf '%s%s' "$harness" ' __MODELFLAG____EFFORTFLAG__-e __PIEXT__ "$(__OPINPUT__ encode launch-brief < __BRIEF__)"' fi ;; # grok (Grok Build TUI): a positional prompt starts the supervised interactive @@ -447,6 +470,11 @@ launch_template() { # launch command - it is a Stop-event hook installed below (global hook + # per-task pointer), so the template is identical for ship/scout/secondmate. grok) printf '%s' 'grok --always-approve __MODELFLAG____EFFORTFLAG__"$(__OPINPUT__ encode launch-brief < __BRIEF__)"' ;; + # Kimi Code rejects a positional prompt, so it launches bare and receives + # only an absolute brief pointer after the TUI readiness gate below. + # Its turn-end signal is a globally configured Stop hook plus a guarded + # per-task worktree token, so no launch placeholder belongs here. + kimi) printf '%s' '__KIMIBIN__ __MODELFLAG__--auto' ;; *) return 1 ;; esac } @@ -487,6 +515,18 @@ case "$ARG3" in ;; esac +case "$HARNESS" in + pi|pi-signed) LAUNCH="FM_PI_HARNESS=$HARNESS $LAUNCH" ;; +esac + +# pi-signed is an explicitly selected executable identity, not an alias that may +# silently fall back to pi. Resolve it from PATH before creating an endpoint and +# retain the literal name in the launch command and task metadata. +if [ "$HARNESS" = pi-signed ] && ! command -v pi-signed >/dev/null 2>&1; then + echo "error: pi-signed executable not found on PATH; install the signed Pi wrapper or select a different verified harness" >&2 + exit 1 +fi + # config/secondmate-harness may carry optional model/effort tokens alongside the # harness ("<harness> [<model>] [<effort>]"). They apply only when this is a # --secondmate spawn and no explicit per-spawn harness/raw launch was supplied, so @@ -530,11 +570,35 @@ shell_quote() { printf "'" } +resolve_kimi_binary() { + local candidate dir fallback + candidate=$(command -v kimi 2>/dev/null || true) + if [ -n "$candidate" ] && [ -x "$candidate" ]; then + case "$candidate" in + /*) printf '%s\n' "$candidate"; return 0 ;; + *) + dir=$(cd "$(dirname "$candidate")" 2>/dev/null && pwd -P) || dir= + if [ -n "$dir" ]; then + printf '%s/%s\n' "$dir" "$(basename "$candidate")" + return 0 + fi + ;; + esac + fi + fallback="${HOME:-}/.kimi-code/bin/kimi" + if [ -n "${HOME:-}" ] && [ -x "$fallback" ]; then + printf '%s\n' "$fallback" + return 0 + fi + echo "error: kimi executable not found; searched PATH for 'kimi' and fallback '$fallback'" >&2 + return 1 +} + model_flag_for_harness() { local harness=$1 model=$2 [ -n "$model" ] && [ "$model" != default ] || return 0 case "$harness" in - claude|codex|opencode|pi|grok) + claude|codex|opencode|pi|pi-signed|grok|kimi) printf -- '--model %s ' "$(shell_quote "$model")" ;; esac @@ -566,7 +630,7 @@ effort_flag_for_harness() { low|medium|high) printf -- '--reasoning-effort %s ' "$(shell_quote "$effort")" ;; esac ;; - pi) + pi|pi-signed) # Pi 0.80.6 accepts the full shared effort vocabulary, including max, through # its --thinking flag. case "$effort" in @@ -576,9 +640,24 @@ effort_flag_for_harness() { # opencode's interactive `opencode --prompt` launch has a verified --model # flag but no verified effort flag. Its `opencode run --variant` flag belongs # to a different, non-interactive launch mode, so fm-spawn does not pass it. + # kimi likewise has no reasoning-effort flag; the requested axis stays in + # task metadata but never reaches the launch command. esac } +case "$LAUNCH" in + *__KIMIBIN__*) + KIMI_BIN=$(resolve_kimi_binary) || exit 1 + LAUNCH=${LAUNCH//__KIMIBIN__/$(shell_quote "$KIMI_BIN")} + if [ "$KIND" != secondmate ]; then + "$FM_ROOT/bin/fm-kimi-turnend-hook.sh" install || { + echo "error: refusing Kimi spawn because the global turn-end hook could not be installed safely" >&2 + exit 1 + } + fi + ;; +esac + json_escape() { printf '%s' "$1" | sed 's/\\/\\\\/g; s/"/\\"/g' } @@ -754,6 +833,8 @@ else BRIEF="$DATA/$ID/brief.md" fi [ -f "$BRIEF" ] || { echo "error: no brief at $BRIEF" >&2; exit 1; } +BRIEF_DIR_REAL=$(cd "$(dirname "$BRIEF")" && pwd -P) +BRIEF_REAL="$BRIEF_DIR_REAL/$(basename "$BRIEF")" # PROJ_ABS can still carry a symlinked path component (e.g. macOS's /tmp -> # /private/tmp) when it came from the ship/scout branch's logical `pwd` above. @@ -1125,6 +1206,58 @@ spawn_send_key() { # <target> <key> cmux) fm_backend_cmux_send_key "$1" "$2" "$W" ;; esac } + +kimi_capture() { + fm_backend_capture "$BACKEND" "$T" 120 "$W" 2>/dev/null || true +} + +kimi_capture_has_empty_composer() { # <plain-pane-capture> + printf '%s\n' "$1" \ + | grep -Eq '^[[:space:]]*(│|┃|\|)[[:space:]]*>[[:space:]]*(│|┃|\|)[[:space:]]*$' +} + +kimi_wait_for_ready() { + local pane i=0 max=${FM_KIMI_READY_POLLS:-60} interval=${FM_KIMI_POLL_INTERVAL:-0.5} + while [ "$i" -lt "$max" ]; do + pane=$(kimi_capture) + if printf '%s\n' "$pane" | grep -Fq 'Welcome to Kimi Code!' \ + || kimi_capture_has_empty_composer "$pane"; then + return 0 + fi + i=$((i + 1)) + [ "$i" -ge "$max" ] || sleep "$interval" + done + return 1 +} + +kimi_delivery_is_confirmed() { # <plain-pane-capture> + local pane=$1 + kimi_capture_has_empty_composer "$pane" || return 1 + if { printf '%s\n' "$pane" | grep -Fq '✨' \ + && printf '%s\n' "$pane" | grep -Fq 'Read the brief at'; } \ + || printf '%s\n' "$pane" \ + | grep -qiE 'context:[[:space:]]*(0\.[0-9]*[1-9][0-9]*|[1-9][0-9]*([.][0-9]+)?)[[:space:]]*%'; then + return 0 + fi + return 1 +} + +kimi_wait_for_delivery() { + local pane i=0 max=${FM_KIMI_DELIVERY_POLLS:-40} interval=${FM_KIMI_POLL_INTERVAL:-0.5} + while [ "$i" -lt "$max" ]; do + pane=$(kimi_capture) + kimi_delivery_is_confirmed "$pane" && return 0 + i=$((i + 1)) + [ "$i" -ge "$max" ] || sleep "$interval" + done + return 1 +} + +kimi_spawn_fail() { # <detail> + printf 'failed: %s\n' "$1" >> "$STATE/$ID.status" + echo "error: $1; inspect window $T" >&2 +} + if [ "$KIND" != secondmate ] && [ "$BACKEND" != orca ]; then spawn_send_text_line "$WT_TARGET" 'treehouse get' @@ -1183,9 +1316,10 @@ fi TASK_TMP="/tmp/fm-$ID" mkdir -p "$TASK_TMP/gotmp" -# Per-harness turn-end hook: a file that touches state/<id>.turn-ended when the -# agent finishes a turn. Worktree-resident hooks are kept out of git's view so -# they never block teardown's dirty check or leak into a commit. +# Per-harness turn-end hook where enabled: a file that touches +# state/<id>.turn-ended when the agent finishes a turn. Worktree-resident hooks +# and token pointers stay out of git's view so they never block teardown's dirty +# check or leak into a commit. mkdir -p "$STATE" STATE_REAL=$(cd "$STATE" && pwd -P) TURNEND="$STATE_REAL/$ID.turn-ended" @@ -1216,7 +1350,7 @@ export const FmTurnEnd = async ({ \$ }) => ({ EOF exclude_path '.opencode/plugins/fm-turn-end.js' ;; - pi*) + pi|pi-signed) # Written OUTSIDE the worktree: pi's project-trust gate fires on any extension # loaded from inside the project (verified live), but an explicit -e path # elsewhere loads without a dialog. Lives in state/, cleaned by teardown. @@ -1283,6 +1417,21 @@ EOF printf 'token=%s\n' "${auth_file##*/}" > "$WT/.fm-grok-turnend" exclude_path '.fm-grok-turnend' ;; + kimi*) + # Kimi's Stop hook is global, but it is inert unless cwd contains this + # task's token pointer and the token resolves through Firstmate's private + # registry. The installer above owns the format-preserving config edit and + # the always-zero, silent hook script. + KIMI_AUTH_DIR="$HOME/.kimi-code/fm-turn-end.d" + old_umask=$(umask) + umask 077 + auth_file=$(mktemp "$KIMI_AUTH_DIR/fm.XXXXXXXXXXXX") + umask "$old_umask" + printf '%s\n' "$TURNEND" > "$auth_file" + printf '%s\n' "${auth_file##*/}" > "$STATE/$ID.kimi-turnend-token" + printf 'token=%s\n' "${auth_file##*/}" > "$WT/.fm-kimi-turnend" + exclude_path '.fm-kimi-turnend' + ;; esac fi @@ -1306,6 +1455,7 @@ META_WINDOW=$T [ "$BACKEND" = orca ] && META_WINDOW=$W { echo "window=$META_WINDOW" + echo "endpoint_task_id=$ID" echo "worktree=$WT" echo "project=$PROJ_ABS" echo "harness=$HARNESS" @@ -1361,6 +1511,16 @@ LAUNCH=${LAUNCH//__PIEXT__/$sq_piext} LAUNCH=${LAUNCH//__PITURNEND__/$sq_piturnend} LAUNCH=${LAUNCH//__PIWATCH__/$sq_piwatch} LAUNCH=${LAUNCH//__OPINPUT__/$sq_opinput} +# Crewmate panes are created by a long-lived tmux/herdr daemon that does not +# inherit firstmate's current environment, so a bare `claude` in the pane falls +# back to the default ~/.claude store even when firstmate itself runs under a +# different CLAUDE_CONFIG_DIR (for example a work-vs-personal subscription split). +# Forward firstmate's own resolved store onto the claude launch so the crewmate +# uses the same credential/config firstmate is authenticated with. Only when set; +# an unset value is the single-store default and needs no prefix. +if [ "$HARNESS" = claude ] && [ -n "${CLAUDE_CONFIG_DIR:-}" ]; then + LAUNCH="CLAUDE_CONFIG_DIR=$(shell_quote "$CLAUDE_CONFIG_DIR") $LAUNCH" +fi if [ "$KIND" = secondmate ]; then sq_home=$(shell_quote "$PROJ_ABS") LAUNCH="FM_ROOT_OVERRIDE= FM_STATE_OVERRIDE= FM_DATA_OVERRIDE= FM_PROJECTS_OVERRIDE= FM_CONFIG_OVERRIDE= FM_HOME=$sq_home $LAUNCH" @@ -1377,6 +1537,30 @@ if [ "${HERDR_PROJECTED:-0}" -eq 1 ]; then spawn_herdr_presentation_order_lock_release fi spawn_send_key "$T" Enter +if [ "$HARNESS" = kimi ]; then + if ! kimi_wait_for_ready; then + kimi_spawn_fail "kimi did not show a verified ready signal before brief delivery" + exit 1 + fi + KIMI_POINTER="Read the brief at $BRIEF_REAL and follow it exactly." + KIMI_SUBMIT_RETRIES=${FM_KIMI_SUBMIT_RETRIES:-3} + KIMI_SUBMIT_SLEEP=${FM_KIMI_SUBMIT_SLEEP:-${FM_KIMI_POLL_INTERVAL:-0.5}} + KIMI_SUBMIT_SETTLE=${FM_KIMI_SUBMIT_SETTLE:-0} + KIMI_SUBMIT_VERDICT=$(fm_backend_send_text_submit \ + "$BACKEND" "$T" "$KIMI_POINTER" "$KIMI_SUBMIT_RETRIES" \ + "$KIMI_SUBMIT_SLEEP" "$KIMI_SUBMIT_SETTLE" "$W") || { + kimi_spawn_fail "kimi brief pointer could not be submitted" + exit 1 + } + if [ "$KIMI_SUBMIT_VERDICT" = send-failed ]; then + kimi_spawn_fail "kimi brief pointer could not be submitted" + exit 1 + fi + if ! kimi_wait_for_delivery; then + kimi_spawn_fail "kimi brief pointer delivery was not confirmed" + exit 1 + fi +fi if [ "$KIND" = secondmate ]; then if ! fm_config_reread_discard_pending "$PROJ_ABS" "$ID" "$FM_HOME"; then if fm_config_reread_quarantine_pending "$PROJ_ABS" "$ID" "$FM_HOME"; then diff --git a/bin/fm-startup-memory-budget-lib.sh b/bin/fm-startup-memory-budget-lib.sh new file mode 100644 index 00000000000..f2c06014b8e --- /dev/null +++ b/bin/fm-startup-memory-budget-lib.sh @@ -0,0 +1,224 @@ +# shellcheck shell=bash +# Startup-memory budget primitives. +# Usage: . bin/fm-startup-memory-budget-lib.sh +# +# The local, primary-authoritative config/startup-memory-budget setting is one +# strictly formatted positive decimal value followed by one newline. The +# locked primary bootstrap owns first materialization. This library owns safe +# parsing, default publication, and the portable prompt-memory estimate used by +# bin/fm-startup-memory-budget.sh and the internal /stow skill. + +FM_STARTUP_MEMORY_BUDGET_FILE="startup-memory-budget" +FM_STARTUP_MEMORY_BUDGET_DEFAULT="7500" +FM_STARTUP_MEMORY_BUDGET_ERROR="" +FM_STARTUP_MEMORY_BUDGET_VALUE="" +FM_STARTUP_MEMORY_MEASURE_BYTES="" +FM_STARTUP_MEMORY_MEASURE_TOKENS="" +FM_STARTUP_MEMORY_MEASURE_PRESENCE="" + +fm_startup_memory_budget_fail() { + FM_STARTUP_MEMORY_BUDGET_ERROR=$1 + return 1 +} + +fm_startup_memory_budget_link_count() { + if [ "$(uname)" = Darwin ]; then + stat -f %l "$1" 2>/dev/null + else + stat -c %h "$1" 2>/dev/null + fi +} + +fm_startup_memory_budget_config_dir_safe() { + local dir=$1 + if [ -L "$dir" ]; then + fm_startup_memory_budget_fail "config directory is symlinked" + return 1 + fi + if [ ! -d "$dir" ]; then + fm_startup_memory_budget_fail "config directory is not a directory" + return 1 + fi + return 0 +} + +# fm_startup_memory_budget_file_valid <path> +# Sets FM_STARTUP_MEMORY_BUDGET_VALUE only for a regular, single-linked file +# containing exactly one positive decimal value and one terminating newline. +fm_startup_memory_budget_file_valid() { + local path=$1 links value + FM_STARTUP_MEMORY_BUDGET_VALUE="" + if [ -L "$path" ]; then + fm_startup_memory_budget_fail "file is symlinked" + return 1 + fi + if [ ! -e "$path" ]; then + fm_startup_memory_budget_fail "file is absent" + return 1 + fi + if [ ! -f "$path" ]; then + fm_startup_memory_budget_fail "file is not a regular file" + return 1 + fi + links=$(fm_startup_memory_budget_link_count "$path") || { + fm_startup_memory_budget_fail "could not inspect file link count" + return 1 + } + if [ "$links" != 1 ]; then + fm_startup_memory_budget_fail "file is hardlinked" + return 1 + fi + value=$(<"$path") || { + fm_startup_memory_budget_fail "could not read file" + return 1 + } + case "$value" in + ''|0|*[!0-9]*|0*) + fm_startup_memory_budget_fail "value must be one positive decimal integer" + return 1 + ;; + esac + if ! printf '%s\n' "$value" | cmp -s "$path" -; then + fm_startup_memory_budget_fail "file must contain exactly one value followed by one newline" + return 1 + fi + FM_STARTUP_MEMORY_BUDGET_VALUE=$value + return 0 +} + +# fm_startup_memory_budget_read <config-dir> +# Prints the validated decimal value. It never treats an absent or unsafe file +# as an implicit default because callers need a visible, auditable setting. +fm_startup_memory_budget_read() { + local config_dir=$1 path + fm_startup_memory_budget_config_dir_safe "$config_dir" || return 1 + path="$config_dir/$FM_STARTUP_MEMORY_BUDGET_FILE" + fm_startup_memory_budget_file_valid "$path" || return 1 + printf '%s\n' "$FM_STARTUP_MEMORY_BUDGET_VALUE" +} + +# fm_startup_memory_budget_materialize <config-dir> +# Atomically publishes the visible default only when the file is absent. A +# concurrent valid creator is accepted; every unsafe or malformed existing +# artifact is rejected without replacement. +fm_startup_memory_budget_materialize() { + local config_dir=$1 path tmp + if [ -e "$config_dir" ] || [ -L "$config_dir" ]; then + fm_startup_memory_budget_config_dir_safe "$config_dir" || return 1 + else + mkdir -p "$config_dir" 2>/dev/null || { + fm_startup_memory_budget_fail "could not create config directory" + return 1 + } + fm_startup_memory_budget_config_dir_safe "$config_dir" || return 1 + fi + + path="$config_dir/$FM_STARTUP_MEMORY_BUDGET_FILE" + if [ -e "$path" ] || [ -L "$path" ]; then + fm_startup_memory_budget_read "$config_dir" >/dev/null || return 1 + return 0 + fi + + tmp=$(umask 077; mktemp "$config_dir/.startup-memory-budget.XXXXXX" 2>/dev/null) || { + fm_startup_memory_budget_fail "could not create default temporary file" + return 1 + } + if ! printf '%s\n' "$FM_STARTUP_MEMORY_BUDGET_DEFAULT" > "$tmp" \ + || ! fm_startup_memory_budget_file_valid "$tmp"; then + rm -f "$tmp" + [ -n "$FM_STARTUP_MEMORY_BUDGET_ERROR" ] \ + || fm_startup_memory_budget_fail "could not write default value" + return 1 + fi + + # link(2) gives no-clobber publication in this directory. Removing the + # temporary name leaves the published file with exactly one link. + if ln "$tmp" "$path" 2>/dev/null; then + rm -f "$tmp" + fm_startup_memory_budget_read "$config_dir" >/dev/null || return 1 + return 0 + fi + rm -f "$tmp" + # Another actor may have created the file. Accept it only if it now meets + # the same safe, exact format - never replace or guess at it. + fm_startup_memory_budget_read "$config_dir" >/dev/null +} + +# fm_startup_memory_estimated_tokens_for_bytes <non-negative bytes> +# The estimate is ceil(UTF-8 bytes / 3): stable, dependency-free, and +# deliberately conservative for ordinary prompt text without claiming provider +# exactness. +fm_startup_memory_estimated_tokens_for_bytes() { + local bytes=$1 tokens + case "$bytes" in + ''|*[!0-9]*) return 1 ;; + esac + tokens=$((bytes / 3)) + if [ $((bytes % 3)) -ne 0 ]; then + tokens=$((tokens + 1)) + fi + printf '%s\n' "$tokens" +} + +# fm_startup_memory_measure_file <path> +# Prints "<bytes> <estimated-tokens> <present|absent>". Memory files must be +# ordinary files when present so a measurement never follows a symlink or reads +# a special file. +fm_startup_memory_measure_file() { + local path=$1 bytes tokens + FM_STARTUP_MEMORY_MEASURE_BYTES="" + FM_STARTUP_MEMORY_MEASURE_TOKENS="" + FM_STARTUP_MEMORY_MEASURE_PRESENCE="" + if [ ! -e "$path" ] && [ ! -L "$path" ]; then + FM_STARTUP_MEMORY_MEASURE_BYTES=0 + FM_STARTUP_MEMORY_MEASURE_TOKENS=0 + FM_STARTUP_MEMORY_MEASURE_PRESENCE=absent + printf '0 0 absent\n' + return 0 + fi + if [ -L "$path" ] || [ ! -f "$path" ]; then + fm_startup_memory_budget_fail "memory file is not an ordinary regular file: $path" + return 1 + fi + bytes=$(LC_ALL=C wc -c < "$path" 2>/dev/null | tr -d '[:space:]') || { + fm_startup_memory_budget_fail "could not measure memory file: $path" + return 1 + } + case "$bytes" in + ''|*[!0-9]*) + fm_startup_memory_budget_fail "invalid byte count for memory file: $path" + return 1 + ;; + esac + tokens=$(fm_startup_memory_estimated_tokens_for_bytes "$bytes") || { + fm_startup_memory_budget_fail "could not estimate memory tokens for: $path" + return 1 + } + # shellcheck disable=SC2034 # Public measurement result consumed by the caller after sourcing. + FM_STARTUP_MEMORY_MEASURE_BYTES=$bytes + # shellcheck disable=SC2034 # Public measurement result consumed by the caller after sourcing. + FM_STARTUP_MEMORY_MEASURE_TOKENS=$tokens + # shellcheck disable=SC2034 # Public measurement result consumed by the caller after sourcing. + FM_STARTUP_MEMORY_MEASURE_PRESENCE=present + printf '%s %s present\n' "$bytes" "$tokens" +} + +# fm_startup_memory_decimal_le <left> <right> +# Decimal comparison without shell arithmetic overflow. Inputs are normalized +# non-negative decimal strings. +fm_startup_memory_decimal_le() { + local left=$1 right=$2 left_len right_len + case "$left:$right" in + *[!0-9:]*|:*|*:) return 1 ;; + esac + left_len=${#left} + right_len=${#right} + if [ "$left_len" -lt "$right_len" ]; then + return 0 + fi + if [ "$left_len" -gt "$right_len" ]; then + return 1 + fi + [ "$left" = "$right" ] && return 0 + [[ "$left" < "$right" ]] +} diff --git a/bin/fm-startup-memory-budget.sh b/bin/fm-startup-memory-budget.sh new file mode 100755 index 00000000000..715da549482 --- /dev/null +++ b/bin/fm-startup-memory-budget.sh @@ -0,0 +1,94 @@ +#!/usr/bin/env bash +# Read and account for the local startup-memory budget. +# Usage: +# fm-startup-memory-budget.sh read +# fm-startup-memory-budget.sh report +# +# `read` prints the one validated effective budget from +# config/startup-memory-budget. `report` prints the stable local estimate for +# data/captain.md, data/captain-shared.md, and data/learnings.md together. +# Bootstrap owns default materialization; this command never creates or repairs +# configuration, so an absent, malformed, symlinked, hardlinked, or otherwise +# unsafe value is a concrete error rather than an inferred default. +set -eu + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" +FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" +CONFIG="${FM_CONFIG_OVERRIDE:-$FM_HOME/config}" +DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" + +# shellcheck source=bin/fm-startup-memory-budget-lib.sh +. "$SCRIPT_DIR/fm-startup-memory-budget-lib.sh" + +usage() { + sed -n '2,11{s/^# \{0,1\}//;p;}' "$0" +} + +print_error() { + printf 'startup-memory-budget: %s\n' "$1" >&2 +} + +read_budget() { + if ! fm_startup_memory_budget_read "$CONFIG" >/dev/null; then + print_error "invalid config/$FM_STARTUP_MEMORY_BUDGET_FILE - $FM_STARTUP_MEMORY_BUDGET_ERROR" + return 1 + fi + printf '%s\n' "$FM_STARTUP_MEMORY_BUDGET_VALUE" +} + +report() { + local budget bytes tokens presence total=0 shared_tokens=0 role=primary + if ! budget=$(read_budget); then + return 2 + fi + + if [ -e "$FM_HOME/.fm-secondmate-home" ] || [ -L "$FM_HOME/.fm-secondmate-home" ]; then + role=secondmate + fi + + printf 'estimator=ceil(UTF-8 bytes / 3) conservative-local-estimate\n' + printf 'role=%s\n' "$role" + printf 'effective_budget_tokens=%s\n' "$budget" + for file in captain.md captain-shared.md learnings.md; do + if ! fm_startup_memory_measure_file "$DATA/$file" >/dev/null; then + print_error "$FM_STARTUP_MEMORY_BUDGET_ERROR" + return 2 + fi + bytes=$FM_STARTUP_MEMORY_MEASURE_BYTES + tokens=$FM_STARTUP_MEMORY_MEASURE_TOKENS + presence=$FM_STARTUP_MEMORY_MEASURE_PRESENCE + total=$((total + tokens)) + [ "$file" != captain-shared.md ] || shared_tokens=$tokens + printf 'file=data/%s bytes=%s estimated_tokens=%s status=%s\n' \ + "$file" "$bytes" "$tokens" "$presence" + done + printf 'total_estimated_tokens=%s\n' "$total" + if fm_startup_memory_decimal_le "$total" "$budget"; then + printf 'budget_status=within-budget\n' + else + printf 'budget_status=over-budget\n' + fi + if [ "$role" = secondmate ] \ + && ! fm_startup_memory_decimal_le "$shared_tokens" "$budget"; then + printf 'exception=primary-owned-shared-file-alone-exceeds-budget\n' + fi +} + +case "${1:-}" in + read) + [ "$#" -eq 1 ] || { usage >&2; exit 2; } + read_budget + ;; + report) + [ "$#" -eq 1 ] || { usage >&2; exit 2; } + report + ;; + -h|--help) + usage + ;; + *) + usage >&2 + exit 2 + ;; +esac diff --git a/bin/fm-subagent-pretool-check.sh b/bin/fm-subagent-pretool-check.sh index 9d4334469a8..8edb507218b 100755 --- a/bin/fm-subagent-pretool-check.sh +++ b/bin/fm-subagent-pretool-check.sh @@ -4,11 +4,11 @@ # A firstmate primary that delegates through a harness's own delegation, # scheduling, or background-work tool creates work with no `state/<id>.meta` and # no `data/<id>/brief.md`. Only `bin/fm-spawn.sh` writes that metadata, and -# every firstmate guard keys off it (bin/fm-supervision-lib.sh counts -# `state/*.meta`; bin/fm-turnend-guard.sh exits silently at zero). So such work -# is not merely unsupervised: it makes the whole guard stack structurally inert, -# and it dies with the primary session instead of living in its own backend -# session. +# untracked project work contributes nothing to the in-flight branch of +# bin/fm-supervision-lib.sh or bin/fm-turnend-guard.sh. So such work is not +# merely unsupervised: absent an independent X-mode need, it makes the whole +# guard stack structurally inert, and it dies with the primary session instead +# of living in its own backend session. # # This scoped PreToolUse guard is the shipped mechanism. # Claude primaries should also use an untracked per-home local @@ -65,6 +65,19 @@ DELEGATION_STEMS='agent subagent task workflow cron schedul worktree delegate sp # reason a runaway task cannot be stopped. OBSERVE_ONLY_TOOLS='taskoutput taskstop taskget tasklist cronlist bashoutput killshell' +# Exact lowercase tool names that match a stem above but create no RUNNABLE +# work. These write only the harness's session-local todo list, which has no +# executor: it spawns no agent, allocates no worktree, registers no schedule, +# and starts nothing that could outlive the session or escape a firstmate +# guard. Denying them stops the primary tracking its own plan while granting no +# delegation power, and the deny text would tell it to run bin/fm-brief.sh for a +# todo entry, so the stem match here is a false positive rather than a policy. +# This is a separate list from OBSERVE_ONLY_TOOLS on purpose: these tools WRITE, +# so folding them into a list documented as observe-or-stop would make that +# contract untrue. Both lists are exact-name, never substring, so neither can +# widen by accident. +PLAN_ONLY_TOOLS='taskcreate taskupdate' + TOOL="" TOOL_SET=0 CLAUDE_MODE=0 @@ -139,7 +152,7 @@ case "$TOOL" in mcp__*) exit 0 ;; esac -for allowed in $OBSERVE_ONLY_TOOLS; do +for allowed in $OBSERVE_ONLY_TOOLS $PLAN_ONLY_TOOLS; do [ "$NORMALIZED" != "$allowed" ] || exit 0 done diff --git a/bin/fm-supervise-daemon.sh b/bin/fm-supervise-daemon.sh index 1c5940d211d..30554edbbd4 100755 --- a/bin/fm-supervise-daemon.sh +++ b/bin/fm-supervise-daemon.sh @@ -96,7 +96,7 @@ # (default 300) # FM_HOUSEKEEPING_TICK seconds between housekeeping passes while # the watcher is mid-cycle (default 15) -# FM_BUSY_REGEX OR-ed busy signatures (mirrors fm-watch.sh) +# FM_BUSY_REGEX optional global busy-signature override # FM_COMPOSER_IDLE_RE empty-composer regex applied after dim-ghost # and structural border stripping (default: # bare prompt glyphs plus busy footers) @@ -198,9 +198,8 @@ WEDGE_ALARM_NOTIFIER_PID= # The captain-relevant verb set and the status classifiers (last_status_line, # status_is_captain_relevant, window_to_task, scan_captain_relevant_statuses) now # live in bin/fm-classify-lib.sh, shared with the always-on watcher. -# Composer-empty detection and the tmux busy-footer fallback live in -# bin/fm-tmux-lib.sh (FM_TMUX_BUSY_REGEX_DEFAULT / fm_tmux_composer_state); -# FM_BUSY_REGEX still overrides the fallback busy set here, as before. +# Composer-empty detection and harness-scoped busy-footer matching live in +# bin/fm-tmux-lib.sh; FM_BUSY_REGEX still overrides every fallback here. INJECT_FAIL_SLEEP_DEFAULT=30 INJECT_CONFIRM_RETRIES_DEFAULT=3 INJECT_CONFIRM_SLEEP_DEFAULT=0.5 @@ -549,24 +548,21 @@ mark_escalated_seen() { # <kind> <arg> <state> # (one source of truth with fm-send.sh). These thin wrappers keep the daemon's # call sites and the unit tests stable. # -# pane_input_pending returns 0 (pending) when the cursor line holds real -# unsubmitted text - a human's half-typed line (the return race) or a previous -# injection whose Enter was swallowed. The detector drops dim/faint ghost text and -# strips the harness's composer box borders, so a ghost-only or idle bordered -# claude composer ("│ > … │") is correctly read as empty, not pending (incidents -# afk-invx-i5 and composer-robust). +# pane_input_pending returns 0 unless the composer is positively proven empty. +# This includes real unsubmitted text, ambiguous structure, unreadable state, +# and future verdicts. The detector drops dim/faint ghost text and strips the +# harness's composer box borders, so an aligned ghost-only or idle bordered +# claude composer ("│ > … │") is correctly proven empty. # pane_is_busy / pane_input_pending: BACKEND-AWARE now (previously tmux-only # direct calls). <backend> defaults to tmux when omitted, so every existing # caller/test that passes only <target> is unaffected. Dispatch goes through # bin/fm-backend.sh's generic per-backend primitives (fm_backend_busy_state, # fm_backend_capture, fm_backend_composer_state) rather than hand-rolling a -# case statement here, mirroring the same fallback pattern -# stale_window_is_busy already uses for per-task panes: try the backend's -# native busy-state first, and fall back to the shared regex-over-capture -# reader whenever it does not report "busy" (tmux has no native busy-state -# primitive, so it always takes this fallback path - byte-identical to the -# pre-existing fm_pane_is_busy, since fm_backend_capture's tmux arm runs the -# exact same `tmux capture-pane -p -t <target> -S -40`). +# case statement here, mirroring the fallback order stale_window_is_busy uses +# for per-task panes: try the backend's native busy state first, then match +# captured output. The supervisor pane has no recorded task harness and uses +# the historical combined fallback; stale task panes select the recorded +# harness's verified signature. pane_is_busy() { # <target> [backend] local target=$1 backend=${2:-tmux} bs tail40 bs=$(fm_backend_busy_state "$backend" "$target" 2>/dev/null) @@ -574,22 +570,16 @@ pane_is_busy() { # <target> [backend] busy) return 0 ;; esac tail40=$(fm_backend_capture "$backend" "$target" 40 2>/dev/null) || return 1 - printf '%s' "$tail40" | grep -v '^[[:space:]]*$' | tail -6 \ - | grep -qiE "${FM_BUSY_REGEX:-$FM_TMUX_BUSY_REGEX_DEFAULT}" + printf '%s' "$tail40" | grep -v '^[[:space:]]*$' | tail -12 \ + | fm_busy_lines_match } -# pane_input_pending: the standalone "is there real unsubmitted text" predicate, -# dispatching through fm_backend_composer_state (byte-identical to a direct -# fm_tmux_composer_state call for the default/omitted-backend case). inject_msg -# no longer routes its composer-guard through this boolean: a safe injection -# target must be affirmatively 'empty', and a boolean pending/not-pending check -# cannot distinguish an empty agent composer from a bare dead-shell prompt or an -# unreadable pane (both 'unknown'), so inject_msg reads the full tri-state -# verdict directly. This predicate is retained as the shared pending check and -# as the vehicle for the composer-classifier dispatch regression tests. +# pane_input_pending dispatches through fm_backend_composer_state and treats +# every verdict except exact empty as unsafe. inject_msg reads the full verdict +# directly and applies the same positive-proof boundary. pane_input_pending() { # <target> [backend] local target=$1 backend=${2:-tmux} - [ "$(fm_backend_composer_state "$backend" "$target" 2>/dev/null)" = pending ] + [ "$(fm_backend_composer_state "$backend" "$target" 2>/dev/null)" != empty ] } task_window_backend() { # <window> <state> @@ -599,17 +589,25 @@ task_window_backend() { # <window> <state> fm_backend_of_meta "$meta" } +task_window_harness() { # <window> <state> + local win=$1 state=$2 task meta + task=$(window_to_task "$win" "$state") + meta="$state/$task.meta" + grep '^harness=' "$meta" | cut -d= -f2- || true +} + stale_window_is_busy() { # <window> <state> - local win=$1 state=$2 backend label tail40 bs + local win=$1 state=$2 backend harness label tail40 bs backend=$(task_window_backend "$win" "$state") + harness=$(task_window_harness "$win" "$state") label="fm-$(window_to_task "$win" "$state")" tail40=$(fm_backend_capture "$backend" "$win" 40 "$label" 2>/dev/null) || return 2 bs=$(fm_backend_busy_state "$backend" "$win" 2>/dev/null) case "$bs" in busy) return 0 ;; esac - printf '%s' "$tail40" | grep -v '^[[:space:]]*$' | tail -6 \ - | grep -qiE "${FM_BUSY_REGEX:-$FM_TMUX_BUSY_REGEX_DEFAULT}" + printf '%s' "$tail40" | grep -v '^[[:space:]]*$' | tail -12 \ + | fm_busy_lines_match "$harness" } escalate_add() { # <state> <distilled-item> @@ -1183,7 +1181,7 @@ is_wake_reason() { # <reason> # --- dispatch one wake reason to self-handle or escalate -------------------- # Side effects: logging, marker records, escalation buffer appends. handle_wake() { # <reason> <state> - local reason=$1 state=$2 decision action distilled task last + local reason=$1 state=$2 decision action distilled task last stale_detail local kind="" arg="" if should_force_self "$reason"; then log "wake force-self (FM_INJECT_SKIP): $reason" @@ -1192,8 +1190,13 @@ handle_wake() { # <reason> <state> case "$reason" in signal:*) kind=signal; arg="${reason#signal: }" decision=$(classify_signal "$arg" "$state") ;; - stale:*) kind=stale; arg="${reason#stale: }" - decision=$(classify_stale "$arg" "$state") ;; + stale:*) kind=stale; arg="${reason#stale: }"; stale_detail="${arg#"$arg"}" + case "$arg" in *" ("*) stale_detail="${arg#*" ("}"; arg="${arg%% \(*}" ;; esac + decision=$(classify_stale "$arg" "$state") + case "$stale_detail" in + idle\ *s,\ possible\ wedge,\ escalation\ *) + decision="escalate|${reason#stale: }" ;; + esac ;; check:*) decision=$(classify_check "$reason") ;; heartbeat|heartbeat:*) decision=$(classify_heartbeat) ;; *) decision=$(classify_unknown "$reason") ;; diff --git a/bin/fm-supervision-instructions.sh b/bin/fm-supervision-instructions.sh index e4529b51dda..6cd87699b0b 100755 --- a/bin/fm-supervision-instructions.sh +++ b/bin/fm-supervision-instructions.sh @@ -82,6 +82,7 @@ fi case "$HARNESS" in claude|codex|opencode|pi|grok) SNIPPET="$DOC_DIR/$HARNESS.md" ;; + pi-signed) SNIPPET="$DOC_DIR/pi.md" ;; *) HARNESS=unknown; SNIPPET="$DOC_DIR/unknown.md" ;; esac [ -f "$SNIPPET" ] || SNIPPET="$DOC_DIR/unknown.md" @@ -139,7 +140,7 @@ repair_line() { codex) printf '%s%s%s%s\n' "$prefix" 'repair missing watcher supervision with a foreground checkpoint: bin/fm-watch-checkpoint.sh --seconds ' "$checkpoint_seconds" '.' ;; - pi) + pi|pi-signed) printf '%s%s%s%s%s%s\n' "$prefix" 'repair a missing or failed watcher cycle with the Pi tool fm_watch_arm_pi, or restart Pi with -e ' "$pi_turnend_ext" ' -e ' "$pi_ext" ' if the extensions are not loaded.' ;; opencode) @@ -157,12 +158,12 @@ repair_line() { ordinary_wake_line() { case "$HARNESS" in claude) - printf '%s\n' '- Ordinary wake: re-arm exactly one bin/fm-watch-arm.sh Claude Code background task as directed below.' + printf '%s\n' '- Ordinary wake: the Stop-owned auto-arm (bin/fm-claude-stop-autoarm.sh) already owns watcher continuity; drain and handle the wake, and do not arm another cycle yourself.' ;; codex) printf '%s\n' '- Ordinary wake: take the next foreground bin/fm-watch-checkpoint.sh checkpoint as directed below.' ;; - pi) + pi|pi-signed) printf '%s\n' '- Ordinary wake: the Pi extension already owns watcher continuity; do not arm another cycle.' ;; opencode) diff --git a/bin/fm-supervision-lib.sh b/bin/fm-supervision-lib.sh index c89747a94fc..1930700d2af 100644 --- a/bin/fm-supervision-lib.sh +++ b/bin/fm-supervision-lib.sh @@ -2,12 +2,14 @@ # Shared "supervision missing" predicate. # Usage: . bin/fm-supervision-lib.sh # -# True exactly when a firstmate home has in-flight work (a state/<id>.meta -# exists) but no watcher has a fresh liveness beacon (state/.last-watcher-beat, -# touched every poll cycle, within the grace window). bin/fm-guard.sh uses this -# grace-based warning predicate directly; bin/fm-turnend-guard.sh uses the status -# fields here for its banner but performs its end-of-turn block decision with the -# live watcher lock check in bin/fm-wake-lib.sh. +# Reports whether a firstmate home needs supervision because it has in-flight +# work (a state/<id>.meta exists) or an X-mode relay poll +# (state/x-watch.check.sh), and whether its watcher has a fresh liveness beacon +# (state/.last-watcher-beat, touched every poll cycle, within the grace window). +# bin/fm-guard.sh keeps its task-specific grace-based warning predicate; +# bin/fm-turnend-guard.sh uses the status fields here for its banner but performs +# its end-of-turn block decision with the live watcher lock check in +# bin/fm-wake-lib.sh. # Portable mtime; Linux stat lacks -f, macOS stat lacks -c. fm_sup_stat_mtime() { @@ -21,6 +23,7 @@ fm_sup_stat_mtime() { # fm_supervision_status <state-dir> [grace-seconds] # Populates, for the state dir at $1: # FM_SUP_IN_FLIGHT count of state/*.meta (in-flight tasks) +# FM_SUP_NEEDED true/false - in-flight work or an X-mode relay poll # FM_SUP_WATCHER_FRESH true/false - a watcher beacon within the grace window # FM_SUP_BEACON_DESC human-readable beacon age, for banners ("never" if absent) # FM_SUP_QUEUE_PENDING true/false - state/.wake-queue has unread records @@ -29,6 +32,7 @@ fm_sup_stat_mtime() { fm_supervision_status() { local state=$1 grace=${2:-${FM_GUARD_GRACE:-300}} meta beat m age FM_SUP_IN_FLIGHT=0 + FM_SUP_NEEDED=false FM_SUP_WATCHER_FRESH=false FM_SUP_BEACON_DESC=never FM_SUP_QUEUE_PENDING=false @@ -37,6 +41,9 @@ fm_supervision_status() { [ -e "$meta" ] || continue FM_SUP_IN_FLIGHT=$((FM_SUP_IN_FLIGHT + 1)) done + if [ "$FM_SUP_IN_FLIGHT" -gt 0 ] || [ -f "$state/x-watch.check.sh" ]; then + FM_SUP_NEEDED=true + fi beat="$state/.last-watcher-beat" if [ -e "$beat" ]; then @@ -56,6 +63,14 @@ fm_supervision_status() { return 0 } +# fm_supervision_needed <state-dir> [grace-seconds] +# Exit 0 (true) exactly when in-flight work or an X-mode relay poll needs a +# watcher. Exit 1 (false) for an idle home. +fm_supervision_needed() { + fm_supervision_status "$@" + [ "$FM_SUP_NEEDED" = true ] +} + # fm_supervision_unhealthy <state-dir> [grace-seconds] # Exit 0 (true) exactly in the dangerous state: in-flight work exists and no # watcher has a fresh beacon. Exit 1 (false) otherwise, including zero in-flight. diff --git a/bin/fm-teardown.sh b/bin/fm-teardown.sh index 63f830c3add..6164ebdd785 100755 --- a/bin/fm-teardown.sh +++ b/bin/fm-teardown.sh @@ -116,18 +116,20 @@ FORCE=${2:-} # down a worktree (see bin/fm-gate-refuse-lib.sh). fm_refuse_if_gate_agent FM_LOCK_LOG_PREFIX=teardown -"$FM_ROOT/bin/fm-guard.sh" || true META="$STATE/$ID.meta" [ -f "$META" ] || { echo "error: no meta for task $ID at $META" >&2; exit 1; } -WT=$(grep '^worktree=' "$META" | cut -d= -f2-) -T=$(grep '^window=' "$META" | cut -d= -f2-) -PROJ=$(grep '^project=' "$META" | cut -d= -f2-) -BACKEND=$(fm_backend_of_meta "$META") -if [ "$BACKEND" = orca ]; then - T_ORCA=$(grep '^terminal=' "$META" | tail -1 | cut -d= -f2- || true) - [ -n "$T_ORCA" ] && T=$T_ORCA -fi +# This is the first cleanup authorization check. It is metadata-only and must +# complete before fm-guard, a backend command, file removal, branch deletion, +# worktree return, registry change, or process termination can run. +fm_backend_validate_task_endpoint "$META" "$ID" || exit 1 +BACKEND=$FM_BACKEND_VALIDATED_BACKEND +T=$FM_BACKEND_VALIDATED_TARGET +WT=$(fm_meta_get "$META" worktree) +PROJ=$(fm_meta_get "$META" project) +T_ORCA= +[ "$BACKEND" != orca ] || T_ORCA=$T +"$FM_ROOT/bin/fm-guard.sh" || true HOME_PATH=$(grep '^home=' "$META" | cut -d= -f2- || true) PR_URL=$(grep '^pr=' "$META" | tail -1 | cut -d= -f2- || true) # tasktmp is recorded by fm-spawn for tasks that set up a per-task temp root @@ -196,6 +198,14 @@ remove_grok_turnend_auth() { rm -f "$hooks_dir/$token" } +remove_kimi_turnend_auth() { + local state_dir=$1 id=$2 token hooks_dir + token=$(cat "$state_dir/$id.kimi-turnend-token" 2>/dev/null || true) + case "$token" in ''|*[!A-Za-z0-9._-]*) return 0 ;; esac + hooks_dir="$HOME/.kimi-code/fm-turn-end.d" + rm -f "$hooks_dir/$token" +} + validate_pr_poll_cleanup() { local state_dir=$1 id=$2 quarantine state_device artifact has_artifact=0 fm_task_id_path_safe "$id" || return 0 @@ -672,7 +682,7 @@ validate_worktree_teardown_safety() { echo "Restore the git index state, or get the captain's explicit OK to discard, then --force." >&2 return 1 fi - dirty=$(printf '%s\n' "$dirty_raw" | grep -vE '^\?\? (\.claude/|\.fm-grok-turnend$)' | head -1 || true) + dirty=$(printf '%s\n' "$dirty_raw" | grep -vE '^\?\? (\.claude/|\.fm-(grok|kimi)-turnend$)' | head -1 || true) if ! unpushed_raw=$(git -C "$WT" log --oneline HEAD --not --remotes -- 2>/dev/null); then if worktree_safety_blocked_by_lock "commits not on a remote"; then @@ -936,6 +946,7 @@ validate_firstmate_home_children_removal() { for child_meta in "$sub_state"/*.meta; do [ -e "$child_meta" ] || continue child_id=$(basename "$child_meta" .meta) + fm_backend_validate_task_endpoint "$child_meta" "$child_id" || return 1 validate_pr_poll_cleanup "$sub_state" "$child_id" || return 1 child_wt=$(meta_value "$child_meta" worktree) child_kind=$(meta_value "$child_meta" kind) @@ -1002,12 +1013,14 @@ cleanup_firstmate_home_children() { elif [ "$child_backend" = orca ]; then if [ -n "$child_wt" ] && [ -d "$child_wt" ]; then validate_child_worktree_for_removal "$child_wt" "$child_proj" >/dev/null || return 1 - rm -f "$child_wt/.claude/settings.local.json" "$child_wt/.opencode/plugins/fm-turn-end.js" "$child_wt/.fm-grok-turnend" + rm -f "$child_wt/.claude/settings.local.json" "$child_wt/.opencode/plugins/fm-turn-end.js" \ + "$child_wt/.fm-grok-turnend" "$child_wt/.fm-kimi-turnend" fi fm_backend_remove_worktree "$child_backend" "$child_orca_worktree_id" || return 1 elif [ -n "$child_wt" ] && [ -d "$child_wt" ]; then validate_child_worktree_for_removal "$child_wt" "$child_proj" >/dev/null || return 1 - rm -f "$child_wt/.claude/settings.local.json" "$child_wt/.opencode/plugins/fm-turn-end.js" "$child_wt/.fm-grok-turnend" + rm -f "$child_wt/.claude/settings.local.json" "$child_wt/.opencode/plugins/fm-turn-end.js" \ + "$child_wt/.fm-grok-turnend" "$child_wt/.fm-kimi-turnend" if [ -n "$child_proj" ] && [ -d "$child_proj" ] && command -v treehouse >/dev/null 2>&1; then if teardown_treehouse_return "$child_wt" "$child_proj" "child worktree"; then : @@ -1023,8 +1036,11 @@ cleanup_firstmate_home_children() { fi fi remove_grok_turnend_auth "$sub_state" "$child_id" + remove_kimi_turnend_auth "$sub_state" "$child_id" remove_pr_poll_artifacts "$sub_state" "$child_id" || return 1 - rm -f "$sub_state/$child_id.status" "$sub_state/$child_id.turn-ended" "$sub_state/$child_id.meta" "$sub_state/$child_id.pi-ext.ts" "$sub_state/$child_id.grok-turnend-token" + rm -f "$sub_state/$child_id.status" "$sub_state/$child_id.turn-ended" \ + "$sub_state/$child_id.meta" "$sub_state/$child_id.pi-ext.ts" \ + "$sub_state/$child_id.grok-turnend-token" "$sub_state/$child_id.kimi-turnend-token" done } @@ -1114,7 +1130,8 @@ if [ "$BACKEND" = orca ] && [ "$KIND" != secondmate ]; then git -C "$WT" branch -D "$branch" >/dev/null 2>&1 || true fi fi - rm -f "$WT/.claude/settings.local.json" "$WT/.opencode/plugins/fm-turn-end.js" "$WT/.fm-grok-turnend" + rm -f "$WT/.claude/settings.local.json" "$WT/.opencode/plugins/fm-turn-end.js" \ + "$WT/.fm-grok-turnend" "$WT/.fm-kimi-turnend" fi [ -z "$T_ORCA" ] || fm_backend_kill "$BACKEND" "$T" "$(meta_value "$META" zellij_tab_id)" "fm-$ID" 2>/dev/null || true fm_backend_remove_worktree "$BACKEND" "$ORCA_WORKTREE_ID" @@ -1126,7 +1143,8 @@ elif [ -d "$WT" ] && [ "$KIND" != secondmate ]; then fi fi # Remove our hook file so a reused pool worktree cannot fire signals for a dead task. - rm -f "$WT/.claude/settings.local.json" "$WT/.opencode/plugins/fm-turn-end.js" "$WT/.fm-grok-turnend" + rm -f "$WT/.claude/settings.local.json" "$WT/.opencode/plugins/fm-turn-end.js" \ + "$WT/.fm-grok-turnend" "$WT/.fm-kimi-turnend" # Kills remaining processes in the worktree (including the agent), resets, returns # to pool. treehouse resolves the pool from the working directory, so run it from # the project. teardown_treehouse_return tolerates transient and stale git locks @@ -1205,12 +1223,15 @@ if [ "$KIND" = secondmate ]; then remove_secondmate_registry_entry "$ID" fi remove_grok_turnend_auth "$STATE" "$ID" +remove_kimi_turnend_auth "$STATE" "$ID" fm_backend_clear_transition "$BACKEND" "$STATE" "$T" || true # Remove the per-task temp root (/tmp/fm-<id>/, incl. its gotmp/) recorded by spawn. # Read before the state-file rm below; empty (pre-fix tasks without tasktmp=) is a no-op. [ -n "$TASK_TMP" ] && rm -rf "$TASK_TMP" remove_pr_poll_artifacts "$STATE" "$ID" || exit 1 -rm -f "$STATE/$ID.status" "$STATE/$ID.turn-ended" "$STATE/$ID.meta" "$STATE/$ID.pi-ext.ts" "$STATE/$ID.grok-turnend-token" +rm -f "$STATE/$ID.status" "$STATE/$ID.turn-ended" "$STATE/$ID.meta" \ + "$STATE/$ID.pi-ext.ts" "$STATE/$ID.grok-turnend-token" \ + "$STATE/$ID.kimi-turnend-token" if [ "$KIND" != scout ] && [ "$KIND" != secondmate ] && [ "$MODE" != local-only ]; then "$FM_ROOT/bin/fm-fleet-sync.sh" "$PROJ" || true fi diff --git a/bin/fm-test-isolation-proof.sh b/bin/fm-test-isolation-proof.sh index c9e468d0dfc..f84f8ed09ef 100755 --- a/bin/fm-test-isolation-proof.sh +++ b/bin/fm-test-isolation-proof.sh @@ -87,9 +87,6 @@ now_ms() { # evidence; do not re-add a basename without clearing its reason. exclusion_reason() { case "$1" in - fm-continuity-pretool-check.test.sh) - printf '%s\n' 'background sleep 300 holder process for live-lock identity; process-leak risk under concurrent load' - ;; fm-test-isolation-proof.test.sh) printf '%s\n' 'isolation-proof harness contract itself; must not re-enter concurrent matrix' ;; @@ -108,6 +105,9 @@ exclusion_reason() { fm-teardown.test.sh) printf '%s\n' 'landed-work + lock-race teardown matrix; keep serial with forge/git stress peers' ;; + fm-herdr-session-cleanup.test.sh) + printf '%s\n' 'session-start task/presentation lock matrix; keep serial until dedicated concurrent proof' + ;; fm-daemon.test.sh|fm-guard-stale-banner.test.sh|fm-pi-watch-extension.test.sh|\ fm-supervision-events.test.sh|fm-turnend-guard.test.sh|fm-wake-daemon-lifecycle-e2e.test.sh|\ fm-wake-queue.test.sh|fm-watch-checkpoint.test.sh|fm-watch-triage.test.sh|\ @@ -118,7 +118,7 @@ exclusion_reason() { fm-afk-launch.test.sh) printf '%s\n' 'AFK lifecycle / inject path; exclusive daemon and pane control' ;; - fm-afk-pi-herdr-return-e2e.test.sh|fm-claude-continuity-live-e2e.test.sh|\ + fm-afk-pi-herdr-return-e2e.test.sh|\ fm-codex-continuity-live-e2e.test.sh|fm-grok-continuity-live-e2e.test.sh|\ fm-opencode-primary-live-e2e.test.sh|fm-pi-primary-live-e2e.test.sh|\ fm-send-secondmate-marker-herdr-e2e.test.sh) @@ -127,7 +127,7 @@ exclusion_reason() { fm-backend-autodetect-smoke.test.sh|fm-backend-herdr-eventwait-smoke.test.sh|\ fm-backend-herdr-presentation-e2e.test.sh|fm-backend-herdr-prune-safety-e2e.test.sh|\ fm-backend-herdr-respawn-idem-e2e.test.sh|fm-backend-herdr-smoke.test.sh|\ - fm-backend-herdr-workspace-per-home-e2e.test.sh) + fm-backend-herdr-workspace-per-home-e2e.test.sh|fm-herdr-session-cleanup-e2e.test.sh) printf '%s\n' 'real Herdr-gated; Herdr lane is a later phase' ;; fm-backend-cmux.test.sh|fm-backend-cmux-smoke.test.sh) @@ -152,20 +152,15 @@ list_parallel_candidates() { tests/fm-arm-pretool-check.test.sh tests/fm-backend-herdr.test.sh tests/fm-brief.test.sh -tests/fm-captain-translation-contract.test.sh tests/fm-cd-pretool-check.test.sh tests/fm-composer-ghost.test.sh tests/fm-composer-lib.test.sh tests/fm-crew-state.test.sh tests/fm-decision-hold-lifecycle.test.sh -tests/fm-dispatch-select.test.sh tests/fm-ensure-agents-md.test.sh tests/fm-grok-harness.test.sh tests/fm-herdr-lab.test.sh -tests/fm-instruction-owners.test.sh tests/fm-lint.test.sh -tests/fm-nm-test-contract.test.sh -tests/fm-no-mistakes-ownership.test.sh tests/fm-pi-primary-types.test.sh tests/fm-pr-merge.test.sh tests/fm-review-diff.test.sh @@ -173,7 +168,6 @@ tests/fm-send-popup-settle.test.sh tests/fm-send-settle.test.sh tests/fm-send-strict.test.sh tests/fm-spawn-batch.test.sh -tests/fm-stow-contract.test.sh tests/fm-supervision-instructions.test.sh tests/fm-test-run.test.sh tests/fm-tmux-submit-busy.test.sh @@ -191,7 +185,6 @@ list_exclusions_for_report() { printf '%s\t%s\n' "$base" "$reason" fi done <<'EOF' -fm-continuity-pretool-check.test.sh fm-test-isolation-proof.test.sh fm-backend-tmux-smoke.test.sh fm-backend.test.sh diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index f1c6b77a065..6c4b4aae463 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -38,7 +38,7 @@ # silently pass as a gate skip. # --jobs N run the selected scripts with up to N concurrent workers. # Default is 1 (serial). N>1 is allowed only when every -# selected script is in the Phase 2 proven-isolated set +# selected script is in the proven-isolated set # (bin/fm-test-isolation-proof.sh --list). Cap is 8. Stateful # families never schedule under --jobs. # -h, --help print this header @@ -118,14 +118,14 @@ now_ms() { family_for_basename() { case "$1" in fm-arm-pretool-check.test.sh|fm-ask-user-authority.test.sh|fm-brief.test.sh|\ - fm-calm-pi-extension.test.sh|fm-captain-translation-contract.test.sh|fm-cd-pretool-check.test.sh|\ + fm-calm-pi-extension.test.sh|fm-cd-pretool-check.test.sh|\ fm-composer-ghost.test.sh|fm-composer-lib.test.sh|\ - fm-continuity-pretool-check.test.sh|fm-crew-state.test.sh|fm-decision-hold-lifecycle.test.sh|\ - fm-dispatch-select.test.sh|fm-ensure-agents-md.test.sh|fm-grok-harness.test.sh|\ - fm-herdr-lab.test.sh|fm-instruction-owners.test.sh|fm-lint.test.sh|fm-secrets-check.test.sh|\ - fm-install-herdr.test.sh|fm-nm-test-contract.test.sh|fm-no-mistakes-ownership.test.sh|\ + fm-crew-state.test.sh|fm-decision-hold-lifecycle.test.sh|\ + fm-documentation-audiences.test.sh|fm-ensure-agents-md.test.sh|fm-grok-harness.test.sh|\ + fm-herdr-lab.test.sh|fm-kimi-harness.test.sh|fm-lint.test.sh|fm-secrets-check.test.sh|\ + fm-captain-translation-contract.test.sh|\ fm-operational-input.test.sh|fm-pi-primary-types.test.sh|\ - fm-send-popup-settle.test.sh|fm-send-settle.test.sh|fm-stow-contract.test.sh|\ + fm-send-popup-settle.test.sh|fm-send-settle.test.sh|\ fm-subagent-pretool-check.test.sh|\ fm-supervision-instructions.test.sh|fm-tmux-submit-busy.test.sh|fm-transition-lib.test.sh|\ fm-test-run.test.sh|fm-test-isolation-proof.test.sh|fm-toolchain-mirror.test.sh) @@ -140,11 +140,13 @@ family_for_basename() { fm-afk-inject-herdr-e2e.test.sh|fm-afk-launch.test.sh|fm-backend-autodetect-smoke.test.sh|\ fm-backend-herdr-eventwait-smoke.test.sh|fm-backend-herdr-presentation-e2e.test.sh|\ fm-backend-herdr-prune-safety-e2e.test.sh|fm-backend-herdr-respawn-idem-e2e.test.sh|\ + fm-herdr-session-cleanup-e2e.test.sh|\ fm-backend-herdr-smoke.test.sh|fm-backend-herdr-workspace-per-home-e2e.test.sh) printf '%s\n' real-herdr-gated ;; fm-backlog-handoff.test.sh|fm-secondmate-harness.test.sh|fm-secondmate-lifecycle-e2e.test.sh|\ fm-secondmate-liveness.test.sh|fm-secondmate-safety.test.sh|fm-secondmate-sync.test.sh|\ + fm-startup-memory-budget.test.sh|\ fm-send-secondmate-marker.test.sh|fm-shared-captain-inheritance.test.sh) printf '%s\n' secondmate ;; @@ -153,15 +155,16 @@ family_for_basename() { fm-update.test.sh) printf '%s\n' session-bootstrap ;; - fm-afk-pi-herdr-return-e2e.test.sh|fm-claude-continuity-live-e2e.test.sh|\ + fm-afk-pi-herdr-return-e2e.test.sh|\ fm-codex-continuity-live-e2e.test.sh|fm-grok-continuity-live-e2e.test.sh|\ - fm-opencode-primary-live-e2e.test.sh|fm-pi-primary-live-e2e.test.sh|\ + fm-grok-stop-live-e2e.test.sh|fm-opencode-primary-live-e2e.test.sh|fm-pi-primary-live-e2e.test.sh|\ fm-send-secondmate-marker-herdr-e2e.test.sh) printf '%s\n' live-harness-optin ;; fm-backend-herdr.test.sh|fm-backend-tmux-smoke.test.sh|fm-backend.test.sh|\ - fm-send-strict.test.sh|fm-spawn-batch.test.sh|fm-spawn-dispatch-profile.test.sh|\ - fm-spawn-worktree-settle.test.sh) + fm-herdr-session-cleanup.test.sh|fm-send-strict.test.sh|fm-spawn-batch.test.sh|\ + fm-spawn-dispatch-profile.test.sh|fm-spawn-worktree-settle.test.sh|\ + fm-teardown-endpoint-safety.test.sh) printf '%s\n' backend-dispatch ;; fm-pr-check-security.test.sh|fm-pr-merge.test.sh|fm-review-diff.test.sh|\ @@ -227,7 +230,7 @@ real-herdr-gated EOF } -# Exact Phase 2 proven-isolated candidate set (same paths as +# Exact proven-isolated candidate set (same paths as # bin/fm-test-isolation-proof.sh --list). Do not expand without a new concurrent # isolation proof archive. list_proven_isolated() { @@ -235,20 +238,15 @@ list_proven_isolated() { tests/fm-arm-pretool-check.test.sh tests/fm-backend-herdr.test.sh tests/fm-brief.test.sh -tests/fm-captain-translation-contract.test.sh tests/fm-cd-pretool-check.test.sh tests/fm-composer-ghost.test.sh tests/fm-composer-lib.test.sh tests/fm-crew-state.test.sh tests/fm-decision-hold-lifecycle.test.sh -tests/fm-dispatch-select.test.sh tests/fm-ensure-agents-md.test.sh tests/fm-grok-harness.test.sh tests/fm-herdr-lab.test.sh -tests/fm-instruction-owners.test.sh tests/fm-lint.test.sh -tests/fm-nm-test-contract.test.sh -tests/fm-no-mistakes-ownership.test.sh tests/fm-pi-primary-types.test.sh tests/fm-pr-merge.test.sh tests/fm-review-diff.test.sh @@ -256,7 +254,6 @@ tests/fm-send-popup-settle.test.sh tests/fm-send-settle.test.sh tests/fm-send-strict.test.sh tests/fm-spawn-batch.test.sh -tests/fm-stow-contract.test.sh tests/fm-supervision-instructions.test.sh tests/fm-test-run.test.sh tests/fm-tmux-submit-busy.test.sh @@ -265,48 +262,41 @@ tests/fm-x-mode.test.sh EOF } -# Portable parallel shard 1: LPT balance of the proven-isolated set using -# Phase 1 serial duration averages from CI timing artifacts on main after -# #825/#832/#834 (docs/fm-test-portable-shards.md). Execution order is longest -# first so wall-clock stays near the balanced sum. +# Portable parallel shard 1: LPT balance of the proven-isolated set using the +# current concurrent-proof durations in docs/fm-test-isolation-proof.json. +# Execution order is longest first so wall-clock stays near the balanced sum. list_portable_parallel_1() { cat <<'EOF' -tests/fm-arm-pretool-check.test.sh +tests/fm-x-mode.test.sh tests/fm-cd-pretool-check.test.sh -tests/fm-backend-herdr.test.sh -tests/fm-pr-merge.test.sh +tests/fm-decision-hold-lifecycle.test.sh tests/fm-test-run.test.sh -tests/fm-send-popup-settle.test.sh +tests/fm-composer-ghost.test.sh +tests/fm-grok-harness.test.sh +tests/fm-lint.test.sh +tests/fm-pi-primary-types.test.sh tests/fm-review-diff.test.sh tests/fm-brief.test.sh -tests/fm-dispatch-select.test.sh -tests/fm-ensure-agents-md.test.sh -tests/fm-instruction-owners.test.sh -tests/fm-pi-primary-types.test.sh tests/fm-transition-lib.test.sh -tests/fm-composer-lib.test.sh -tests/fm-stow-contract.test.sh EOF } # Portable parallel shard 2: the complementary LPT half of the proven set. list_portable_parallel_2() { cat <<'EOF' -tests/fm-decision-hold-lifecycle.test.sh -tests/fm-x-mode.test.sh -tests/fm-herdr-lab.test.sh +tests/fm-backend-herdr.test.sh +tests/fm-arm-pretool-check.test.sh tests/fm-crew-state.test.sh -tests/fm-grok-harness.test.sh -tests/fm-spawn-batch.test.sh -tests/fm-send-strict.test.sh +tests/fm-herdr-lab.test.sh +tests/fm-pr-merge.test.sh +tests/fm-send-popup-settle.test.sh tests/fm-tmux-submit-busy.test.sh -tests/fm-composer-ghost.test.sh tests/fm-send-settle.test.sh +tests/fm-send-strict.test.sh +tests/fm-spawn-batch.test.sh tests/fm-supervision-instructions.test.sh -tests/fm-lint.test.sh -tests/fm-nm-test-contract.test.sh -tests/fm-captain-translation-contract.test.sh -tests/fm-no-mistakes-ownership.test.sh +tests/fm-ensure-agents-md.test.sh +tests/fm-composer-lib.test.sh EOF } @@ -621,6 +611,11 @@ families_for_changed_path() { printf '%s\n' backend-dispatch printf '%s\n' pure-contract-unit ;; + bin/fm-herdr-session-cleanup.sh) + printf '%s\n' session-bootstrap + printf '%s\n' real-herdr-gated + printf '%s\n' backend-dispatch + ;; bin/backends/zellij*|tests/zellij-test-safety.sh) printf '%s\n' zellij printf '%s\n' backend-dispatch @@ -651,6 +646,10 @@ families_for_changed_path() { printf '%s\n' live-harness-optin printf '%s\n' afk ;; + bin/fm-startup-memory-budget.sh|bin/fm-startup-memory-budget-lib.sh) + printf '%s\n' secondmate + printf '%s\n' session-bootstrap + ;; bin/fm-secondmate*|bin/fm-home-seed.sh|bin/fm-backlog-handoff.sh|\ bin/fm-config-inherit-lib.sh|bin/fm-config-push.sh|bin/fm-shared*) printf '%s\n' secondmate @@ -664,7 +663,7 @@ families_for_changed_path() { bin/fm-x-*|bin/fm-check*) printf '%s\n' pr-forge ;; - bin/fm-spawn.sh|bin/fm-send.sh|bin/fm-dispatch-select.sh|bin/fm-harness.sh|\ + bin/fm-spawn.sh|bin/fm-send.sh|bin/fm-harness.sh|\ bin/fm-peek.sh|bin/fm-composer*) printf '%s\n' backend-dispatch printf '%s\n' pure-contract-unit @@ -689,6 +688,9 @@ families_for_changed_path() { bin/fm-ff-lib.sh|bin/fm-gotmp*|bin/*pretool*) printf '%s\n' pure-contract-unit ;; + .agents/skills/*/SKILL.md) + printf '%s\n' pure-contract-unit + ;; .github/workflows/ci.yml|.no-mistakes.yaml) printf '%s\n' pure-contract-unit printf '%s\n' real-herdr-gated diff --git a/bin/fm-tmux-lib.sh b/bin/fm-tmux-lib.sh index 889def95bff..cf8c3f7fa5e 100755 --- a/bin/fm-tmux-lib.sh +++ b/bin/fm-tmux-lib.sh @@ -19,15 +19,15 @@ # "suggestion" as dim/faint text inside an otherwise-empty composer. A plain # capture cannot tell it apart from text a human typed, so the old reader saw an # idle pane as holding pending input and the daemon deferred injection / firstmate -# misjudged the pane. The composer reader now captures just the cursor line WITH -# ANSI styling (tmux capture-pane -e) and extracts the real typed content with the -# shared, fleet-wide fm_composer_strip_ghost (bin/fm-composer-lib.sh), which drops -# every de-emphasised run - dim/faint (SGR 2) AND a dark/muted truecolor -# foreground - so ghost/placeholder text never counts as real input. The styled -# capture is consumed internally and parsed into a boolean here; it is NEVER -# surfaced (fm-peek and every human/LLM-facing path stay plain), and only the -# single composer row is captured, so no escape-laden pane bulk is produced. This -# is harness-generic: any harness that de-emphasises placeholder/ghost text +# misjudged the pane. The composer reader now captures the visible pane WITH ANSI +# styling (tmux capture-pane -e), locates a bordered composer structurally, and +# extracts the real typed content from every row with the shared, fleet-wide +# fm_composer_strip_ghost (bin/fm-composer-lib.sh), which drops every +# de-emphasised run - dim/faint (SGR 2) AND a dark/muted truecolor foreground - +# so ghost/placeholder text never counts as real input. The styled capture is +# consumed internally and parsed into a boolean here; it is NEVER surfaced +# (fm-peek and every human/LLM-facing path stay plain). This is harness-generic: +# any harness that de-emphasises placeholder/ghost text # benefits, and the herdr adapter routes through the same owner (task # afk-herdr-false-pending), so the two backends cannot drift. # @@ -45,9 +45,9 @@ # as a known gap in `docs/herdr-backend.md` rather than patched here, so the # tmux adapter does not paper over a herdr-specific shape. # -# Per-harness override: FM_COMPOSER_IDLE_RE matches an empty composer after -# ghost and structural border stripping. FM_BUSY_REGEX overrides the busy -# footer set (mirrors fm-watch.sh / the daemon). +# Overrides: FM_COMPOSER_IDLE_RE matches an empty composer after ghost and +# structural border stripping. FM_BUSY_REGEX globally overrides harness-scoped +# busy-footer matching (mirrors fm-watch.sh / the daemon). # # All functions are `set -u` and `set -e` safe (guarded tmux calls, explicit # returns) so they can be sourced into either context. @@ -61,9 +61,52 @@ . "$(dirname -- "${BASH_SOURCE[0]}")/fm-composer-lib.sh" # Busy footers per harness (mirror fm-watch.sh). claude/codex: "esc to -# interrupt"; opencode: "esc interrupt"; pi: "Working..."; grok: "Ctrl+c:cancel" -# (grok's mid-turn cancel hint, shown iff a turn is running - verified grok 0.2.73). +# interrupt"; opencode: "esc interrupt"; pi: "Working..."; grok: "Ctrl+c:cancel". +# Claude's current spinner has a rotating glyph and word, but every active-turn +# line has an ellipsis followed by a parenthesized elapsed duration. Keep this +# signature separate from the shared default because that shape is not generic +# enough to classify arbitrary harness output safely. +# Kimi's anchored moon-phase spinner is separate because bare moon glyphs in +# ordinary output must not classify another harness as busy. Leading whitespace is +# OPTIONAL; whitespace on both sides of the separator is REQUIRED because every +# captured spinner row had it. A zero-whitespace form has NEVER been observed and +# is deliberately not matched. The line end is intentionally unanchored because +# rotating tip text follows and is not required to be present. The idle status +# bar's lowercase `thinking` label and independently rotating tip text are not +# busy signals on their own. +# The full moon-phase set remains locale- and emoji-font-sensitive because Kimi +# exposes no stable ASCII busy token. FM_TMUX_BUSY_REGEX_DEFAULT='esc (to )?interrupt|Working\.\.\.|Ctrl\+c:cancel' +FM_TMUX_CLAUDE_BUSY_REGEX_DEFAULT='esc to interrupt|…[[:space:]]+\([0-9]+[smh]' +FM_TMUX_CODEX_BUSY_REGEX_DEFAULT='esc to interrupt' +FM_TMUX_OPENCODE_BUSY_REGEX_DEFAULT='esc interrupt' +FM_TMUX_PI_BUSY_REGEX_DEFAULT='Working\.\.\.' +FM_TMUX_GROK_BUSY_REGEX_DEFAULT='Ctrl\+c:cancel' +FM_TMUX_KIMI_BUSY_REGEX_DEFAULT='^[[:space:]]*(🌑|🌒|🌓|🌔|🌕|🌖|🌗|🌘)[[:space:]]+·[[:space:]]+' + +fm_busy_lines_match() { # [harness] + local harness=${1:-} lines regex + IFS= read -r -d '' lines || true + if [ -n "${FM_BUSY_REGEX:-}" ]; then + regex=$FM_BUSY_REGEX + else + case "$harness" in + claude) regex=$FM_TMUX_CLAUDE_BUSY_REGEX_DEFAULT ;; + codex) regex=$FM_TMUX_CODEX_BUSY_REGEX_DEFAULT ;; + opencode) regex=$FM_TMUX_OPENCODE_BUSY_REGEX_DEFAULT ;; + pi|pi-signed) regex=$FM_TMUX_PI_BUSY_REGEX_DEFAULT ;; + grok) regex=$FM_TMUX_GROK_BUSY_REGEX_DEFAULT ;; + kimi) regex=$FM_TMUX_KIMI_BUSY_REGEX_DEFAULT ;; + '') regex=$FM_TMUX_BUSY_REGEX_DEFAULT ;; + *) + # A supplied harness must never borrow another harness's signature. + # Register its verified signature explicitly before classifying it busy. + regex= + ;; + esac + fi + [ -n "$regex" ] && printf '%s' "$lines" | grep -qiE "$regex" +} # fm_tmux_strip_ghost: thin adapter over the shared, fleet-wide ghost extractor # fm_composer_strip_ghost (bin/fm-composer-lib.sh). It drops de-emphasised @@ -74,108 +117,293 @@ FM_TMUX_BUSY_REGEX_DEFAULT='esc (to )?interrupt|Working\.\.\.|Ctrl\+c:cancel' # so the tmux and herdr adapters cannot drift apart on what counts as ghost text. fm_tmux_strip_ghost() { fm_composer_strip_ghost; } -# fm_tmux_composer_state: classify the cursor/composer line of <target> as -# empty - no pending input (blank, a busy footer, an empty agent composer, or -# only de-emphasised ghost/placeholder text). Safe to inject; also the positive -# acknowledgement that a submit landed. -# pending - real, unsubmitted text on the cursor line (a human mid-typing, or a -# previous injection whose Enter was swallowed). Defer / retry. -# unknown - the pane could not be read (tmux error), OR the cursor line is a -# bare shell prompt (`$`/`%`/`#`/`>`) - a dead shell, not an agent -# composer, so NOT a safe injection target. The caller decides. -# -# The cursor line is captured WITH ANSI styling (capture-pane -e) and bounded to -# the single composer row (-S/-E). The bordered flag (a genuine composer box) is -# read from the PLAIN row (fm_composer_strip_ansi keeps ghost text so the box -# border is still visible), while the real-typed CONTENT is extracted with the -# shared fm_composer_strip_ghost so dim/faint AND dark-truecolor ghost text drops -# out before classification (grok's dark box border drops with the ghost, which -# is why the bordered flag is read from the plain row, not the ghost-stripped -# one). Both are internal only, never surfaced. The detector strips the harness's -# box-drawing composer borders ("│ … │", heavy "┃", or a plain ASCII "|") using -# literal-string substitution (bash 3.2 safe, locale-independent - no \u escapes, -# no multibyte character classes), and delegates the empty/pending/unknown -# decision to the shared owner fm_composer_classify_content -# (bin/fm-composer-lib.sh). The bordered flag is what lets a bordered `│ > │` -# (claude's own idle composer) read empty while a bare, unbordered `$ ` dead-shell -# prompt reads unknown. -fm_tmux_composer_state() { # <target> -> empty|pending|unknown - local target=$1 cy raw plain stripped bordered=0 - cy=$(tmux display-message -p -t "$target" '#{cursor_y}' 2>/dev/null) || { printf 'unknown'; return 0; } - case "$cy" in ''|*[!0-9]*) printf 'unknown'; return 0 ;; esac - raw=$(tmux capture-pane -e -p -t "$target" -S "$cy" -E "$cy" 2>/dev/null) || { printf 'unknown'; return 0; } - # bordered: from the plain row (borders survive an all-ANSI strip). +# fm_tmux_composer_row_state: classify one raw styled candidate row. +# A structural caller forces bordered=1; the compatibility fallback passes 0 +# and may recognize a busy footer. +fm_tmux_composer_row_state() { # <raw-row> [bordered] [allow-busy] -> empty|pending|unknown + local raw=$1 bordered=${2:-0} allow_busy=${3:-1} plain stripped plain=$(printf '%s\n' "$raw" | fm_composer_strip_ansi) plain="${plain#"${plain%%[![:space:]]*}"}" plain="${plain%"${plain##*[![:space:]]}"}" - case "$plain" in - '│'*'│'|'┃'*'┃'|'|'*'|') bordered=1 ;; - esac - # content: from the ghost-stripped row (real typed text only). stripped=$(printf '%s\n' "$raw" | fm_composer_strip_ghost) stripped="${stripped#"${stripped%%[![:space:]]*}"}" stripped="${stripped%"${stripped##*[![:space:]]}"}" case "$stripped" in '│'*'│') stripped=${stripped#│}; stripped=${stripped%│} ;; '┃'*'┃') stripped=${stripped#┃}; stripped=${stripped%┃} ;; + '║'*'║') stripped=${stripped#║}; stripped=${stripped%║} ;; '|'*'|') stripped=${stripped#|}; stripped=${stripped%|} ;; esac stripped="${stripped#"${stripped%%[![:space:]]*}"}" stripped="${stripped%"${stripped##*[![:space:]]}"}" - # A busy footer landing on the cursor line is not pending input (tmux-specific: - # only tmux captures the raw cursor row, which may BE the footer). - if [ -n "$stripped" ] \ + if [ "$allow_busy" = 1 ] && [ -n "$stripped" ] \ && printf '%s' "$stripped" | grep -qiE "${FM_BUSY_REGEX:-$FM_TMUX_BUSY_REGEX_DEFAULT}"; then printf 'empty'; return 0 fi fm_composer_classify_content "$bordered" "$stripped" "${FM_COMPOSER_IDLE_RE:-}" insensitive "$plain" } -# fm_pane_input_pending: 0 (pending) if the cursor line holds real unsubmitted -# text, 1 otherwise. An unreadable pane is treated as NOT pending (fail-safe: -# the same bias the old daemon used — an unknown pane defers nothing here). +fm_tmux_row_has_composer_edge() { # <plain-row> + local row=$1 + row="${row#"${row%%[![:space:]]*}"}" + row="${row%"${row##*[![:space:]]}"}" + case "$row" in + '│'*|*'│'|'┃'*|*'┃'|'║'*|*'║'|'╭'*|*'╭'|'╮'*|*'╮'|\ + '┌'*|*'┌'|'┐'*|*'┐'|'╔'*|*'╔'|'╗'*|*'╗'|'┏'*|*'┏'|'┓'*|*'┓'|\ + '╰'*|*'╰'|'╯'*|*'╯'|'└'*|*'└'|'┘'*|*'┘'|'╚'*|*'╚'|'╝'*|*'╝'|\ + '┗'*|*'┗'|'┛'*|*'┛'|'─'*|*'─'|'━'*|*'━'|'═'*|*'═'|'|'*|*'|'|'+'*|*'+') + return 0 + ;; + esac + return 1 +} + +fm_tmux_composer_geometry_spaces() { # <content-inner> -> spaces + local content=$1 probe + probe="${content#"${content%%[![:space:]]*}"}" + case "$probe" in + '>'*) content=${content/>/ } ;; + '❯'*) content=${content/❯/ } ;; + '›'*) content=${content/›/ } ;; + esac + content=$(printf '%s' "$content" | LC_ALL=C sed 's/[!-~]/ /g') + case "$content" in + *[![:space:]]*) return 1 ;; + esac + printf '%s' "$content" +} + +# fm_tmux_find_composer_box: print the zero-based top and bottom rows of the +# complete bordered box that structurally contains the cursor, plus whether its +# geometry is ambiguous. The cursor may be on any content row or on the bottom +# border; no fixed cursor offset is used. +fm_tmux_find_composer_box() { # <cursor-y> <plain-visible-pane> -> "<top> <bottom> <ambiguous>" + local cy=$1 pane=$2 line indent left_stripped trimmed kind family current_family= + local side_family top_inner top_spaces='' geometry_check=0 geometry_ambiguous=0 + local content_inner content_spaces bottom_inner bottom_spaces + local current_indent= + local row=0 top=-1 valid=0 content_rows=0 unsafe=0 cursor_structural=0 + while IFS= read -r line; do + indent=${line%%[![:space:]]*} + left_stripped="${line#"${line%%[![:space:]]*}"}" + trimmed="${left_stripped%"${left_stripped##*[![:space:]]}"}" + kind= + family= + case "$trimmed" in + '╭'*'╮') kind=top; family=rounded ;; + '┌'*'┐') kind=top; family=light ;; + '╔'*'╗') kind=top; family=double ;; + '┏'*'┓') kind=top; family=heavy ;; + '╰'*'╯') kind=bottom; family=rounded ;; + '└'*'┘') kind=bottom; family=light ;; + '╚'*'╝') kind=bottom; family=double ;; + '┗'*'┛') kind=bottom; family=heavy ;; + '+'*'+') kind=ascii; family=ascii ;; + esac + if [ "$row" -eq "$cy" ] && fm_tmux_row_has_composer_edge "$trimmed"; then + cursor_structural=1 + fi + if [ "$kind" = top ] || { [ "$kind" = ascii ] && [ "$top" -lt 0 ]; }; then + if [ "$top" -ge 0 ] && [ "$top" -lt "$cy" ] && [ "$cy" -le "$row" ]; then + unsafe=1 + fi + top=$row + current_family=$family + current_indent=$indent + valid=1 + content_rows=0 + geometry_ambiguous=0 + geometry_check=1 + top_inner=$trimmed + case "$family" in + rounded) top_inner=${top_inner#╭}; top_inner=${top_inner%╮}; top_spaces=${top_inner//─/ } ;; + light) top_inner=${top_inner#┌}; top_inner=${top_inner%┐}; top_spaces=${top_inner//─/ } ;; + double) top_inner=${top_inner#╔}; top_inner=${top_inner%╗}; top_spaces=${top_inner//═/ } ;; + heavy) top_inner=${top_inner#┏}; top_inner=${top_inner%┓}; top_spaces=${top_inner//━/ } ;; + ascii) top_inner=${top_inner#+}; top_inner=${top_inner%+}; top_spaces=${top_inner//-/ } ;; + esac + case "$top_spaces" in + *[![:space:]]*) geometry_check=0; geometry_ambiguous=1 ;; + esac + elif [ "$kind" = bottom ] || { [ "$kind" = ascii ] && [ "$top" -ge 0 ]; }; then + if [ "$top" -ge 0 ] && [ "$family" = "$current_family" ] \ + && [ "$valid" = 1 ] && [ "$content_rows" -gt 0 ] \ + && [ "$top" -lt "$cy" ] && [ "$cy" -le "$row" ]; then + [ "$indent" = "$current_indent" ] || geometry_ambiguous=1 + if [ "$geometry_check" = 1 ]; then + bottom_inner=$trimmed + case "$family" in + rounded) bottom_inner=${bottom_inner#╰}; bottom_inner=${bottom_inner%╯}; bottom_spaces=${bottom_inner//─/ } ;; + light) bottom_inner=${bottom_inner#└}; bottom_inner=${bottom_inner%┘}; bottom_spaces=${bottom_inner//─/ } ;; + double) bottom_inner=${bottom_inner#╚}; bottom_inner=${bottom_inner%╝}; bottom_spaces=${bottom_inner//═/ } ;; + heavy) bottom_inner=${bottom_inner#┗}; bottom_inner=${bottom_inner%┛}; bottom_spaces=${bottom_inner//━/ } ;; + ascii) bottom_inner=${bottom_inner#+}; bottom_inner=${bottom_inner%+}; bottom_spaces=${bottom_inner//-/ } ;; + esac + [ "$bottom_spaces" = "$top_spaces" ] || geometry_ambiguous=1 + fi + printf '%s %s %s' "$top" "$row" "$geometry_ambiguous" + return 0 + fi + if { [ "$top" -ge 0 ] && [ "$top" -lt "$cy" ] && [ "$cy" -le "$row" ]; } \ + || [ "$row" -eq "$cy" ]; then + unsafe=1 + fi + top=-1 + current_family= + current_indent= + valid=0 + content_rows=0 + elif [ "$top" -ge 0 ]; then + side_family= + case "$trimmed" in + '│'*'│') side_family=single ;; + '┃'*'┃') side_family=heavy ;; + '║'*'║') side_family=double ;; + '|'*'|') side_family=ascii ;; + esac + case "$current_family:$side_family" in + rounded:single|light:single|heavy:heavy|double:double|ascii:ascii) + content_rows=$((content_rows + 1)) + [ "$indent" = "$current_indent" ] || geometry_ambiguous=1 + if [ "$geometry_check" = 1 ]; then + content_inner=$trimmed + case "$side_family" in + single) content_inner=${content_inner#│}; content_inner=${content_inner%│} ;; + heavy) content_inner=${content_inner#┃}; content_inner=${content_inner%┃} ;; + double) content_inner=${content_inner#║}; content_inner=${content_inner%║} ;; + ascii) content_inner=${content_inner#|}; content_inner=${content_inner%|} ;; + esac + if content_spaces=$(fm_tmux_composer_geometry_spaces "$content_inner"); then + [ "$content_spaces" = "$top_spaces" ] || geometry_ambiguous=1 + else + geometry_ambiguous=1 + fi + fi + ;; + *) valid=0 ;; + esac + fi + row=$((row + 1)) + done <<EOF +$pane +EOF + if [ "$top" -ge 0 ] && [ "$top" -lt "$cy" ]; then + unsafe=1 + fi + if [ "$unsafe" = 1 ] || [ "$cursor_structural" = 1 ]; then + return 2 + fi + return 1 +} + +# fm_tmux_composer_state classification contract: +# A row is structural only when its first or last non-whitespace character is a +# composer edge. A complete box has matching border families and bounded top and +# bottom rows. The proof-carrying verdict is empty for proven emptiness, pending +# for proven text in established structure, pending-unproven for text in +# ambiguous structure, and unknown for unreadable state. Consumers that can +# overwrite input or confirm delivery must accept only the exact positive proof +# they require, so unrecognized future verdicts fail safe by default. Empty +# requires positive proof: a genuinely empty composer, an all-empty unambiguous +# box, an empty non-bordered fallback row, or the submit core's proven +# busy-queued Enter conversion. +fm_tmux_composer_state() { # <target> -> empty|pending|pending-unproven|unknown + local target=$1 cy raw pane plain box box_status top bottom geometry_ambiguous + local row row_raw state unknown_seen=0 + cy=$(tmux display-message -p -t "$target" '#{cursor_y}' 2>/dev/null) || { printf 'unknown'; return 0; } + case "$cy" in ''|*[!0-9]*) printf 'unknown'; return 0 ;; esac + pane=$(tmux capture-pane -e -p -t "$target" -S 0 -E - 2>/dev/null) || { printf 'unknown'; return 0; } + plain=$(printf '%s\n' "$pane" | fm_composer_strip_ansi) + if box=$(fm_tmux_find_composer_box "$cy" "$plain"); then + top=${box%% *} + box=${box#* } + bottom=${box%% *} + geometry_ambiguous=${box#* } + row=$((top + 1)) + while [ "$row" -lt "$bottom" ]; do + row_raw=$(printf '%s\n' "$pane" | sed -n "$((row + 1))p") + state=$(fm_tmux_composer_row_state "$row_raw" 1 0) + case "$state" in + pending) + if [ "$geometry_ambiguous" = 1 ]; then + printf 'pending-unproven' + else + printf 'pending' + fi + return 0 + ;; + unknown) unknown_seen=1 ;; + esac + row=$((row + 1)) + done + if [ "$unknown_seen" = 1 ] || [ "$geometry_ambiguous" = 1 ]; then + printf 'unknown' + else + printf 'empty' + fi + return 0 + else + box_status=$? + if [ "$box_status" -eq 2 ]; then + printf 'unknown' + return 0 + fi + fi + raw=$(tmux capture-pane -e -p -t "$target" -S "$cy" -E "$cy" 2>/dev/null) \ + || { printf 'unknown'; return 0; } + if fm_tmux_row_has_composer_edge "$(printf '%s\n' "$raw" | fm_composer_strip_ansi)"; then + printf 'unknown' + return 0 + fi + fm_tmux_composer_row_state "$raw" 0 +} + +# fm_pane_input_pending: 0 when the composer is not proven empty, so pending +# text, ambiguous structure, unreadable state, and future verdicts all defer. fm_pane_input_pending() { # <target> - [ "$(fm_tmux_composer_state "$1")" = pending ] + [ "$(fm_tmux_composer_state "$1")" != empty ] } # fm_pane_is_busy: 0 if the pane's last few non-blank lines show a busy footer # (an agent mid-turn). Scans a 40-line tail like fm-watch.sh. -fm_pane_is_busy() { # <target> - local win=$1 tail40 +fm_pane_is_busy() { # <target> [harness] + local win=$1 harness=${2:-} tail40 tail40=$(tmux capture-pane -p -t "$win" -S -40 2>/dev/null) || return 1 - printf '%s' "$tail40" | grep -v '^[[:space:]]*$' | tail -6 \ - | grep -qiE "${FM_BUSY_REGEX:-$FM_TMUX_BUSY_REGEX_DEFAULT}" + printf '%s' "$tail40" | grep -v '^[[:space:]]*$' | tail -12 \ + | fm_busy_lines_match "$harness" } # fm_tmux_submit_core: type <text> into <target> ONCE, then submit with Enter, # verifying the composer cleared. Retries Enter ONLY — never retypes, because a # swallowed Enter leaves our text in the composer and retyping would duplicate -# it. Echoes the final verdict on stdout (empty|pending|unknown|send-failed) so callers can -# pick their own success policy: -# - the daemon clears its buffer only on "empty" (strict: an unknown pane must -# not be mistaken for a delivered escalation). -# - fm-send fails only on "pending" (lenient: a positively-confirmed swallow), -# so an unreadable pane never turns a normal steer into a false error. +# it. Echoes the final proof-carrying verdict on stdout so callers can require +# exact `empty` before treating submission as confirmed. # Busy-queued Enter (opencode 1.18.4): the harness accepts Enter while mid-turn # and queues it for after the current turn, but keeps the typed text visible in -# the composer. Once the Enter-retry budget is spent and the composer still -# reads "pending", the submit core falls back to `fm_pane_is_busy`: a busy pane -# means the Enter was accepted and queued (report `empty` so the caller does -# not re-send), while an idle pane keeps `pending` as a genuine swallow. This -# is the only place that exception lives, so the daemon's strict and -# fm-send's lenient success policies both treat a busy-queued Enter as -# delivered. +# the composer. Once the Enter-retry budget is spent and a structurally proven +# composer still reads "pending", the submit core falls back to +# `fm_pane_is_busy`: a busy pane means the Enter was accepted and queued (report +# `empty` so the caller does not re-send), while an idle pane keeps `pending` as +# a genuine swallow. Pending-unproven receives the same Enter retry budget but +# never reaches this exception. fm_tmux_submit_enter_core() { # <target> <retries> <enter-sleep> local target=$1 retries=$2 sleep_s=$3 i=0 state while :; do tmux send-keys -t "$target" Enter 2>/dev/null || true sleep "$sleep_s" state=$(fm_tmux_composer_state "$target") - [ "$state" = pending ] || { printf '%s' "$state"; return 0; } + case "$state" in + pending|pending-unproven) ;; + *) printf '%s' "$state"; return 0 ;; + esac i=$((i + 1)) [ "$i" -lt "$retries" ] || break done - # Retries exhausted, composer still shows pending. + if [ "$state" != pending ]; then + printf '%s' "$state" + return 0 + fi + # Retries exhausted, composer still shows proven pending. # If the pane is busy (agent mid-turn), the harness accepted the Enter # and queued the message for processing when the current turn ends. # Treat it as submitted so the caller does not re-send. diff --git a/bin/fm-turnend-guard-grok.sh b/bin/fm-turnend-guard-grok.sh index 5cc33e9c54e..3dcfd6f2f64 100755 --- a/bin/fm-turnend-guard-grok.sh +++ b/bin/fm-turnend-guard-grok.sh @@ -1,30 +1,69 @@ #!/usr/bin/env bash # Grok Stop-hook adapter for the firstmate PRIMARY turn-end guard. # -# Grok Stop hooks are passive: exit 2 does not block or feed stderr back to the -# model. This adapter still uses the shared primary-scoped predicate in -# fm-turnend-guard.sh. When that predicate says the primary would end blind, the -# adapter forces one same-session follow-up by running `grok --resume <session>` -# with a guard instruction. GROK_TURNEND_GUARD_ACTIVE is the loop guard: the -# nested turn's own Stop hook exits without spawning another nested turn. +# The exact running Stop payload selects one path. A typed native capability +# field delegates the shared guard's exit status and stderr directly back to +# that Grok process. Field absence preserves the pre-native one-resume fallback. +# Invalid or unreadable input starts neither path. Camel case has typed +# precedence over the legacy snake-case spelling when both are present. set -u PAYLOAD=$(cat 2>/dev/null || true) [ -n "$PAYLOAD" ] || exit 0 -[ -n "${GROK_TURNEND_GUARD_ACTIVE:-}" ] && exit 0 +command -v jq >/dev/null 2>&1 || exit 0 +printf '%s' "$PAYLOAD" | jq -n --stream -e ' + reduce inputs as $item ( + {}; + if ( + ($item | length) == 2 + and ($item[0] | length) > 0 + and ( + $item[0][0] == "sessionId" + or $item[0][0] == "stopHookActive" + or $item[0][0] == "stop_hook_active" + ) + ) then + .[$item[0][0]] = ((.[$item[0][0]] // 0) + 1) + else + . + end + ) + | all(.[]; . == 1) +' >/dev/null 2>&1 || exit 0 +CAPABILITY=$(printf '%s' "$PAYLOAD" | jq -ser ' + if length != 1 then error("payload count") + elif ((.[0] | type) != "object") then error("payload") + else .[0] | + if has("stopHookActive") then + if ((.stopHookActive | type) == "boolean") then "native" else error("stopHookActive") end + elif has("stop_hook_active") then + if ((.stop_hook_active | type) == "boolean") then "native" else error("stop_hook_active") end + else "legacy" + end + end +' 2>/dev/null) || exit 0 ROOT=${GROK_WORKSPACE_ROOT:-${CLAUDE_PROJECT_DIR:-}} [ -n "$ROOT" ] || exit 0 ROOT=${ROOT%/} [ -x "$ROOT/bin/fm-turnend-guard.sh" ] || exit 0 -if ! command -v jq >/dev/null 2>&1; then - exit 0 +if [ "$CAPABILITY" = native ]; then + printf '%s' "$PAYLOAD" | "$ROOT/bin/fm-turnend-guard.sh" + RC=$? + case "$RC" in + 0|2) exit "$RC" ;; + *) exit 0 ;; + esac fi -SESSION_ID=$(printf '%s' "$PAYLOAD" | jq -r '.sessionId // empty' 2>/dev/null) || exit 0 -[ -n "$SESSION_ID" ] || exit 0 +# Only a genuine pre-native payload reaches this bounded compatibility path. +[ -n "${GROK_TURNEND_GUARD_ACTIVE:-}" ] && exit 0 +SESSION_ID=$(printf '%s' "$PAYLOAD" | jq -er ' + .sessionId | select(type == "string" and length > 0) +' 2>/dev/null) || exit 0 +command -v grok >/dev/null 2>&1 || exit 0 ERR=$(mktemp "${TMPDIR:-/tmp}/fm-turnend-grok.XXXXXX") || exit 0 trap 'rm -f "$ERR"' EXIT diff --git a/bin/fm-turnend-guard.sh b/bin/fm-turnend-guard.sh index ff82b65873b..2e96fb33e48 100755 --- a/bin/fm-turnend-guard.sh +++ b/bin/fm-turnend-guard.sh @@ -11,8 +11,10 @@ # This script is push-based: verified harness turn-end hooks invoke it every time # the primary is about to end a turn. # Claude and codex can block directly by preserving exit status 2 and stderr. -# OpenCode, pi, and grok adapters use the same predicate and force one bounded -# follow-up because their turn-end events are passive. +# OpenCode and pi adapters use the same predicate and force one bounded +# follow-up because their turn-end events are passive. Grok delegates native +# blocking when its running Stop payload advertises that capability, with one +# bounded resume fallback for payloads from pre-native processes. # See docs/turnend-guard.md for the per-harness mechanics, validation evidence, # and fail-open tradeoffs. # @@ -26,15 +28,35 @@ # primary checkout - the main home or a genuinely marked secondmate home - and # stay a silent, fast no-op inside child task worktrees. # -# Loop-guard: never block twice in the same turn. Claude Code and codex Stop -# payloads carry stop_hook_active=true when the CURRENT stop attempt was itself -# already forced by an earlier block this turn; on that signal we always allow -# the stop, whether or not watcher supervision actually got resumed. Passive -# harness adapters provide their own one-follow-up guard before calling this -# script. -# That bounds this to at most one forced continuation per turn - never a wedged, -# un-endable session - while still nagging again on a later turn if the problem -# persists. +# Loop-guard, codex/Grok (default) mode: never block twice in the same turn. +# Codex uses stop_hook_active and Grok uses stopHookActive; typed camel-case +# takes precedence when both spellings are present. A true value means the +# current stop attempt already follows a block, so this guard always allows it. +# Passive harness adapters provide their own one-follow-up guard before calling +# this script. +# That bounds those harnesses to at most one forced continuation per turn - +# never a wedged, un-endable session - while still nagging again on a later turn +# if the problem persists. +# +# Loop-guard, --claude mode (Stop-owned auto-arm cooperation): Claude Code +# marks EVERY stop after ANY stop-hook-driven continuation stop_hook_active=true, +# including turns started by the asyncRewake auto-arm, so the one-shot allow +# would re-open the exact blind window this guard exists to close +# (docs/turnend-guard.md records the 2026-07-21 incident). In --claude mode this +# guard ignores stop_hook_active and instead cooperates with the Stop-owned +# auto-arm (bin/fm-claude-stop-autoarm.sh), which fires on the same Stop event: +# 1. a live identity-matched watcher with a fresh beacon allows immediately; +# 2. otherwise wait briefly (FM_CLAUDE_AUTOARM_SYNC_WAIT_MS, default 800ms) +# for the auto-arm to claim this home (state/.claude-autoarm.lock owner +# alive) or to record a fresh rewake outcome (state/.claude-autoarm-epoch) +# for this event epoch - either proof allows without consuming a +# continuation, so one event epoch yields exactly one recovery turn; +# 3. only when neither materializes is the auto-arm genuinely absent: re-block +# with the repair banner, bounded to FM_CLAUDE_TURNEND_BLOCK_BUDGET +# (default 3) consecutive blocks per session - safely below Claude Code's +# hard 8-consecutive-block override - then allow degraded with a visible +# systemMessage so the session can always end. +# Any allow resets the consecutive-block budget. set -u SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -44,6 +66,20 @@ STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" CONFIG="${FM_CONFIG_OVERRIDE:-$FM_HOME/config}" GRACE=${FM_GUARD_GRACE:-300} WATCH="$SCRIPT_DIR/fm-watch.sh" +CLAUDE_MODE=0 +SYNC_WAIT_MS=${FM_CLAUDE_AUTOARM_SYNC_WAIT_MS:-800} +EPOCH_FRESH=${FM_CLAUDE_AUTOARM_EPOCH_FRESH:-15} +BLOCK_BUDGET=${FM_CLAUDE_TURNEND_BLOCK_BUDGET:-3} +case "$SYNC_WAIT_MS" in ''|*[!0-9]*) SYNC_WAIT_MS=800 ;; esac +case "$EPOCH_FRESH" in ''|*[!0-9]*|0) EPOCH_FRESH=15 ;; esac +case "$BLOCK_BUDGET" in ''|*[!0-9]*|0) BLOCK_BUDGET=3 ;; esac + +for arg in "$@"; do + case "$arg" in + --claude) CLAUDE_MODE=1 ;; + *) echo "usage: $(basename "$0") [--claude]" >&2; exit 2 ;; + esac +done # shellcheck source=bin/fm-supervision-lib.sh . "$SCRIPT_DIR/fm-supervision-lib.sh" @@ -60,8 +96,18 @@ PAYLOAD=$(cat 2>/dev/null || true) # loop-guard field, so we must never block - fail open, not noisy. command -v jq >/dev/null 2>&1 || exit 0 -STOP_HOOK_ACTIVE=$(printf '%s' "$PAYLOAD" | jq -r '.stop_hook_active // false' 2>/dev/null) || exit 0 -[ "$STOP_HOOK_ACTIVE" = "true" ] && exit 0 +STOP_HOOK_ACTIVE=$(printf '%s' "$PAYLOAD" | jq -r ' + if type != "object" then error("payload") + elif has("stopHookActive") then + if ((.stopHookActive | type) == "boolean") then .stopHookActive else error("stopHookActive") end + elif has("stop_hook_active") then + if ((.stop_hook_active | type) == "boolean") then .stop_hook_active else error("stop_hook_active") end + else false + end +' 2>/dev/null) || exit 0 +if [ "$CLAUDE_MODE" -eq 0 ] && [ "$STOP_HOOK_ACTIVE" = "true" ]; then + exit 0 +fi # --- scope precisely to a PRIMARY checkout ---------------------------------- # A genuinely-marked secondmate home runs its OWN primary firstmate session, so @@ -81,22 +127,113 @@ fm_primary_scope_matches "$FM_ROOT" "$STATE" || exit 0 # shellcheck source=bin/fm-wake-lib.sh . "$SCRIPT_DIR/fm-wake-lib.sh" +BUDGET_FILE="$STATE/.turnend-claude-blocks" +budget_reset() { + [ "$CLAUDE_MODE" -eq 1 ] || return 0 + rm -f "$BUDGET_FILE" 2>/dev/null || true +} + fm_supervision_status "$STATE" "$GRACE" -[ "$FM_SUP_IN_FLIGHT" -gt 0 ] || exit 0 -fm_watcher_healthy "$STATE" "$WATCH" "$GRACE" "$FM_HOME" && exit 0 +if [ "$CLAUDE_MODE" -eq 1 ]; then + if [ "$FM_SUP_NEEDED" = false ]; then + budget_reset + exit 0 + fi +else + if [ "$FM_SUP_IN_FLIGHT" -eq 0 ]; then + budget_reset + exit 0 + fi +fi +if fm_watcher_healthy "$STATE" "$WATCH" "$GRACE" "$FM_HOME"; then + budget_reset + exit 0 +fi + +block_stop() { + local afk x_mode reason rule + afk=0 + [ -e "$STATE/.afk" ] && afk=1 + x_mode=0 + [ -f "$CONFIG/x-mode.env" ] && x_mode=1 + reason=$("$SCRIPT_DIR/fm-supervision-instructions.sh" --afk "$afk" --x-mode "$x_mode" --repair-line 2>/dev/null \ + || printf '%s\n' 'tasks in flight, no live watcher - repair missing watcher supervision according to the session-start operating block before ending the turn') + rule='━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━' + { + printf '●%s\n' "$rule" + printf '● TURN WOULD END BLIND - SUPERVISION IS OFF\n' + if [ "$FM_SUP_IN_FLIGHT" -gt 0 ]; then + printf '● %s task(s) in flight, but no live watcher holds this home lock (last beat: %s).\n' "$FM_SUP_IN_FLIGHT" "$FM_SUP_BEACON_DESC" + else + printf '● X-mode relay polling needs supervision, but no live watcher holds this home lock (last beat: %s).\n' "$FM_SUP_BEACON_DESC" + fi + if [ "$CLAUDE_MODE" -eq 1 ]; then + printf '● The Stop-owned auto-arm did not claim this home either, so recovery is NOT already under way.\n' + fi + printf '● %s\n' "$reason" + printf '●%s\n' "$rule" + } >&2 + exit 2 +} + +if [ "$CLAUDE_MODE" -eq 0 ]; then + block_stop +fi + +# --- --claude cooperative path ----------------------------------------------- +# The Stop-owned auto-arm fires on the same Stop event. Give it a brief bounded +# window to prove it owns recovery for this event epoch before consuming one of +# Claude's bounded continuations. +autoarm_owns_recovery() { + local pid outcome age + fm_watcher_healthy "$STATE" "$WATCH" "$GRACE" "$FM_HOME" && return 0 + pid=$(cat "$STATE/.claude-autoarm.lock/pid" 2>/dev/null || true) + fm_pid_alive "$pid" && return 0 + outcome=$(sed -n 's/^.*outcome=\([a-z][a-z]*\) .*$/\1/p' "$STATE/.claude-autoarm-epoch" 2>/dev/null || true) + if [ "$outcome" = rewake ]; then + age=$(fm_path_age "$STATE/.claude-autoarm-epoch") + [ "$age" -lt "$EPOCH_FRESH" ] && return 0 + fi + return 1 +} + +i=0 +while [ "$i" -lt $((SYNC_WAIT_MS / 100)) ]; do + if autoarm_owns_recovery; then + budget_reset + exit 0 + fi + sleep 0.1 + i=$((i + 1)) +done +if autoarm_owns_recovery; then + budget_reset + exit 0 +fi -afk=0 -[ -e "$STATE/.afk" ] && afk=1 -x_mode=0 -[ -f "$CONFIG/x-mode.env" ] && x_mode=1 -REASON=$("$SCRIPT_DIR/fm-supervision-instructions.sh" --afk "$afk" --x-mode "$x_mode" --repair-line 2>/dev/null \ - || printf '%s\n' 'tasks in flight, no live watcher - repair missing watcher supervision according to the session-start operating block before ending the turn') -rule='━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━' -{ - printf '●%s\n' "$rule" - printf '● TURN WOULD END BLIND - SUPERVISION IS OFF\n' - printf '● %s task(s) in flight, but no live watcher holds this home lock (last beat: %s).\n' "$FM_SUP_IN_FLIGHT" "$FM_SUP_BEACON_DESC" - printf '● %s\n' "$REASON" - printf '●%s\n' "$rule" -} >&2 -exit 2 +# The auto-arm genuinely failed to establish: re-block, but never past the +# budget so the session can always end and Claude's 8-block override is never +# approached. +SESSION_ID=$(printf '%s' "$PAYLOAD" | jq -r '.session_id // "unknown"' 2>/dev/null || printf 'unknown') +COUNT=0 +if [ -f "$BUDGET_FILE" ]; then + old_session=$(sed -n '1s/^session=//p' "$BUDGET_FILE" 2>/dev/null || true) + old_count=$(sed -n '2s/^count=//p' "$BUDGET_FILE" 2>/dev/null || true) + case "$old_count" in + ''|*[!0-9]*) old_count=0 ;; + esac + [ "$old_session" = "$SESSION_ID" ] && COUNT=$old_count +fi +COUNT=$((COUNT + 1)) +if [ "$COUNT" -gt "$BLOCK_BUDGET" ]; then + budget_reset + if [ "$FM_SUP_IN_FLIGHT" -gt 0 ]; then + NEED_DESC="$FM_SUP_IN_FLIGHT task(s) in flight" + else + NEED_DESC="X-mode relay polling active" + fi + printf '{"systemMessage":"firstmate turn-end guard: %s with no live watcher and no Stop auto-arm claim; block budget exhausted, allowing this stop. Repair supervision (bin/fm-watch-arm.sh as a Claude Code background task) or investigate why bin/fm-claude-stop-autoarm.sh is not claiming this home."}\n' "$NEED_DESC" + exit 0 +fi +printf 'session=%s\ncount=%s\n' "$SESSION_ID" "$COUNT" > "$BUDGET_FILE" 2>/dev/null || true +block_stop diff --git a/bin/fm-wake-lib.sh b/bin/fm-wake-lib.sh index 929c7231a47..8cec58bec1d 100755 --- a/bin/fm-wake-lib.sh +++ b/bin/fm-wake-lib.sh @@ -9,6 +9,10 @@ STATE="${FM_STATE_OVERRIDE:-${STATE:-$FM_HOME/state}}" FM_WAKE_QUEUE="${FM_WAKE_QUEUE:-$STATE/.wake-queue}" FM_WAKE_QUEUE_LOCK="${FM_WAKE_QUEUE_LOCK:-$STATE/.wake-queue.lock}" FM_LOCK_STALE_AFTER="${FM_LOCK_STALE_AFTER:-2}" +# Resolved once at source time: fm_pid_identity and fm_path_mtime run inside 0.2s +# confirm and 0.5s attach polls, and forking uname per call is a measurable cost on +# the platform (Git Bash/MSYS) that already pays the highest fork price. +_FM_UNAME=$(uname 2>/dev/null || echo unknown) mkdir -p "$STATE" fm_current_pid() { @@ -24,17 +28,19 @@ fm_pid_alive() { } fm_pid_identity() { - local pid=$1 out proc_root stat_line starttime cmdline_hex + local pid=$1 out proc_root stat_line starttime cmdline_hex identity_key local -a stat_fields case "$pid" in ''|*[!0-9]*) return 1 ;; esac proc_root=${FM_PROC_ROOT_OVERRIDE:-/proc} - # Prefer /proc on Linux: stat field 22 (starttime, clock ticks since boot) is + # Prefer a Linux-compatible /proc when present: stat field 22 (starttime, clock ticks since boot) is # immune to the wall-clock steps that re-render the ps lstart fallback's date # (observed as WSL2 btime drift) and would evict a live watcher; combining the # full NUL-separated cmdline keeps PID reuse a mismatch even on a tick collision. - if [ "$(uname)" = Linux ] && [ -r "$proc_root/$pid/stat" ] && [ -r "$proc_root/$pid/cmdline" ]; then + # Git Bash/MSYS exposes these compatible files but its Cygwin ps rejects the + # portable fallback's -o fields, so capability detection must not key on uname. + if [ -r "$proc_root/$pid/stat" ] && [ -r "$proc_root/$pid/cmdline" ]; then stat_line=$(cat "$proc_root/$pid/stat" 2>/dev/null) || return 1 # After the final comm delimiter, array index 19 is proc stat field 22. read -r -a stat_fields <<< "${stat_line##*)}" @@ -45,7 +51,9 @@ fm_pid_identity() { esac cmdline_hex=$(od -An -v -tx1 "$proc_root/$pid/cmdline" 2>/dev/null | tr -d '[:space:]') || return 1 [ -n "$cmdline_hex" ] || return 1 - printf 'linux-starttime=%s cmdline-hex=%s\n' "$starttime" "$cmdline_hex" + identity_key=proc-starttime + [ "$_FM_UNAME" != Linux ] || identity_key=linux-starttime + printf '%s=%s cmdline-hex=%s\n' "$identity_key" "$starttime" "$cmdline_hex" return 0 fi # Pin LC_ALL=C so lstart's date format is locale-invariant: the identity is @@ -57,7 +65,7 @@ fm_pid_identity() { } fm_path_mtime() { - if [ "$(uname)" = Darwin ]; then + if [ "$_FM_UNAME" = Darwin ]; then stat -f %m "$1" 2>/dev/null else stat -c %Y "$1" 2>/dev/null diff --git a/bin/fm-watch-arm.sh b/bin/fm-watch-arm.sh index 5c20fbaf5e6..3c2df49c891 100755 --- a/bin/fm-watch-arm.sh +++ b/bin/fm-watch-arm.sh @@ -6,8 +6,11 @@ # daemon owns triage and the watcher exits on every wake for the daemon to # classify. Reliability depends on arming through a mechanism that SURVIVES the # call and NOTIFIES on exit, so firstmate must run this script as the harness's -# own tracked background task (e.g. run_in_background). Run it as its own -# standalone background task, never bundled onto the tail of another command. +# own tracked background task (e.g. run_in_background), or - for a Claude +# primary - inside the Stop asyncRewake hook's foreground process tree +# (bin/fm-claude-stop-autoarm.sh), where the harness owns the process group and +# the hook's exit-2 rewake is the notification. Run it as its own standalone +# background task, never bundled onto the tail of another command. # NEVER fire it and forget with a shell `&` inside another call: that backgrounded # child is reaped when the call returns, leaving NO watcher running and a false # "already running" off the dying process. That exact mistake silently took @@ -65,7 +68,13 @@ BEAT="$STATE/.last-watcher-beat" # "Fresh" reuses the guard's threshold so there is one definition of liveness. GRACE=${FM_GUARD_GRACE:-300} # How long to wait for a freshly forked watcher to acquire the lock and beat. -CONFIRM_TIMEOUT=${FM_ARM_CONFIRM_TIMEOUT:-10} +# Git Bash/MSYS pays a much higher fork cost while the watcher completes its +# required pre-lock migration, so its bounded default covers that cold start. +case "${OSTYPE:-}" in + msys*|mingw*|cygwin*) ARM_CONFIRM_DEFAULT=30 ;; + *) ARM_CONFIRM_DEFAULT=10 ;; +esac +CONFIRM_TIMEOUT=${FM_ARM_CONFIRM_TIMEOUT:-$ARM_CONFIRM_DEFAULT} # Poll interval while attached to an existing healthy watcher. ATTACH_POLL=${FM_ARM_ATTACH_POLL:-0.5} CYCLE_LOG="$STATE/.watch-cycle-exits.log" diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index 1bb9f05e6cb..e5501f852b3 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -30,7 +30,16 @@ # also carries a "demand-deep-inspection" marker so the # wake payload itself, not just repetition, forces a # closer look instead of another routine supervision -# resume. Unless afk is active. +# resume. Unless afk is active. A genuinely busy pane +# (window_is_busy true) is exempt from the above, but +# only up to BUSY_TURN_MAX_SECS with no completed turn +# (state/<id>.turn-ended, or the spawn record before any +# turn completes); past that bound busy_turn_over_age +# routes it through the same wedge timer, so it surfaces +# with the identical "stale: ..." reason, escalation +# count, and demand-deep-inspection marker, for human +# inspection only - never an automatic interrupt, +# signal, or restart of the worker or its tool process. # check: <script>: <out> authenticated check output, always actionable # check: rejected unauthenticated state checks: <paths> # unsafe state checks were refused without execution @@ -100,11 +109,13 @@ CHECK_TIMEOUT=${FM_CHECK_TIMEOUT:-30} # seconds allowed per *.check.sh SIGNAL_GRACE=${FM_SIGNAL_GRACE:-30} # seconds to linger after a signal so trailing # signals (a status write, then the same turn's # turn-end hook) coalesce into one wake -# Busy signatures per harness, OR-ed. Extend via env when new adapters are verified. +# Busy signatures are selected by recorded harness unless FM_BUSY_REGEX globally +# overrides them. # claude/codex: "esc to interrupt"; opencode: "esc interrupt"; pi: "Working..."; -# grok: "Ctrl+c:cancel" (the mid-turn cancel hint in grok's keybind bar, shown iff a -# turn is running; absent when idle - verified grok 0.2.73, ASCII to avoid the -# locale fragility of matching grok's braille spinner glyph directly). +# grok: "Ctrl+c:cancel". Claude's current spinner signature is matched only for +# a recorded Claude task because an ellipsis followed by elapsed time is not a +# safe shared signature for arbitrary harness output. Kimi's moon-plus-middot +# spinner signature is likewise matched only for a recorded Kimi task. BUSY_REGEX=${FM_BUSY_REGEX:-'esc (to )?interrupt|Working\.\.\.|Ctrl\+c:cancel'} # Always-on wake triage: most wakes during a long crew validation are benign (a # working: note or turn-end while a pipeline runs, a no-change heartbeat). Rather @@ -125,6 +136,19 @@ BUSY_REGEX=${FM_BUSY_REGEX:-'esc (to )?interrupt|Working\.\.\.|Ctrl\+c:cancel'} # daemon owns triage, so this watcher reverts to one-shot (enqueue + exit on every # wake) and never double-triages - and never runs the costly provably-working read. STALE_ESCALATE_SECS=${FM_STALE_ESCALATE_SECS:-240} # idle secs before a provably-working stale escalates as a possible wedge +# A busy pane is unconditional proof of liveness with no built-in duration bound, +# so a hung foreground call can remain hidden even while its rendered busy +# footer changes every poll. BUSY_TURN_MAX_SECS bounds how long any busy pane +# may go with no completed turn: once its task's +# state/<id>.turn-ended marker (or, before any turn has completed, the task's +# spawn record) is this old, busy_turn_over_age routes the pane through the +# same STALE_ESCALATE_SECS-paced wedge_timer_check used for a provably-working +# non-busy stale, so it escalates via the existing stale reason, escalation +# counter, and demand-deep-inspection marker for human inspection only - never +# an automatic interrupt, signal, or restart. A completed turn touches +# turn-ended and resets the age. Set generously above any legitimate interval +# between completed turns, including long tool calls, builds, or test runs. +BUSY_TURN_MAX_SECS=${FM_BUSY_TURN_MAX_SECS:-3600} # A crew that declared a pause is idling on a known external wait, so its stale # pane is absorbed rather than wedge-escalated. # A captain-held or paused crew whose agent has confidently exited uses the same @@ -158,19 +182,24 @@ hash_pane() { # window_is_busy: 0 (busy) iff the task's harness is actively working. Prefers # a backend's native semantic busy state (fm_backend_busy_state - herdr's # agent.get; herdr-addendum "busy state" row, "the first backend where -# fm_session_busy_state gets real semantics"); falls back to the existing -# pane-tail regex ONLY when the backend reports unknown (tmux always does, so -# its path is unchanged byte-for-byte). <tail40> is the same bounded capture -# already read for hashing, so this adds no extra backend calls on the -# regex-fallback path. +# fm_session_busy_state gets real semantics"); when the backend reports unknown, +# falls back to the recorded harness's verified pane-tail signature. <tail40> is +# the same bounded capture already read for hashing, so this adds no extra +# backend calls on the regex-fallback path. window_is_busy() { # <window> <tail40> - local w=$1 tail40=$2 bs + local w=$1 tail40=$2 bs harness lines bs=$(fm_backend_busy_state "$(window_backend "$w")" "$w" 2>/dev/null) case "$bs" in busy) return 0 ;; idle) return 1 ;; *) - printf '%s' "$tail40" | grep -v '^[[:space:]]*$' | tail -6 | grep -qiE "$BUSY_REGEX" + lines=$(printf '%s' "$tail40" | grep -v '^[[:space:]]*$' | tail -12) + harness=$(window_harness "$w") + if [ -n "${FM_BUSY_REGEX:-}" ]; then + printf '%s' "$lines" | grep -qiE "$BUSY_REGEX" + else + printf '%s' "$lines" | fm_busy_lines_match "$harness" + fi ;; esac } @@ -202,6 +231,13 @@ window_backend() { echo tmux } +window_harness() { + local w=$1 meta + meta=$(fm_backend_meta_for_window "$w" "$STATE" 2>/dev/null || true) + [ -n "$meta" ] || return 0 + grep '^harness=' "$meta" | cut -d= -f2- || true +} + window_label() { local w=$1 task task=$(window_to_task "$w" "$STATE") @@ -267,6 +303,20 @@ wedge_timer_check() { # <window> <since-file> <triage-label> <escalation-count- esac } +# busy_turn_over_age: 0 iff <task>'s latest completed-turn marker is at least +# BUSY_TURN_MAX_SECS old. Ages the per-task turn-ended marker, the harness-neutral +# signal every verified harness's turn-end hook touches; before any turn has +# completed, ages the task's spawn record instead so a fresh task still gets a +# bound. The caller checks that the pane is busy and routes a crossed bound +# through the existing wedge_timer_check, never anything that touches the +# worker itself. +busy_turn_over_age() { # <task> + local task=$1 f + f="$STATE/$task.turn-ended" + [ -e "$f" ] || f="$STATE/$task.meta" + [ "$(age_of "$f")" -ge "$BUSY_TURN_MAX_SECS" ] +} + # Absorb a stale pane under a declared external-wait pause (paused:) or a # dead-agent captain-held transfer, and re-surface it once every # PAUSE_RESURFACE_SECS for a recheck so it cannot rot invisibly. Called on any @@ -837,14 +887,16 @@ EOF ewf="$STATE/.wedge-escalations-$key" pf="$STATE/.paused-$key" # flag: this key's stale is using the bounded pause cadence prev=$(cat "$hf" 2>/dev/null || true) + # Busy match: a backend's native semantic state when available (herdr), else + # the last 6 non-blank lines only (the TUI footer area, where every verified + # harness renders its busy indicator) so busy-looking strings in displayed + # content cannot suppress stale detection. Read once per window per poll and + # reused below so a busy verdict is consistent within one cycle. + if window_is_busy "$w" "$tail40"; then busy_now=0; else busy_now=1; fi if [ "$h" = "$prev" ]; then n=$(( $(cat "$cf" 2>/dev/null || echo 0) + 1 )) echo "$n" > "$cf" - # Busy match: a backend's native semantic state when available (herdr), - # else the last 6 non-blank lines only (the TUI footer area, where every - # verified harness renders its busy indicator) so busy-looking strings - # in displayed content cannot suppress stale detection. - if [ "$n" -ge 2 ] && ! window_is_busy "$w" "$tail40"; then + if [ "$n" -ge 2 ] && [ "$busy_now" -ne 0 ]; then # The pane is idle/stale at hash $h. Triage decides whether this wakes # firstmate. Detection itself is unchanged from above. if [ "$kind" = secondmate ]; then @@ -944,8 +996,14 @@ EOF fi fi else - # Pane busy or not yet stably stale: reset pending escalation bookkeeping. - rm -f "$ssf" "$ewf" + # Pane busy or not yet stably stale: reset pending escalation bookkeeping, + # unless a genuinely busy pane has gone too long with no completed turn - + # then route it through the same wedge timer instead of erasing it. + if [ "$busy_now" -eq 0 ] && busy_turn_over_age "$task"; then + wedge_timer_check "$w" "$ssf" "busy (no completed turn)" "$ewf" + else + rm -f "$ssf" "$ewf" + fi if [ -e "$pf" ] && { [ "$n" -ge 2 ] || ! status_is_paused_or_captain_held "$(last_status_line "$STATE/$(window_to_task "$w" "$STATE").status")"; }; then clear_pause_tracking "$w" fi @@ -953,9 +1011,13 @@ EOF else printf '%s' "$h" > "$hf" echo 0 > "$cf" - rm -f "$ssf" "$ewf" + if [ "$busy_now" -eq 0 ] && busy_turn_over_age "$task"; then + wedge_timer_check "$w" "$ssf" "busy (no completed turn)" "$ewf" + else + rm -f "$ssf" "$ewf" + fi task=$(window_to_task "$w" "$STATE") - if ! afk_present && status_is_paused_or_captain_held "$(last_status_line "$STATE/$task.status")" && ! window_is_busy "$w" "$tail40"; then + if ! afk_present && status_is_paused_or_captain_held "$(last_status_line "$STATE/$task.status")" && [ "$busy_now" -ne 0 ]; then case "$(pause_state_class "$w" "$task")" in paused) handle_paused_stale "$w" "$task" "$h" ;; *) clear_pause_tracking "$w" ;; diff --git a/docs/architecture.md b/docs/architecture.md index cb1f6df3e7b..bf8b5cb3ec1 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -11,6 +11,7 @@ firstmate's always-loaded operating contract and routing index for conditional p A zero-token bash watcher (`bin/fm-watch.sh`) sleeps on the fleet, classifies detected wakes in bash, and wakes the first mate only when something is actionable. Actionable wakes include captain-relevant status signals, no-verb signals whose crew is not provably working, authenticated check output such as PR merge polling or an X-mode mention, stale panes whose crew is not provably working whether their status log looks terminal or non-terminal, provably-working stale panes that persist past `FM_STALE_ESCALATE_SECS`, declared external waits that remain paused past `FM_PAUSE_RESURFACE_SECS`, and heartbeat backstop hits. Repeated provably-working stale escalations on the same unchanged pane add an escalation count to the wake reason and, at `FM_WEDGE_DEMAND_INSPECT_COUNT`, a `demand-deep-inspection` marker. +A busy pane is otherwise exempt from staleness, but only until its latest `state/<id>.turn-ended` marker reaches `FM_BUSY_TURN_MAX_SECS`, or its `state/<id>.meta` spawn record reaches that age before any turn completes; past that bound it is routed through the same wedge escalation, with the identical reason, escalation count, and `demand-deep-inspection` marker, for inspection only - never an automatic interrupt, signal, or restart. Those actionable wakes are written to a durable local queue (`state/.wake-queue`) before detector state advances, so a missed process exit can be recovered by draining the queue. When a canonical validated PR poll returns exactly `merged`, the watcher appends that durable notification before publishing a private receipt bound to the poll's registration, bytes, file identities, metadata, provider, URL, and task ID. The receipt makes retirement safely retryable across restarts: fixed-path recovery revalidates the same evidence, removes the runnable check first, removes its registration and data sidecars, removes the receipt last, and preserves task metadata including `pr=` and `pr_head=`. @@ -35,7 +36,7 @@ The most recent recognized ci log marker wins, so checks-green monitoring report Only when no matching run exists does it fall back to the pane busy-signature and then a status-log event whose verb maps to a recognized run-state; a dead pane without a run reports unknown instead of trusting a stale log. Decision-only events such as `resolved` never become current state or leak their prose into the current-state detail. In that status-log fallback, a declared external wait reports the distinct `paused` state with its reason. -For herdr, that pane fallback trusts a native `busy` verdict outright, but corroborates native `idle` or unknown verdicts against the rendered busy signature before deciding the crew is not working. +For herdr, that pane fallback trusts a native `busy` verdict outright, but corroborates native `idle` or unknown verdicts against the recorded harness's rendered busy signature before deciding the crew is not working. For whole-fleet read-only review, `bin/fm-fleet-snapshot.sh --json` emits schema `fm-fleet-snapshot.v1` from the backlog, task metadata, current crew state, endpoint probes, PR/report pointers, scout reports, bounded current summaries from registered secondmate homes, and secondmate return-channel guidance. `bin/fm-fleet-view.sh` renders that snapshot as Markdown for humans, while `bin/fm-bearings-snapshot.sh` provides the bounded bearings projection, so both views consume one structured contract instead of reparsing raw fleet files. The script header owns the exact JSON schema. @@ -54,13 +55,14 @@ The default path remains local-only; live GitHub enrichment exists only behind t Optional X mode integrates with the watcher only after explicit opt-in; [configuration.md](configuration.md#x-mode-env) owns its generated-artifact and dispatch mechanics. At session start, `bin/fm-session-start.sh` emits exactly one primary-harness supervision block rendered by `bin/fm-supervision-instructions.sh` from `docs/supervision-protocols/`. -That block owns the live wait shape for the running primary harness: Claude and Grok use background-notify cycles, Codex uses bounded foreground checkpoints, Pi uses its two tracked primary extensions, and OpenCode uses its TUI plugin. +That block owns the live wait shape for the running primary harness: Claude's Stop `asyncRewake` hook owns tokenless re-arm cycles, Grok uses background-notify cycles, Codex uses bounded foreground checkpoints, Pi and pi-signed use the same two tracked primary extensions, and OpenCode uses its TUI plugin. `bin/fm-watch-arm.sh` remains the verified arm wrapper for protocols that call it; it forks the watcher as a tracked child, verifies it is genuinely alive with a fresh liveness beacon, and prints an honest `started`, `attached`, or nonzero `FAILED` status. On `attached` it stays live across identity-matched successors, and an unexplained clean child close either attaches to a verified healthy successor or becomes the typed nonzero `watcher: FAILED - cycle ended without an actionable reason` result. The arm layer records one bounded lifecycle row per observed cycle in `state/.watch-cycle-exits.log`; `state/.watch-triage.log` remains exclusively the absorbed-wake debug log. Pi and OpenCode verify session-lock ownership and launch one singleton successor from their child-close handlers before delivering an actionable wake prompt, with bounded exponential retry for failed restoration. -Claude keeps its tracked background-task protocol and adds a narrow PreToolUse continuity gate that allows drain, arm recovery, and fail-closed teardown while refusing only other fleet commands when tasks are in flight and no identity-matched live watcher holds the home lock. -The existing turn-end guard is unchanged and remains the final backstop for all five harness protocols. +Claude's `bin/fm-claude-stop-autoarm.sh` hook fires on every Stop and, when the home is eligible and still needs supervision, claims one home-scoped cycle, foregrounds the arm wrapper, and translates an actionable close or typed failure into one exit-2 rewake. +[`watcher-continuity.md`](watcher-continuity.md) owns Claude's residual active-turn coverage and watcher-status command-gating boundary. +The existing turn-end guard remains the final backstop for all five harness-engine protocols, with pi-signed sharing Pi's protocol and the `--claude` mode cooperating with the auto-arm claim. Its `--restart` mode signals only the watcher recorded in the current home's `state/.watch.lock`, so restarting one home cannot kill sibling secondmate watchers. A pull-based guard (`bin/fm-guard.sh`) warns through supervision tool output if the primary checkout is tangled, or if tasks are in flight and that watcher stops running or queued wakes are waiting to be drained. The drain script calls that guard after emptying the queue, which avoids repeating the queued-wakes warning for records it just consumed while still warning on stale watcher liveness. @@ -78,7 +80,7 @@ Its supervisor injection path supports tmux and herdr panes, with `FM_SUPERVISOR Pane existence, busy checks, composer checks, capture, and verified submit route through `bin/fm-backend.sh`: tmux keeps the same submit core used by the tmux send backend, while herdr uses native busy state, native agent-state submit confirmation on idle baselines, and its ANSI-aware structural composer classifier for pending-input guards and submit fallback. The tmux submit core (shared `fm_tmux_submit_enter_core`) treats a busy pane + retries-exhausted + composer-still-pending as a queued Enter (opencode 1.18.4 accepts Enter mid-turn and queues it for after the turn), reported as `empty` so the daemon and `fm-send` do not re-send; an idle pane keeps the `pending` verdict as a genuine swallow. The same opencode busy-queue case is a known gap on the herdr adapter and is recorded in `docs/herdr-backend.md` rather than patched here. Composer-content classification has one shared owner, `bin/fm-composer-lib.sh`, used by tmux, herdr, Orca, and cmux after each adapter performs its own capture and composer-row recognition. -The daemon injects only into an affirmatively `empty` composer, so both `pending` and `unknown` defer and a bare dead-shell prompt cannot receive an escalation; the complete policy is in [Composer-emptiness safety](herdr-backend.md#composer-emptiness-safety-2026-07-10-fleet-wide-across-all-four-backends). +The daemon injects only into an affirmatively `empty` composer, so both `pending` and `unknown` defer and a bare dead-shell prompt cannot receive an escalation; the current boundary is in [Composer and injection safety](herdr-backend.md#composer-and-injection-safety). Unsupported supervisor backends refuse at daemon startup. Stalled escalation delivery writes `state/.subsuper-inject-wedged` and attempts a configured backend-independent active alert after `FM_MAX_DEFER_SECS` instead of silently deferring forever. On an unmarked return, `bin/fm-afk-return.sh` owns ordered shutdown, durable catch-up evidence, and the fail-closed gate that keeps ordinary work behind every live firstmate-actionable blocker. @@ -88,22 +90,25 @@ On an unmarked return, `bin/fm-afk-return.sh` owns ordered shutdown, durable cat The runtime backend is the session-provider layer below firstmate's scripts. It owns task endpoint creation, bounded capture, text/key sends, current-path reads for spawn-time worktree discovery when the backend does not create the worktree itself, live-window fallback lookup, agent-process liveness probes where verified, and endpoint teardown. -`bin/fm-backend.sh` centralizes backend selection, `state/<id>.meta` helpers, selector resolution, and operation dispatch; `bin/backends/tmux.sh` is the verified reference adapter ([`docs/tmux-backend.md`](tmux-backend.md)), and `bin/backends/herdr.sh` (P2), `bin/backends/zellij.sh` (P3), `bin/backends/orca.sh` (P4), and `bin/backends/cmux.sh` (P5) are experimental task-spawn adapters. +`bin/fm-backend.sh` centralizes backend selection, `state/<id>.meta` helpers, metadata-only cleanup identity validation, selector resolution, and operation dispatch; `bin/backends/tmux.sh` is the verified reference adapter ([`docs/tmux-backend.md`](tmux-backend.md)), and `bin/backends/herdr.sh` (P2), `bin/backends/zellij.sh` (P3), `bin/backends/orca.sh` (P4), and `bin/backends/cmux.sh` (P5) are experimental task-spawn adapters. New spawns select a backend from `--backend`, then `FM_BACKEND`, then local `config/backend`, then runtime auto-detection from `$TMUX`, `HERDR_ENV=1`, or cmux runtime signals, then default `tmux`. Runtime auto-detection is innermost-first: `$TMUX` wins over `HERDR_ENV=1`, which wins over cmux's primary `CMUX_WORKSPACE_ID` marker and documented fallback signals; auto-detected herdr or cmux prints a one-time opt-out notice, auto-detected tmux stays silent, and zellij and orca are never auto-detected (only explicit selection). Unknown backend names fail loudly. For compatibility, default tmux tasks do not write `backend=tmux`; every reader treats a missing `backend=` field as `tmux`. -`fm-watch.sh` polls each window's backend for a busy state: tmux, zellij, orca, and cmux have no native primitive and always report unknown, preserving the original pane-tail-regex detection unchanged; herdr's `agent.get` semantic state (working/idle/done/blocked) is consulted first for stale detection, with unknown native states falling back to the same regex. +`fm-watch.sh` polls each window's backend for a busy state: tmux, zellij, orca, and cmux have no native primitive and always report unknown, so their pane-tail fallback matches only the recorded harness's verified signature; herdr's `agent.get` semantic state (working/idle/done/blocked) is consulted first for stale detection, with unknown native states using the same harness-scoped fallback. +This scope prevents cross-harness false positives such as Kimi's rotating idle tip `ctrl+c: cancel` borrowing Grok's busy token, and keeps Claude's broader elapsed-spinner shape from matching ordinary output in other panes. +Unknown supplied harnesses match no default signature, while callers that have no harness metadata retain the historical combined-pattern compatibility fallback. That poll loop is the default event source for backends with no native push events, so this stays an extraction of the abstraction rather than a watcher rewrite. -For capable herdr sessions, the same watcher replaces its terminal sleep with a bounded native event wait that immediately surfaces `blocked`; [herdr-backend.md](herdr-backend.md#native-paneagent_status_changed-push-escalation-immediate-blocked-wake) owns the mechanism, capability gates, and verification evidence. +For capable Herdr sessions, the same watcher replaces its terminal sleep with a bounded native event wait that immediately surfaces `blocked`; [Push events and polling fallback](herdr-backend.md#push-events-and-polling-fallback) owns the current mechanism and capability gates, while [runtime backend verification](verification/runtime-backends.md#native-blocked-event) owns the active evidence. The deeper session-start agent-process liveness probe is separate from that busy-state poll: tmux and Herdr have verified classifiers for secondmate recovery, Zellij remains unverified, and Orca and cmux do not support secondmate spawns. -Herdr is experimental and can be selected explicitly or by runtime auto-detection: treehouse remains the worktree provider for it exactly as it is for tmux (herdr is a session provider only), and its full verification - the container shape decision, created-vs-adopted default-tab prune safety, restored-layout husk respawn idempotency, verified CLI facts, ANSI-preserved ghost/placeholder classification through the shared extractor, a verified small-`--lines` capture bug and its workaround, and known gaps - is recorded in `docs/herdr-backend.md`. +Herdr is experimental and can be selected explicitly or by runtime auto-detection: Treehouse remains its worktree provider, [`herdr-backend.md`](herdr-backend.md) owns current setup and safety limits, and [`verification/runtime-backends.md`](verification/runtime-backends.md#herdr) owns active empirical evidence. Herdr's durable default container shape is workspace-per-home plus tab-per-task: the primary home uses workspace label `firstmate`, secondmate homes use `2ndmate-<secondmate-id>`, and recovery/list-live scopes to the current `FM_HOME`'s workspace. -Its optional default-off presentation projection may place one clean new task in a disposable workspace without changing endpoint authority or lifecycle ownership; [`docs/herdr-backend.md`](herdr-backend.md#optional-disposable-single-task-presentation-spaces) owns that conditional design. -Zellij is experimental and selected only explicitly: treehouse remains its worktree provider too, and its full verification - the resolved "gaps to verify" list from the original design report, the unconditional-exit-0 CLI quirk and its mitigation, the focus-steal-on-new-tab finding, the home-scoped tab-title collision fix, and known gaps - is recorded in `docs/zellij-backend.md`. +Its optional default-off presentation projection may place one clean new task in a disposable workspace without changing endpoint authority or lifecycle ownership; [Optional presentation spaces](herdr-backend.md#optional-presentation-spaces) owns that conditional design and its narrow home-local restored-shell cleanup at locked session start. +Zellij is experimental and selected only explicitly: Treehouse remains its worktree provider, [`zellij-backend.md`](zellij-backend.md) owns current setup and limits, and [`verification/runtime-backends.md`](verification/runtime-backends.md#zellij) owns active empirical evidence. Zellij's container shape is simpler than herdr's: one shared `firstmate` session, one tab per task, with no per-home workspace split; visible tab titles are scoped by the active home label plus a short hash of the resolved `FM_ROOT` path. -Orca is experimental and selected only explicitly: Orca owns both worktree and terminal lifecycle, records `orca_worktree_id=` and `terminal=`, and removes worktrees through `orca worktree rm` only after the usual firstmate teardown checks pass. Its current behavior and limitations are recorded in `docs/orca-backend.md`. -cmux is experimental, GUI-first, macOS-only, and can be selected explicitly or by runtime auto-detection from its primary `CMUX_WORKSPACE_ID` marker plus documented fallback signals: treehouse remains its worktree provider (cmux is a session provider only, like herdr/zellij), and its full verification - the socket access setup requirement with Automation mode recommended, the read-screen-fails-on-a-fresh-surface finding, the close-surface-refuses-on-the-last-surface finding, the source-verified runtime marker and fallback behavior, and known gaps - is recorded in `docs/cmux-backend.md`. +Orca is experimental and selected only explicitly: Orca owns both worktree and terminal lifecycle, records `orca_worktree_id=` and `terminal=`, and removes worktrees through `orca worktree rm` only after the usual firstmate teardown checks pass. +[`orca-backend.md`](orca-backend.md) owns current behavior and limitations, while [`verification/runtime-backends.md`](verification/runtime-backends.md#orca) owns active smoke evidence. +cmux is experimental, GUI-first, macOS-only, and can be selected explicitly or by runtime auto-detection from its primary `CMUX_WORKSPACE_ID` marker plus documented fallback signals: Treehouse remains its worktree provider, [`cmux-backend.md`](cmux-backend.md) owns current setup and limits, and [`verification/runtime-backends.md`](verification/runtime-backends.md#cmux) owns active source and live evidence. cmux's container shape is one workspace per task with one surface, no per-home container split; workspace titles are scoped by the active home label plus a short hash of the resolved `FM_ROOT` path, and `--secondmate` spawns are refused, mirroring Orca. Codex App support is recorded in `docs/codex-app-backend.md`; it is not selectable as a runtime backend. @@ -138,13 +143,13 @@ The intake and authority contract in `AGENTS.md` owns when separate scout resear ## Dispatch profiles Crewmate and scout dispatch can stay on the static crewmate harness resolved by `config/crew-harness`, or it can use local dispatch profiles in `config/crew-dispatch.json`. -The dispatch file is intentionally judgment-based: firstmate reads the natural-language rules at intake, chooses the best matching rule, resolves that rule directly or through a supported selector, and passes only concrete `--harness`, `--model`, and `--effort` axes to `fm-spawn.sh`. -The shell scripts validate the JSON shape and verified harness/effort combinations, and `fm-dispatch-select.sh` owns quota-aware array selection plus OS-backed random fallback, but they do not parse task intent or match the natural-language rules. +The dispatch file is intentionally judgment-based: firstmate reads the natural-language rules at intake, chooses the best matching rule, resolves profile arrays itself from current quota output under the `AGENTS.md` section 4 intake boundary and the `quota-array-dispatch` selection procedure, and passes only concrete `--harness`, `--model`, and `--effort` axes to `fm-spawn.sh`. +The shell scripts validate the JSON shape and verified harness/effort combinations, but they do not parse task intent, match natural-language rules, or own array selection. The session-start bootstrap step keeps valid dispatch configuration silent unless verbose facts are enabled and surfaces a concise invalid-config line when validation fails. When the file exists, `fm-spawn.sh` refuses crewmate and scout launches without an explicit harness, so `config/crew-harness` is only automatic when no dispatch profile file is active. Secondmate launches are exempt because they resolve the secondmate harness and any optional secondmate model or effort tokens instead. Unsupported effort values are still recorded in task meta when passed to `fm-spawn.sh`, but the launch template omits any effort flag that the selected harness does not accept. -That keeps spawn launch compatible across claude, codex, grok, pi, and opencode while preserving the requested profile for later audit. +That keeps spawn launch compatible across claude, codex, grok, pi, opencode, and kimi while preserving the requested profile for later audit. ## Optional secondmates @@ -258,7 +263,7 @@ The mechanics are owned by the `/updatefirstmate` skill and firstmate's operatin Fleet state lives in each task's session-provider backend (tmux by hard default, herdr or cmux when selected or auto-detected, zellij/orca when explicitly selected), no-mistakes run records, status event logs, local markdown under `data/` including `data/captain.md`, `data/captain-shared.md`, and `data/learnings.md`, and persistent secondmate homes. For herdr, respawning after a server-restored layout closes and replaces confirmed no-agent or dead task-tab husks instead of requiring manual tab cleanup. -At session start, confirmed agent-less endpoints are closed and relaunched, authoritatively missing endpoints are relaunched directly, and ambiguous or unreadable targets are left untouched to avoid duplicate supervisors. +At session start, confirmed-dead secondmate agent endpoints are closed and relaunched through the same secondmate spawn path, while ambiguous liveness reads are left untouched to avoid duplicate supervisors. Use `/stow` before an intentional reset when the conversation may hold durable knowledge that has not yet been written to disk; after that, the next firstmate session can reconcile and carry on. ## Development notes diff --git a/docs/arm-pretool-check.md b/docs/arm-pretool-check.md index 3d51099202f..f4747e0abdc 100644 --- a/docs/arm-pretool-check.md +++ b/docs/arm-pretool-check.md @@ -15,17 +15,6 @@ The seatbelt rejects those command shapes before execution. This policy is not a post-arm liveness guarantee. `bin/fm-guard.sh`, `bin/fm-turnend-guard.sh`, the watcher lock, and the watcher beacon still prove whether supervision is healthy after an allowed call. -## Claude continuity gate - -Claude also registers `bin/fm-continuity-pretool-check.sh` for Bash PreToolUse events. -This is a separate, tightly bounded recovery gate rather than another watcher-shape policy. -It runs only in a primary home, and it denies only an executed `bin/fm-*.sh` command other than `bin/fm-wake-drain.sh`, `bin/fm-watch-arm.sh`, or the ordinary literal `bin/fm-teardown.sh` when task metadata is in flight and no identity-matched live watcher holds that home's lock. -Ordinary shell commands, fleet-script names used as data, all commands in an idle fleet, child worktrees, wake drain, watcher arm, and ordinary literal teardown remain allowed. -The denial gives Claude reason-specific recovery guidance — drain, re-arm via a tracked Claude background task, and use fail-closed `bin/fm-teardown.sh` for completed tasks — per the contract in [`watcher-continuity.md`](watcher-continuity.md). -`bin/fm-continuity-command-policy.mjs` reuses this document's shell lexer and command-position analysis but owns the recovery-versus-other-fleet classification. -Malformed transport or opaque dynamic syntax fails open so this narrow gate cannot become a blanket Bash block. -The existing `bin/fm-turnend-guard.sh` Stop integration is unchanged and remains the final backstop. - The classifier never executes, sources, evaluates, or expands any part of the submitted command. It tokenizes the bytes and classifies lexical execution positions only. @@ -35,7 +24,7 @@ It tokenizes the bytes and classifies lexical execution positions only. - Stdin JSON at `.tool_input.command` for Claude and Codex. - Stdin JSON at `.toolInput.command` for Grok. -- `--command <exact string>` for OpenCode and Pi. +- `--command <exact string>` for OpenCode, Pi, and pi-signed. - `--background` as a compatibility-only field that never changes the decision. - `--claude` to preserve Claude's stderr-only deny requirement. @@ -162,7 +151,7 @@ Prose may improve without changing adapter behavior. - `--claude` suppresses stdout completely because Claude ignores a PreToolUse deny when stdout is nonempty. - Codex blocks on exit 2 and displays stderr. - OpenCode throws only when the checker exits 2. -- Pi returns `{block: true}` only when the checker exits 2. +- Pi and pi-signed return `{block: true}` only when the checker exits 2. ## Harness wiring @@ -172,7 +161,7 @@ Prose may improve without changing adapter behavior. | Claude | `.tool_input.command` | `.claude/settings.json` forwards stdin with `--claude`, leaving stdout empty and returning the stderr deny object. | | Grok | `.toolInput.command` | `.grok/hooks/fm-primary-pretool-check.json` forwards stdin and Grok consumes the stdout `decision=deny` object. | | OpenCode | `output.args.command` | `.opencode/plugins/fm-primary-pretool-check.js` passes one `--command` argument and throws only for exit 2. | -| Pi | `event.input.command` | `.pi/extensions/fm-primary-turnend-guard.ts` passes one `--command` argument and returns `{block: true}` only for exit 2. | +| Pi / pi-signed | `event.input.command` | `.pi/extensions/fm-primary-turnend-guard.ts` passes one `--command` argument and returns `{block: true}` only for exit 2. | Grok project hooks require folder trust. Every shell variable reference in a Grok hook command must carry an inline default such as `${GROK_WORKSPACE_ROOT:-}` because Grok expands the raw hook command before `bash -lc` runs it. diff --git a/docs/calm-mode-feasibility.md b/docs/calm-mode-feasibility.md index ff76b892baa..b94b6a6aef9 100644 --- a/docs/calm-mode-feasibility.md +++ b/docs/calm-mode-feasibility.md @@ -1,7 +1,7 @@ # Calm-mode harness feasibility This document owns the version-scoped feasibility evidence, Pi transcript taxonomy, and supported-API boundaries for Firstmate calm mode. -The README owns the user-facing `/calm` usage and limitation contract. +[`calm.md`](calm.md) owns the current user-facing `/calm` usage and limitation contract. ## Required extension surface @@ -9,9 +9,17 @@ A qualifying implementation must auto-load from the trusted project, persist the The governing presentation policy allows genuine original user prompts, genuine user-facing assistant text, and Pi's native working activity. Changing persisted context to remove hidden content, filtering provider context, patching installed harness code, or claiming coverage outside a supported renderer does not satisfy that boundary. +## Compatibility evidence + +[`calm.md`](calm.md#pi-compatibility) owns the current Pi compatibility contract. +Pi 0.81.1 was installed when Calm was first built, and Pi 0.82.0 was the later reverification target. +The inspected Pi CHANGELOG shows no relevant presentation API introduced at either version, so those versions remain verification evidence rather than compatibility bounds. +The exported classes used by the adapters (`AssistantMessageComponent` and `InteractiveMode`) are undocumented internals with no stated version guarantee. +`tests/fm-calm-pi-extension.test.sh` records the installed Pi version as evidence without gating on it and covers both newer synthetic versions and an unavailable adapter seam. + ## Pi 0.81.1 end-to-end reproduction -The current installed and regression-supported Pi version was verified on 2026-07-22. +The Pi version installed at the time was verified on 2026-07-22. ```text $ pi --version @@ -57,7 +65,8 @@ The single-thinking, tool-call-only, tool-result, Calm-off, and `clearOnShrink` PR 927 made Calm persistent and described controlled rows as gapless while retaining a documented unsupported boundary for collapsed-thinking spacing. PR 936 removed the unsafe operational-input reroute and preserved legacy zero-height entries but did not change assistant-message layout. -The fix installs one idempotent Pi 0.81.1 presentation adapter on the exported `AssistantMessageComponent.updateContent` method. +The fix installs one idempotent presentation adapter, verified on Pi 0.81.1 through 0.82.0, on the exported `AssistantMessageComponent.updateContent` method. +The adapter probes for that exact method and, per the [compatibility contract](calm.md#pi-compatibility), degrades independently with a diagnostic rather than gating on a version number. Only while Calm is active and Pi has collapsed thinking does the adapter pass a shallow thinking-free presentation copy into Pi's ordinary layout calculation, then retain the original message on the component for invalidation and thinking expansion. The persisted assistant message, provider context, tool execution, export data, and expansion history remain unchanged. Collapsed thinking-only assistant messages now render zero rows, thinking before visible assistant text adds no spacing beyond the text-only baseline, and expanding thinking still renders the original reasoning. @@ -69,7 +78,7 @@ Ordinary user-role near misses remain visible, including quoted current markers, ## Duplicate-turn regression and semantic boundary -The captain-visible regression reproduced three consecutive times in the persisted Pi session at `/Users/kunchen/.pi/agent/sessions/--Users-kunchen-github-kunchenguid-firstmate--/2026-07-23T16-37-24-672Z_019f8fd6-c440-7641-b2bf-8065dab1622a.jsonl`. +The captain-visible regression reproduced three consecutive times in a persisted Pi session under `~/.pi/agent/sessions/`. Assistant `bb83873b` was followed by hidden custom input `9d087b52` and distinct duplicate assistant `f4232aa3`. Assistant `3a388d8c` was followed by adjacent hidden custom inputs `e1914f28` and `cfdefb09` and distinct duplicate assistant `47c81eeb`. Distinct provider response identifiers and signatures prove separate model turns rather than duplicate TUI paint. @@ -114,7 +123,8 @@ The real Pi viewport moved the unchanged assistant text from row 7 to row 2, ren The leading cause would have been falsified if the row or height remained, the provider lost or duplicated the message, or the persisted role or bytes changed. None occurred. -The fix installs a separate idempotent Pi 0.81.1 presentation adapter on the exported `InteractiveMode.addMessageToChat` method. +The fix installs a separate idempotent presentation adapter, verified on Pi 0.81.1 through 0.82.0, on the exported `InteractiveMode.addMessageToChat` method. +The adapter probes for that exact method and, per the [compatibility contract](calm.md#pi-compatibility), degrades independently with a diagnostic rather than gating on a version number. It delegates current recognition to `bin/fm-operational-input.sh`, adds only the evidence-backed bare-U+2063 `Supervisor escalate (` presentation compatibility shape, mounts a `UserMessageComponent` subclass that preserves Pi's stock row plus leading spacer while Calm is off, and returns zero rendered lines while Calm is on. It never intercepts the input event, rewrites the message, changes its role, filters model context, or changes session data. Messages containing an image are left on Pi's ordinary path even when their text equals an operational envelope because Firstmate's authoritative producers are text-only. @@ -148,7 +158,7 @@ Serialized session data and Pi 0.81.1's sidebar tree also retain legacy hidden o The taxonomy was derived from Pi 0.81.1's installed public declarations, documentation, examples, `interactive-mode.js`, and its exported component implementations. The test fixture enumerates every class below through the centralized policy, and the interactive fixture exercises the screenshot classes, current user-role operational input, and legacy synthetic presentation entries. -| Policy class | Pi transcript path | Calm result on Pi 0.81.1 | +| Policy class | Pi transcript path | Calm result (verified on Pi 0.81.1 through 0.82.0) | | --- | --- | --- | | `genuine-user-prompt` | `UserMessageComponent` | Visible, including every tested operational near miss. | | `genuine-agent-response` | Assistant text in `AssistantMessageComponent` | Visible. | @@ -167,12 +177,12 @@ The test fixture enumerates every class below through the centralized policy, an | `system-notice` | `showStatus`, `showError`, compaction, retry, and startup warning rows | Unsupported boundary; remains visible. | | `cache-notice` | Non-persisted cache-miss `Text` row | Unsupported boundary; remains visible. | | `project-trust-warning` | Non-persisted startup `Text` row | Unsupported boundary; remains visible. | -| `synthetic-user` | Firstmate extension `sendUserMessage`, terminal-injected input, Firstmate-generated Pi positional brief, or the already non-displayed session-start nudge | Canonically classified text-only operational user messages stay ordinary semantic user messages but render through the zero-height Pi 0.81.1 adapter under Calm; legacy entries stay gaplessly controllable, and the session-start nudge retains its existing non-displayed custom-message path. | +| `synthetic-user` | Firstmate extension `sendUserMessage`, terminal-injected input, Firstmate-generated Pi positional brief, or the already non-displayed session-start nudge | Canonically classified text-only operational user messages stay ordinary semantic user messages but render through the zero-height adapter (verified on Pi 0.81.1 through 0.82.0) under Calm; legacy entries stay gaplessly controllable, and the session-start nudge retains its existing non-displayed custom-message path. | | `synthetic-assistant` | No authoritative Firstmate source found | Policy-hidden, but Pi exposes no generic assistant-role renderer. | | `unknown` | Future or unclassified transcript component | Policy-hidden, but no generic renderer exists; never claimed as covered. | The installed extension API has no supported global transcript filter, user-message renderer, assistant-message renderer, chat-container API, or generic custom-tool wrapper. -Pi 0.81.1 exports `AssistantMessageComponent` and `InteractiveMode`, so Calm uses separate exact-version, idempotent adapters for assistant thinking layout and the complete operational-user transcript row while leaving all message data and non-Calm rendering unchanged. +Pi 0.81.1 through 0.82.0 export `AssistantMessageComponent` and `InteractiveMode`, so Calm uses separate idempotent, API-probed adapters for assistant thinking layout and the complete operational-user transcript row while leaving all message data and non-Calm rendering unchanged; see the [compatibility contract](calm.md#pi-compatibility) for how a future Pi lacking one of those exports is handled. General component replacement, ANSI cursor erasure, provider-context mutation, and installed-file patching remain rejected as unsupported or preservation-breaking workarounds. ## Cross-harness verification record @@ -197,7 +207,7 @@ grok 0.2.106 (bde89716f679) | Claude Code 2.1.218 | Not feasible through the inspected supported project surface. | Project hooks can observe lifecycle and tool events, while the plugin CLI packages supported components; neither inspected surface exposes a transcript-row renderer or transcript-wide redraw API. | | Codex CLI 0.144.6 | Not feasible through the inspected supported project surface. | The tracked hooks expose session, pre-tool, and stop handling, while the plugin and feature inventories expose no TUI tool-row renderer or transcript redraw control. | | OpenCode 1.17.18 | Not feasible without violating the preservation boundary. | Plugins expose events and tool execution hooks, not a built-in transcript-row renderer; same-name tool replacement changes execution rather than presentation alone. | -| Pi 0.81.1 | Partially feasible with two exact-version exported-class adapters. | Public APIs control working visibility, collapsed labels, known tool slots, custom entries, and expansion redraws; exported assistant and interactive-mode classes provide the version-pinned collapsed-thinking and operational-user layout boundaries, while generic user, tool, and status filtering remains unavailable. | +| Pi (verified 0.81.1 through 0.82.0) | Partially feasible with two API-probed exported-class adapters. | Public APIs control working visibility, collapsed labels, known tool slots, custom entries, and expansion redraws; exported assistant and interactive-mode classes provide the collapsed-thinking and operational-user layout boundaries, gated on the exact method's presence rather than a version number, while generic user, tool, and status filtering remains unavailable. | | Grok CLI 0.2.106 | Not feasible through the inspected supported project surface. | Project hooks expose lifecycle and tool interception, while the plugin CLI exposes no row-renderer contract; `--minimal` changes the whole screen mode rather than selected transcript rows. | These conclusions are deliberately limited to the named versions and supported surfaces. @@ -215,7 +225,7 @@ The operational provider path covers Calm loaded on, loaded off, default prefere It asserts one persisted and rendered captain answer, exact user-role operational envelopes in order, no replacement custom messages, one processing result, zero operational transcript rows, and the two-row neighboring-assistant geometry for live, adjacent, and restart paths. Quoted current markers, ASCII-only labels, ordinary text before a marker, unrelated U+2063 placement, and image-bearing input remain visible in component and native transcript checks. `tests/fm-pi-primary-live-e2e.test.sh` also proves the unchanged built-in `Working...` row while Calm is active on the credentialed provider path before continuing its ordinary watcher lifecycle. -`tests/fm-pi-primary-types.test.sh` performs strict no-emit TypeScript checking against the installed Pi 0.81.1 declarations. +`tests/fm-pi-primary-types.test.sh` performs strict no-emit TypeScript checking against the installed Pi declarations, currently package version 0.81.1. The relevant commands are: @@ -256,3 +266,24 @@ FM_TEST_SUMMARY_FAMILY family=pure-contract-unit count=31 duration_ms=165384 fai $ tests/fm-pi-primary-live-e2e.test.sh skip: set FM_PI_LIVE_E2E=1 to run the isolated interactive Pi regression ``` + +## 2026-07-26 Pi 0.82.0 compatibility verification + +Pi 0.82.0 preserved both API-probed presentation seams and every deterministic Calm TUI guarantee. +The globally installed declaration package remained 0.81.1, so the strict typecheck continued to cover that earlier declaration-evidence version while the real CLI exercised 0.82.0. + +```text +$ pi --version +0.82.0 + +$ tests/fm-calm-pi-extension.test.sh +ok - Pi calm extension is presentation-only with one persisted visibility choice, no Calm status row, native working visibility, supported redraw controls, and the Firstmate watcher-tool integration +ok - Pi calm resolves its persistent home independently of Pi's launch directory +ok - Pi calm centralizes transcript visibility, preserves execution/export data, keeps native working visible, and persists its choice across session starts +ok - Pi operational follow-up E2E processes exact user-role notifications once while Calm hides current and adjacent rows, Calm off and absent render them, and restart preserves semantics +ok - Pi Calm native /skill:ahoy geometry keeps every collapsed thinking and tool block at zero height while preserving expansion, history, restart, and Calm-off rendering +ok - Pi calm native E2E keeps Working and captain turns visible, hides exact operational user rows without changing persistence, restores them Calm-off, survives restart, and preserves export plus Ctrl+O behavior + +$ tests/fm-pi-primary-types.test.sh +ok - tracked Pi extensions pass strict no-emit typecheck against Pi 0.81.1 +``` diff --git a/docs/calm.md b/docs/calm.md new file mode 100644 index 00000000000..8d63b6d0b56 --- /dev/null +++ b/docs/calm.md @@ -0,0 +1,37 @@ +# Pi Calm mode + +Calm is a Pi-only conversation presentation toggle. +It is off by default, and the last `/calm` choice persists for the effective Firstmate home across Pi session starts and resumes. + +While Calm is active, Pi's built-in `Working...` activity remains visible and no separate Calm status row is added. +Calm hides collapsed thinking labels, the shells for Pi's seven built-in tools, the `fm_watch_arm_pi` tool shell, and canonically classified Firstmate operational user rows. +The operational inputs remain ordinary user-role messages, while Pi's transcript layout renders their complete rows at zero height. +The session-start nudge remains on its existing non-displayed custom-message path. + +Calm changes presentation only. +Tool execution, input delivery, ordering, model context, session storage, diagnostics, and `/export` and `/share` operation remain unchanged. +Every hidden Firstmate input remains available to the model and in serialized session data and exported artifacts. +Legacy operational custom messages remain in session data and Pi's sidebar tree, although the main HTML transcript may omit them. +Toggling Calm off restores ordinary rendering, and `Ctrl+O` expansion state is preserved. + +Pi's supported presentation API does not expose a global transcript filter. +Expanded reasoning and its reserved spacing, built-in tool images, user-bash rows, skill and summary rows, generic status notices, and arbitrary custom-tool or extension rows remain visible. +These are supported-API boundaries rather than hidden-content failures. + +## Pi compatibility + +Calm has no numeric Pi version minimum or maximum and never refuses Pi solely because its version is newer than a previously verified version. +The collapsed-thinking and operational-user-row presentation adapters probe the exact Pi API seam they patch when Calm loads. +If Pi removes one of those seams, Calm logs a diagnostic naming the unavailable adapter and skips only that adapter; `/calm`, the other adapter, and unrelated Pi extensions remain available. + +[`calm-mode-feasibility.md`](calm-mode-feasibility.md) owns the version-scoped renderer taxonomy and empirical evidence. +[`configuration.md`](configuration.md#pi-calm-preference-configcalm) owns the persisted preference file and resolution rules. +`.pi/extensions/lib/fm-calm-visibility.ts` owns the visibility policy, and `.pi/extensions/lib/fm-calm-operational-user-layout.ts` owns the zero-height operational-user row adapter. + +Regression entry points: + +```sh +tests/fm-calm-pi-extension.test.sh +tests/fm-pi-primary-types.test.sh +FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh +``` diff --git a/docs/cd-guard.md b/docs/cd-guard.md index 2d8082e1ce8..998a9b540c1 100644 --- a/docs/cd-guard.md +++ b/docs/cd-guard.md @@ -74,13 +74,13 @@ It does not permit `cd /home/project`, because an absolute-path `cd` remains a p ## Transport and fail-open behavior -`bin/fm-cd-pretool-check.sh` supports all five harness entry shapes used by the tracked adapters: +`bin/fm-cd-pretool-check.sh` supports all five harness-engine entry shapes used by the tracked adapters, with pi-signed sharing Pi's shape: - Claude sends stdin JSON at `.tool_input.command` and adds `--claude` to preserve Claude's stderr-only deny requirement. - Codex sends stdin JSON at `.tool_input.command` without `--claude`. - Grok sends stdin JSON at `.toolInput.command`. - OpenCode sends the exact command string through `--command <exact string>`. -- Pi sends the exact command string through `--command <exact string>`. +- Pi and pi-signed send the exact command string through `--command <exact string>`. Processing order is cheapest-first: a strict-superset prefilter, then the primary-checkout scope, then the Node policy owner. The prefilter removes ordinary single quotes, double quotes, backslashes, carriage returns, and newlines before fast-allowing any command that carries no `cd`, `pushd`, or `popd` substring and no quoting-decoder marker (`$'` ANSI-C or `$"` locale), so quoted or escaped command-word fragments delegate to the policy while most commands never pay for the git scoping calls or the Node process. @@ -99,7 +99,7 @@ Identical in shape to `docs/arm-pretool-check.md`: - `--claude` suppresses stdout completely because Claude ignores a PreToolUse deny when stdout is nonempty. - Codex blocks on exit 2 and displays stderr. - OpenCode throws only when the checker exits 2. -- Pi returns `{block: true}` only when the checker exits 2. +- Pi and pi-signed return `{block: true}` only when the checker exits 2. ## Shared classifier ownership diff --git a/docs/cmux-backend.md b/docs/cmux-backend.md index 5859e7f2d34..4bd438bc74e 100644 --- a/docs/cmux-backend.md +++ b/docs/cmux-backend.md @@ -1,374 +1,130 @@ -# cmux runtime backend (experimental) +# cmux runtime backend -This document records the empirical verification behind `bin/backends/cmux.sh`, the cmux session-provider adapter. -It is the cmux equivalent of the tmux facts recorded in the `harness-adapters` skill and of `docs/herdr-backend.md`'s/`docs/zellij-backend.md`'s/`docs/orca-backend.md`'s facts for those backends. - -cmux is [a Ghostty-based macOS terminal](https://cmux.com) built for AI coding agents, with vertical tabs, notifications, and a CLI/socket JSON-RPC control API (`cmux <verb> ...`). -Verified against the real installed app: cmux 0.64.17 (build 97), macOS aarch64. -The feasibility investigation that preceded this build (`data/cmux-backend-feasibility-c7/report.md`) verified the app's CLI surface from source only, flagging a live install-and-poke pass as the remaining gate; that pass is what this document and `tests/fm-backend-cmux-smoke.test.sh` record. -All real-cmux verification here and in the smoke test creates only `fm-test-`-prefixed task workspaces, with one documented exception: the manual last-in-window verification also creates the unnamed default sibling cmux requires to close that task workspace. -It never enumerates-and-closes, touches no existing workspace, closes only its own `fm-test-` task workspaces, and never quits or relaunches the app - the same discipline `tests/herdr-test-safety.sh`/`tests/zellij-test-safety.sh` established for their backends, adapted in `tests/cmux-test-safety.sh` to cmux's shape (there is no isolated, throwaway session to spin up - cmux is one shared, GUI-first app instance, the same posture as Orca). +cmux is an experimental macOS GUI terminal backend. +It provides task workspaces and surfaces while Treehouse continues to provide git worktrees. +[`configuration.md`](configuration.md#runtime-backend-configbackend--fm_backend) owns shared selection and metadata semantics. ## Setup -Pick cmux if you already run it as your terminal and want firstmate crew tabs to live there instead of tmux. -cmux is **macOS-only** and **GUI-first** - selecting this backend means a real GUI window exists and is running, exactly like Orca's posture. +Pick cmux when you already use the app as your terminal and want task workspaces in its sidebar. +cmux is macOS-only, GUI-first, and unsuitable for a headless or SSH-only Firstmate session. Prerequisites: -- The cmux app itself, installed from [cmux.com](https://cmux.com) or `brew install --cask cmux`, version 0.64.17 or newer. -- `jq`, required to parse cmux's JSON output: `brew install jq` (or your platform's package manager). -- The universal firstmate prerequisites - a verified crew harness plus the required toolchain, owned by [`docs/configuration.md`](configuration.md) ("Harness support", "Toolchain"); treehouse still provides the worktree, cmux only provides the session. -- The cmux CLI binary is not guaranteed to be on `PATH` after a plain app install (see "CLI is not on PATH by default" below) - the adapter falls back to the well-known bundle path automatically, so this is not a blocker, just something to be aware of if you want to run `cmux` yourself from a shell. - -**One-time socket access setup (required, not optional):** cmux's control socket defaults to `automation.socketControlMode: "cmuxOnly"`, which rejects any CLI process not spawned inside cmux itself - firstmate always drives cmux from an external shell, so this must be changed before `backend=cmux` can work at all. -Settings > Automation offers five Socket Control Mode values; their exact semantics were verified from cmux source (see "Socket control modes: the full matrix" below for the enforcement points and evidence): - -| Settings label | JSON value | Works for firstmate? | What gates an external client | -|---|---|---|---| -| Off | `off` | no | The socket listener is never started at all. | -| cmux processes only | `cmuxOnly` (the default) | no | A connect-time check that the peer process is a descendant of the cmux app - an external shell never is. | -| Automation mode | `automation` | **yes - recommended** | Only the socket file's owner-only (0600) permissions: any process of YOUR macOS user connects, with no password and no ancestry check. | -| Password mode | `password` | yes, with a password | An `auth <password>` handshake required before any command; socket file owner-only (0600). | -| Full open access | `allowAll` | yes, **not recommended** | Nothing: no auth, and the socket file is world-writable (0666), so EVERY local user can drive cmux. | - -**Recommendation: Automation mode.** -It is the least-friction viable mode, and on a single-user machine it is not materially weaker than Password mode: both expose the socket only to processes of your own macOS user (0600 socket file), and the password's one extra defense - same-user processes that do not know the secret - is largely illusory, because the password itself must sit in a same-user-readable file (`config/cmux-socket-password`, or cmux's own state-dir password file that its CLI auto-reads) for automation to use it at all. -Password mode buys real friction (a secret to set, distribute to firstmate, and rotate) for that marginal defense; pick it if your threat model includes untrusted same-user processes that cannot read your config files. -Full open access hands the socket - which can open workspaces and run arbitrary commands in them - to every local user on a multi-user machine; cmux's own settings UI calls it unsafe, and it buys firstmate nothing over Automation mode (firstmate always runs as your own user), so choose it only as a deliberate, understood trade-off, never as a default. - -To set it up: - -1. Open cmux's Settings > Automation. -2. Set **Socket Control Mode** to **Automation mode** (recommended). -3. There is no step 3 - no password to configure or distribute. - -If you prefer **Password mode** instead: set the mode and a password in Settings > Automation, then make that same password available to firstmate - either as the first line of a local, gitignored `config/cmux-socket-password` file under the effective config directory, or exported as `CMUX_SOCKET_PASSWORD` in the environment firstmate runs in. -`config/cmux-socket-password` is the durable choice; the adapter reads it fresh on every call from `${FM_CONFIG_OVERRIDE:-$FM_HOME/config}` and passes it through without ever overriding an operator's own ambient `CMUX_SOCKET_PASSWORD` when the file is absent. -A configured password is harmless if you later switch to Automation mode: cmux's CLI sends the `auth` handshake preemptively and tolerates the server's "Unknown command 'auth'" reply in non-password modes (verified from source, `cli/cmux.swift` `authenticateSocketClientIfNeeded`). -Do not edit `~/.config/cmux/cmux.json` by hand for any of this: the mode change cannot be applied over the socket that is itself still rejecting connections, and the app's config writer drops a hand-added `socketPassword` key entirely (see "Socket control modes" below for that finding). - -Ask the firstmate crew to select cmux by putting `cmux` in a local `config/backend` file - the durable way to pick it - or by exporting `FM_BACKEND=cmux` for a one-off session; telling the first mate in chat to use cmux also works. -cmux is also selected by **runtime auto-detection**: a firstmate process itself running inside a cmux-spawned terminal (`CMUX_WORKSPACE_ID` set - or, when cmux's bundled claude wrapper stripped that marker, the bundle-id/ancestry fallback signals - checked after `$TMUX`/`HERDR_ENV=1` since cmux is the outermost terminal application, not a nestable multiplexer) spawns new tasks into cmux by default, with no config needed, exactly like herdr's own auto-detection - see "Runtime auto-detection" below. -Auto-detection only ever picks a SESSION provider; it never touches the one-time socket-access setup above, which stays required regardless of how cmux was selected. -A cmux spawn refuses loudly, with an actionable message pointing back to this document, if the app is unreachable, the socket rejects the connection (`cmuxOnly` mode still active), or a password is required but not configured or was rejected; the refusal names every viable mode with Automation mode as the recommendation, plus the `config/backend`/`--backend tmux` opt-out for a caller who ended up on cmux only because auto-detection picked it. - -No first-run provisioning beyond the socket-access setup above and having `jq` installed; firstmate creates the workspace it needs on first spawn, launching the app itself (`open -a cmux`) if it is not already running. - -Watching and attaching: firstmate uses one workspace per task in whatever cmux window is currently open. -Task selectors resolve through the shared contract owned by [`docs/configuration.md`](configuration.md) ("Runtime backend"), while the actual cmux workspace title is home-scoped as `fm-<home-label>-<id>`, for example `fm-firstmate-<8hex>-cmux-e2e-t1` in the primary home or `fm-2ndmate-<secondmate-id>-<8hex>-cmux-e2e-t1` in a secondmate home. -You do not need to bring the window forward for routine supervision: from an active firstmate session, `bin/fm-peek.sh <id>` reads a task's surface without focusing it, and `FM_HOME=<this-firstmate-home> bin/fm-send.sh <id> "<text>"` steers it unless `FM_HOME` is already set to the active firstmate home - workspace/surface/pane creation all default `focus` to `false`, so an unattended spawn never steals your view. - -Verify it works by spawning a trivial task with `--backend cmux` and confirming the task's meta records `backend=cmux` plus `cmux_workspace_id=` and `cmux_surface_id=`. -The cmux sidebar should show a new `fm-firstmate-<8hex>-<id>` workspace in the primary home. - -Limitations: cmux is experimental, macOS-only, GUI-first (never viable for a headless/CI/SSH-only firstmate instance), has no native busy-state signal, and `--secondmate` spawns are refused until a per-home design exists - see "Known gaps left for a follow-up" at the end of this document. - -## Status: experimental - -cmux is experimental, exactly like every non-tmux backend in this design. -Select it by putting `cmux` in a local `config/backend` file, by exporting `FM_BACKEND=cmux`, by telling the first mate in chat to use cmux, or implicitly by runtime auto-detection when firstmate itself is already running inside a cmux-spawned terminal - see "Runtime auto-detection" below. -GUI-first and macOS-only stay unchanged by that: cmux is never a candidate for a headless/CI/SSH-only instance, because auto-detection can only fire from inside a live cmux terminal in the first place, which such an instance never is. -Absent `backend=` in a task's meta always means `tmux`; only a cmux task ever carries an explicit `backend=cmux` line. -A cmux spawn refuses loudly if the `cmux` CLI cannot be found, the installed version is older than the verified minimum (0.64), or the control socket is unreachable/unauthenticated (`fm_backend_cmux_version_check`, `fm_backend_cmux_ensure_running`). - -## Runtime auto-detection - -Verified from the shipped app source (`Packages/macOS/CmuxTerminal/Sources/CmuxTerminal/Spawn/TerminalSurface+StartupEnvironment.swift`'s `applyManagedCmuxContextEnvironment`, cloned read-only from `github.com/manaflow-ai/cmux` at the commit current on 2026-07-04): every terminal surface cmux spawns gets `CMUX_WORKSPACE_ID`, `CMUX_SURFACE_ID`, and `CMUX_SOCKET_PATH` (plus the legacy `CMUX_TAB_ID`/`CMUX_PANEL_ID` aliases) injected into its environment, and all five keys are marked `protectedKeys` - non-overridable by anything the spawned shell or its own env config does afterward. -cmux's own CLI corroborates this is a legitimate ambient-identity marker, not incidental: `cmux_open.swift` reads `CMUX_WORKSPACE_ID`/`CMUX_SURFACE_ID` from the environment as its own fallback target when a caller does not pass `--workspace`/`--surface`, exactly how `$TMUX` and `HERDR_ENV`/`HERDR_PANE_ID` work for their own backends. - -`fm_backend_detect` (`bin/fm-backend.sh`) checks `CMUX_WORKSPACE_ID` (non-empty) as the PRIMARY cmux marker, not `CMUX_SOCKET_PATH`: the latter is separately documented as a user-settable override for pointing the CLI at a non-default socket path, so its mere presence would not reliably mean "running inside a cmux-spawned terminal" the way `CMUX_WORKSPACE_ID` does. -Nesting still resolves innermost-first, exactly as it does for herdr: `$TMUX` is checked first, then `HERDR_ENV=1`, then the cmux checks last. -cmux is checked last deliberately, not because it is a "lesser" backend, but because it is a terminal application - the outermost layer, like iTerm2/Terminal.app - not a session multiplexer. -Both tmux and herdr can run nested inside a cmux-provided shell (someone starts a tmux or herdr session from within a cmux terminal), but cmux itself cannot run nested inside either of them, so whenever a multiplexer marker is present alongside a cmux signal, that multiplexer really is the innermost, currently-executing layer and must win. -An auto-detected cmux spawn prints the same loud stderr `NOTICE` herdr's auto-detection prints, naming the winning signal and the `config/backend`/`--backend tmux` opt-out; a fallback-signal detection (below) says so explicitly in that notice, so it is visibly distinct from the primary-marker case. - -Auto-detection selects the SESSION provider only. -It has no bearing on the one-time socket-access setup ("Setup" above): a viable `automation.socketControlMode` is still required for the very first cmux-backed spawn to succeed, auto-detected or explicit, and the existing loud spawn refusal (`fm_backend_cmux_ensure_running`) still fires when it is missing. -That refusal message names the viable modes and the `config/backend`/`--backend tmux` opt-out, so a captain who never explicitly chose cmux - and only landed on it because firstmate happened to be launched from inside a cmux terminal - gets a self-contained answer either way: finish the socket setup to actually use cmux, or opt out back to tmux. - -The original build's env-injection finding rested on the source read above alone; it has since been corroborated live (2026-07-04, cmux 0.64.17 build 97): the inherited environment of a tmux server started from a cmux tab on the reference machine carries `CMUX_WORKSPACE_ID`, `CMUX_TAB_ID`, `CMUX_SOCKET_PATH`, `CMUX_BUNDLE_ID`, and `__CFBundleIdentifier=com.cmuxterm.app` into every pane, and firstmate separately confirmed the full injected set on a live tab shell via `ps eww`. - -### The bundled claude wrapper strips `CMUX_*` (unanticipated, load-bearing finding) - -Verified live 2026-07-04 against the installed cmux 0.64.17 (build 97), macOS aarch64; the captain's app was not modified, relaunched, or reconfigured for any of this. - -`claude` typed in a cmux tab does not run the real binary: cmux prepends a per-surface shim directory (`$CMUX_CLAUDE_WRAPPER_SHIM_ROOT`) to `PATH`, resolving `claude` to `/Applications/cmux.app/Contents/Resources/bin/cmux-claude-wrapper`, a readable bash script. -Read from that shipped script, the wrapper has three exec paths: - -1. The hooks-injecting main path (in cmux, socket reachable): KEEPS every `CMUX_*` var and adds more (`CMUX_CLAUDE_PID`, launch metadata). -2. The hooks-disabled path (`CMUX_CLAUDE_HOOKS_DISABLED=1`): KEEPS `CMUX_*`, unsets only `CLAUDECODE`. -3. The passthrough path (not in cmux, OR the socket probe fails): when in cmux, runs `for cmux_key in "${!CMUX_@}"; do unset "$cmux_key"; done` plus `unset TERMINFO` (and `CLAUDECODE`) before `exec`'ing the real claude. - -The wrapper's socket probe is `CMUXTERM_CLI_RESPONSE_TIMEOUT_SEC=0.75 cmux --socket "$CMUX_SOCKET_PATH" ping`, with NO password. -Reproduced verbatim on this machine's live password-mode socket: `Error: ERROR: Authentication required - send auth <password> first`, exit 1. -So under Password mode - exactly the setup this document used to require - the probe always fails and the wrapper always takes the stripping passthrough path. - -The strip itself was reproduced end to end with a fake `claude` on `PATH` that dumps its environment, invoking the real wrapper with `CMUX_SURFACE_ID`, `CMUX_WORKSPACE_ID`, `CMUX_SOCKET_PATH` (the live socket), `__CFBundleIdentifier=com.cmuxterm.app`, `TERMINFO`, and `CLAUDECODE` set: the fake claude saw ONLY `__CFBundleIdentifier=com.cmuxterm.app` - every `CMUX_*` var, `TERMINFO`, and `CLAUDECODE` were gone. -The counterfactual run with `CMUX_CLAUDE_HOOKS_DISABLED=1` added preserved every `CMUX_*` var (only `CLAUDECODE` was unset), confirming the strip is specific to the passthrough path. - -Consequence: a claude-harness firstmate launched inside a cmux tab can have zero `CMUX_*` env, so `CMUX_WORKSPACE_ID` alone cannot be the whole detection contract. -Other harnesses launched from a cmux tab are unaffected (cmux ships no wrapper shims for them). - -### Fallback signals: bundle id first, then process ancestry - -When (and only when) `CMUX_WORKSPACE_ID` is absent - and `$TMUX`/`HERDR_ENV` did not already win - `fm_backend_detect` consults two macOS-only fallback signals, in order (`fm_backend_detect_cmux_fallback`, guarded on `uname` = `Darwin` since cmux itself is macOS-only): - -1. **Bundle id:** `__CFBundleIdentifier` equal to `com.cmuxterm.app`. - LaunchServices sets this app-identity variable for every process an app bundle launches, it is inherited down the process tree, and the wrapper does not strip it (verified in the fake-claude repro above). -2. **Process ancestry:** the parent chain from the current process reaches the running cmux app (`fm_backend_detect_cmux_app_is_ancestor`). - The app is resolved by bundle id, never a hardcoded install path: `lsappinfo info -only pid -app com.cmuxterm.app` printed `"pid"=44127` for the live app (and prints nothing, exit 0, for a non-running bundle id), with a bundle-shaped `ps` comm match (`*/cmux.app/Contents/MacOS/cmux`, any install location) as the fallback when lsappinfo cannot resolve a pid. - Live process-table facts recorded 2026-07-04: the cmux app runs as `/Applications/cmux.app/Contents/MacOS/cmux` (pid 44127, ppid 1), and its tab shells are parented through `/usr/bin/login` (e.g. `44204 44127 /usr/bin/login`), so a claude-under-cmux's chain is claude <- tab shell <- login <- cmux app. - -Which signal is authoritative when: - -- **Wrapper-stripped claude directly in a cmux tab** (the common case): both signals are present; the bundle id is checked first because it is a pure env read, and it is authoritative. -- **Environment-scrubbed launch under cmux** (an `env -i`-style invocation with no inherited `__CFBundleIdentifier`): ancestry is the only signal left, and it is authoritative. -- **Inside a tmux server that was started from a cmux tab**: ancestry is structurally UNUSABLE - the tmux server reparents to launchd (verified live: the reference machine's own cmux-started tmux server has ppid 1), so the walk can never reach cmux - while the bundle id IS inherited into every pane and WOULD false-positive. - `$TMUX` winning first is what keeps that correct; the fallbacks are never consulted when a multiplexer marker is present. - `tests/fm-backend.test.sh` pins this exact case (`test_backend_detect_cmux_fallback_tmux_nested_false_positive`), alongside the bundle-id, ancestry (pid and comm), non-Darwin-guard, and launchd-stop paths. -- **SSH sessions, cron, launchd agents**: neither signal fires - sshd/cron reset the environment (no bundle id) and their ancestry ends at launchd. - -The positive ancestry walk itself is exercised by fake `ps`/`lsappinfo` unit tests rather than live (running a probe process genuinely parented under the captain's live cmux tabs was judged too intrusive, the same posture as this document's screenshot note); every negative live fact above - the strip, the wrapper ping failure, the tmux reparenting, the bundle-id inheritance, the lsappinfo resolution shapes - was verified against the real machine on 2026-07-04. - -## Worktree provider stays treehouse - -cmux is a session provider only, exactly like herdr and zellij (unlike Orca, which also owns the task worktree). -Treehouse remains the worktree provider. -The feasibility report searched cmux's source for a shipped git-worktree-owning feature and found only a prototype (`Sources/ExtensionWorktreePrototype.swift`) that is not wired into any CLI verb - `workspace.create --cwd <path>` just opens a terminal at an existing directory with no opinion about how that directory came to exist. +- cmux 0.64 or newer, installed from [cmux.com](https://cmux.com) or with `brew install --cask cmux`. +- `jq` for JSON responses. +- The universal harness and toolchain requirements in [`configuration.md`](configuration.md#toolchain). -## Task container shape: one workspace per task, one surface +The CLI is not always installed on `PATH` with the app. +The adapter prefers `command -v cmux` and otherwise uses `/Applications/cmux.app/Contents/Resources/bin/cmux`. -cmux's hierarchy is macOS window -> workspace (a vertical-tab entry, cmux's rough analogue of a herdr/zellij tab) -> surface (a pane/split within that workspace). -There is no "session" concept to multiplex the way tmux/herdr/zellij have - there is just "the app" (one running GUI instance, optionally split across native macOS windows). -firstmate uses **one cmux workspace per task**, keyed by the caller-facing `fm-<id>` label, with exactly one surface inside it - mirroring tmux's one-window-per-task and zellij's one-tab-per-task shape. -The caller-facing task label stays `fm-<id>`, but the visible cmux workspace title is `fm-<home-label>-<id>`. -The home label keeps the same readable identity as herdr's workspace split - `firstmate` for the primary home, or `2ndmate-<id>` when `$FM_HOME/.fm-secondmate-home` contains a secondmate id - and appends a short stable hash of the resolved `FM_ROOT` path. -That yields labels like `firstmate-<8hex>` or `2ndmate-<id>-<8hex>`, making the visible workspace title `fm-firstmate-<8hex>-<id>` or `fm-2ndmate-<id>-<8hex>-<task>`. -This was hardened in two captain-directed no-mistakes review gate follow-ups: first by adding the home tag for primary-vs-secondmate collisions, then by adding the `FM_ROOT` hash so two distinct primary installations cannot collide either. -Physically moving or relocating a firstmate installation changes its tag, so workspaces titled under the old tag stop matching after a move. -That is acceptable because a task's own recorded worktree path in `state/<id>.meta` does not survive a repo relocation either, so this is consistent with an existing, already accepted limitation, not a new one. -There is still no per-home cmux container split (unlike herdr's later refinement); the home tag is a title discriminator only. +### Required socket access -## Target string and meta fields +cmux defaults to a control mode that rejects external shells, while Firstmate always controls it from an external process. +Open Settings > Automation and choose a viable Socket Control Mode before the first cmux-backed spawn. -A cmux task's `window=` meta field holds `<workspace_uuid>:<surface_uuid>`, for example `F28BB910-E42C-40F6-AC5C-D92635581EED:A3E9D3A8-BE1D-4055-A567-3525320D2ABF`. -Both are bare UUIDs with no embedded colon, so splitting on the first colon is trivially correct (mirrors herdr's/zellij's target-string convention). -The meta target is still the UUID pair, not the human title. -The human title is reconstructed internally from the caller-facing `fm-<id>` label as `fm-<home-label>-<id>` whenever cmux needs to create, recover, or list a workspace. -`<home-label>` includes the readable home prefix and the short `FM_ROOT` path hash described above. -cmux tasks additionally record: +| Setting | Value | Firstmate support | Security boundary | +| --- | --- | --- | --- | +| Off | `off` | No | The socket listener is disabled. | +| cmux processes only | `cmuxOnly` | No | Only descendants of the cmux app can connect. | +| Automation mode | `automation` | Yes, recommended | The owner-only 0600 socket admits processes of the current macOS user. | +| Password mode | `password` | Yes | The 0600 socket also requires an auth handshake. | +| Full open access | `allowAll` | Yes, not recommended | The 0666 socket admits every local user without authentication. | -- `cmux_workspace_id=` - the task's workspace UUID (same value as the `window=` field's first component). -- `cmux_surface_id=` - the task's surface UUID (same value as the `window=` field's second component). +Automation mode is the recommended same-user boundary. +`allowAll` can execute commands through a world-writable control socket and should be selected only as an explicit security tradeoff. -No session field is needed - unlike herdr/zellij there is no session layer to record. +For Password mode, store the password as the first line of local gitignored `config/cmux-socket-password` or provide `CMUX_SOCKET_PASSWORD` in Firstmate's environment. +The adapter reads the file fresh from the effective config directory and does not overwrite an ambient password when the file is absent. +Configure the mode and password through the cmux UI rather than editing `cmux.json`; the app does not retain a hand-added password key, and socket-based reload cannot fix a socket that is rejecting the caller. -## Verified CLI facts +Select cmux with local `config/backend` containing `cmux`, `FM_BACKEND=cmux` for one launch, or an explicit request to Firstmate. +It can also be runtime auto-detected when Firstmate itself runs inside cmux. +A spawn stops with an actionable setup message when the app, minimum version, `jq`, socket access, or password is unavailable. +The adapter may launch the app with `open -a cmux` only when the socket is down; it does not relaunch the app for access-denied or authentication errors. -| Operation | Verified cmux call | What was verified | -|---|---|---| -| Version gate | `cmux version` -> `"cmux 0.64.17 (97) [9ed29d81a]"` | Works with NO socket connection at all - a pure client-version check, verified even while the socket was still rejecting connections. | -| Reachability/auth gate | `cmux ping` -> `"PONG"` or a typed error | Classified into `ok`\|`denied`\|`unauth`\|`down`\|`error` from the error text (`fm_backend_cmux_ping_state`); `fm_backend_cmux_ensure_running` launches the app (`open -a cmux`) only for `down`, and fails fast with an actionable message for `denied`/`unauth` since relaunching cannot fix a configuration problem. | -| Duplicate task check | `cmux workspace list --json --id-format uuids`, match by home-scoped `.title` | cmux enforces NO title uniqueness for workspaces OR surfaces/tabs - verified live: two workspaces, and two surfaces within one workspace, all created successfully sharing one title. The adapter's own duplicate check is required, mirroring herdr/zellij, and it checks the scoped title such as `fm-firstmate-<8hex>-<id>`. | -| Create task workspace | `cmux new-workspace --name <scoped-title> --cwd <dir> --focus false --id-format uuids` | Creates a workspace with exactly one default surface. `--focus` verified to already default to `false` for workspace/surface/pane creation - no focus-restore dance needed, unlike zellij. The caller passes `fm-<id>`, but the adapter creates `fm-<home-label>-<id>`. | -| Workspace/surface id resolution | `cmux workspace list --json --id-format uuids` (find by home-scoped title), then `cmux list-panes --workspace <id> --json --id-format uuids` (`.panes[0].selected_surface_id`) | A freshly created workspace already has exactly one surface, so no separate `new-surface` call is needed. `--id-format uuids` (or `both`) is required to get a bare `id` field in JSON; the default JSON shape returns only short `ref` strings like `"workspace:2"`. | -| Liveness / target readiness | `cmux list-panes --workspace <id> --json --id-format uuids`, checking the surface id appears in `.panes[].surface_ids` | Structural existence check, NOT a content read - see "read-screen fails on a genuinely fresh surface" below for why `read-screen` cannot be used here. Verified reliable on a completely untouched fresh surface, unlike `read-screen`. | -| Send literal (unsubmitted) | `cmux send --workspace <id> --surface <id> -- <text>` | Verified live: does NOT auto-submit - text sits at the prompt, unexecuted, until a separate Enter. Matches every other backend's "literal-then-separate-Enter" contract. The `--` separator keeps option-shaped text such as `--help` literal. | -| Send key | `cmux send-key --workspace <id> --surface <id> <key>` | Verified names: `enter`, `escape`, `ctrl-c` all work directly (lowercase, hyphenated). Escape is natively supported (unlike Orca); Ctrl-C correctly interrupted a running `sleep 100` in a live test. cmux's own key vocabulary is richer still (`ctrl-d`/`ctrl-z`/`ctrl-\\`, semantic aliases `sigint`/`sigtstp`/`sigquit`), but firstmate's shared vocabulary only needs these three today. | -| Send + submit, composed | `send` then `send-key enter` | cmux has no single-call atomic "type and submit" primitive (unlike tmux's `send-keys ... Enter` or herdr's `pane run`); `fm_backend_cmux_send_text_line` composes the two calls, mirroring zellij's equivalent composition. | -| Bounded capture | `cmux read-screen --workspace <id> --surface <id> --scrollback --lines <N> --json`, trimmed locally with `tail` | No herdr-style small-N empty-result bug: N=1..10 all verified to return correctly-clamped, non-empty content on an already-interacted-with surface. A single call is still bounded by the surface's actual current viewport height regardless of the requested `--lines` value (verified: capped at 16 rows in a headless/no-attached-window test run), so "fetch generous, trim locally" is kept for consistency even though the specific herdr bug does not reproduce. | -| Worktree-path discovery | marked active cwd probe + capture-scrape (`fm_backend_cmux_current_path`), NOT `current_directory` | `current_directory` DOES reflect a `cd` run directly in the surface's own top-level shell, but stays FROZEN at wherever that shell was when it launched a foreground subshell (exactly what `treehouse get` does) - zellij-shape, not herdr-shape. See "Worktree-path discovery: current_directory does not track a subshell" below. | -| Busy state | *(no native primitive)* | cmux has agent-awareness elsewhere (Claude Code hooks integration, session-resume tokens) but exposes nothing over the socket API for generic busy/idle classification; `surface.health`/`surface-health` is render health, not agent status. `fm_backend_busy_state`'s dispatcher (`bin/fm-backend.sh`) falls through to `unknown` for cmux via its wildcard case, exactly like tmux/zellij/Orca - the watcher's existing pane-hash + regex path is the only busy-state source for this backend. | -| Kill | `cmux close-workspace --workspace <id>`, preceded by a throwaway `new-workspace --window <win> --focus false --id-format uuids` when the target is the only workspace in its window | See "Closing the last workspace in a window" below. The backend owns the whole task workspace; kill closes it best-effort (`\|\| true`), but cmux silently refuses to close the LAST workspace in a window, so kill first detects that case (`fm_backend_cmux_window_of_workspace`) and adds a throwaway sibling before closing, matching every other backend's `kill` contract. | -| Recovery / list-live | `cmux workspace list --json --id-format uuids`, filter titles starting with this home's `fm-<home-label>-`, then `list-panes` per match for the surface id | Title-based, never trusts a stored workspace uuid blindly - ids do NOT survive an app relaunch (see "Workspace ids do not survive a relaunch" below), so this is the only safe recovery posture. The adapter prints the plain `fm-<id>` label back to callers after stripping the readable home tag and `FM_ROOT` hash. | +Routine supervision uses `bin/fm-peek.sh <id>` and `FM_HOME=<home> bin/fm-send.sh <id> '<text>'` without bringing the cmux window forward. +Task workspace and surface creation use `focus=false`. -## Socket control modes: the full matrix (default `cmuxOnly` rejects external CLIs) +Verify setup by spawning a small task and confirming metadata contains `backend=cmux`, `cmux_workspace_id=`, and `cmux_surface_id=`. -Not anticipated by the feasibility report (which verified the CLI surface from source only, without a live socket connection): cmux's control socket, by default, **rejects any client process that was not itself spawned inside cmux**. -Verified live (2026-07-03 pass): running any socket-backed CLI command (`cmux ping`, `cmux workspace list`, etc.) from an ordinary external shell - exactly how firstmate always drives cmux - returned `Error: ERROR: Access denied - only processes started inside cmux can connect`. +## Runtime detection -The setting is `automation.socketControlMode` in cmux's settings (`~/.config/cmux/cmux.json` or Settings > Automation), with values `off`, `cmuxOnly` (the default), `automation`, `password`, and `allowAll` (three more legacy aliases - `openAccess`, `fullOpenAccess`, `full` - normalize onto `allowAll`; `notifications` normalizes onto `automation`). +`CMUX_WORKSPACE_ID` is the primary cmux runtime marker. +`CMUX_SOCKET_PATH` is not sufficient because operators may set it outside cmux. +Detection checks tmux first, then Herdr, then cmux, so a multiplexer nested inside cmux remains the active backend. -The per-mode enforcement was traced through the shipped source on 2026-07-04 (`github.com/manaflow-ai/cmux` at commit `9c91710e3f58`, cloned read-only as scratch; verified from source, not live, except where noted - the reference machine's live app is the captain's own, in Password mode, and was not reconfigured to exercise the other modes). -There are exactly four enforcement points, and NO per-command/verb restrictions by mode - a mode either admits a client fully or not at all: +cmux's bundled Claude wrapper can remove every `CMUX_*` variable when its internal socket probe fails, including in Password mode. +On macOS only, detection therefore falls back first to `__CFBundleIdentifier=com.cmuxterm.app`, then to process ancestry reaching the running cmux app. +Those fallbacks are consulted only when neither tmux nor Herdr already won. +An environment-scrubbed or launchd-reparented process with no reliable marker is not auto-detected. -- **Listener start** (`Sources/AppDelegate.swift`, `socketListenerConfigurationIfEnabled`): `off` means the listener is never started; every other mode starts it. -- **Socket file permissions** (`SocketControlMode+SocketControl.swift`, `socketFilePermissions`): `allowAll` chmods the socket 0666 (every local user); all other modes 0600 (owner only). -- **Connect-time ancestry check** (`Sources/TerminalController.swift`, `handleClient`): applied ONLY when the mode is `cmuxOnly` - the peer pid must be a process-tree descendant of the cmux app, which an external shell never is (the live 2026-07-03 rejection above is this check firing). -- **Password handshake** (`Sources/TerminalController.swift`, `authResponseIfNeeded`, gated on `requiresPasswordAuth`, true ONLY for `password`): every command line before a successful `auth <password>` gets an auth error; the three auth-failure texts are "Authentication required - send auth <password> first" (no password presented; also reproduced live on this machine's password-mode socket, 2026-07-04), "Password mode is enabled but no socket password is configured in Settings." (app side has no password), and "Invalid password" (wrong password presented). +Auto-detection selects only the backend. +It never changes socket access or grants credentials. +The spawn refusal explains how to finish cmux setup or opt back into tmux. -So for an external, same-user CLI client like firstmate: `automation` admits it with no credential (its only gate is the 0600 socket file), `password` admits it after the handshake, `allowAll` admits it and every other local user too, and `off`/`cmuxOnly` never admit it. -`automation` is the recommended mode for firstmate, with `password` and `allowAll` as supported alternatives - the "Setup" section above carries the decision rationale. -This build's earlier pass (2026-07-03) chose `password` as "the minimum change that unblocks external CLI access"; the matrix trace above superseded that with `automation`, which reaches the same same-user boundary without the shared-secret friction. +## Task shape and metadata -A real wrinkle found during the 2026-07-03 password-mode setup, still true and worth keeping: **`automation.socketPassword` cannot be set durably through `cmux.json`** - the app's own config writer normalizes the file on reload/restart and drops the `socketPassword` key entirely (it is kept only in a dedicated password store/Settings, not the plaintext JSON), so editing the file to include a password has no lasting effect and cmux's own `cmux reload-config` cannot apply the `socketControlMode` change either, because reload-config is itself a socket call, and the socket is what needs the mode change to accept it in the first place. -The practical path (and what "Setup" above describes) is: set the mode (and password, if choosing Password mode) once through Settings > Automation (a GUI action - the socket-access chicken-and-egg problem does not exist there), then, for Password mode, supply that same password to firstmate via `config/cmux-socket-password` or `CMUX_SOCKET_PASSWORD`. -`CMUX_SOCKET_PASSWORD` in the CLI **client's** own environment is confirmed sufficient (per cmux's own CLI contract, it is the documented fallback when `--password` is absent) - `fm_backend_cmux_cli` exports it only when a value is actually configured, so an operator's own ambient `CMUX_SOCKET_PASSWORD` is never clobbered with an empty value. -cmux's CLI also auto-reads the app's own password file (`socket-control-password` in the cmux state directory) when neither `--password` nor the env var is set, but do not rely on that for firstmate: on the reference machine the app's password lives only in the password store (no state-dir file exists), so the explicit `config/cmux-socket-password`/`CMUX_SOCKET_PASSWORD` supply is the dependable path. +Each task owns one cmux workspace with one surface. +The caller-facing label remains `fm-<id>`, while the visible workspace title is `fm-<home-label>-<id>`. +The home label is `firstmate` or `2ndmate-<id>` plus a stable short hash of the resolved Firstmate root. +cmux does not enforce title uniqueness, so create, recovery, list, and cleanup paths all validate this scoped title. +Relocating the Firstmate installation changes the hash and leaves old titles unmatched, consistent with recorded worktree paths also becoming stale. -`fm_backend_cmux_ping_state` classifies the resulting failure text into `denied` (`cmuxOnly` still active) or `unauth` (password mode active but no/wrong password presented, covering all three auth-failure texts above), and `fm_backend_cmux_ensure_running`/`fm_backend_cmux_version_check`'s callers surface an actionable message naming the viable modes and pointing back to this document for either state - never a generic "is cmux installed?" message, and never a retry-via-relaunch (relaunching the app cannot fix a socket-mode/password configuration problem). -`off` is indistinguishable on the wire from "app not running" (`Socket not found`, classified `down`), so the launch-and-wait path's timeout message names the possibility that the app is running with its socket off. - -## `read-screen` fails on a genuinely fresh surface (unanticipated, load-bearing finding) - -Not anticipated by the feasibility report or by the original design sketch (which proposed using `read-screen` as the liveness probe, mirroring Orca's `fm_backend_orca_capture` doubling as its own liveness check). -Verified live: `cmux read-screen --workspace <id> --surface <id>` against a surface that was JUST created and has never been written to yet fails outright with `Error: internal_error: Failed to read terminal text` - for every `--lines` value tried (including no `--lines` flag at all), and regardless of how long you wait (retried up to several seconds later, still failing). -The moment a single `send` actually writes to that same surface, `read-screen` becomes reliably readable forever after. - -This ruled out `read-screen` as the liveness/readiness probe: the very first `send_literal` call on a freshly created task's surface would fail its own pre-flight readiness check before ever getting to write anything, making every task un-spawnable. -`cmux list-panes --workspace <id> --json --id-format uuids`, checking the target surface id appears in `.panes[].surface_ids`, has no such gap - verified correct and immediate on a completely untouched fresh surface - so `fm_backend_cmux_target_ready` uses that instead, mirroring zellij's own structural `pane_exists` check rather than Orca's read-based liveness pattern. - -## Worktree-path discovery: `current_directory` does not track a subshell (zellij-shape, not herdr-shape) - -Verified live, step by step, mirroring the exact test that caught this for zellij: - -1. A plain `cd /var` typed directly into a surface's own top-level shell updates `cmux workspace list --json`'s `current_directory` field immediately. -2. Running a nested subshell as a foreground command (`bash -c 'cd /Users && exec bash'`, standing in for `treehouse get`'s own nested interactive subshell) and confirming on-screen (via `pwd` typed inside the now-interactive nested shell) that it truly is in the new directory - `current_directory` stays **frozen** at `/var`, the directory the TOP-LEVEL shell was in when it launched the subshell. It never updates once a subshell has taken over as the surface's foreground process. - -This is the same shape zellij's `pane_cwd` has, not herdr's live-tracking `foreground_cwd`. -**Workaround, `fm_backend_cmux_current_path`:** reuses zellij's own active pwd-marker-probe technique verbatim in spirit - submit a begin marker, `pwd`, and an end marker via `send_text_line`, briefly settle, capture, and read only the lines between the markers. -Verified against the real binary in both shapes: a direct `cd` in the surface's own shell, and a nested subshell's own `cd` (the load-bearing case matching `treehouse get`'s actual shape) - both confirmed correct in `tests/fm-backend-cmux-smoke.test.sh`. - -## Closing the last surface: a third shape (unanticipated finding) - -The design sketch anticipated two possibilities for closing a workspace's last surface - herdr-shape (auto-closes the whole workspace) or zellij-shape (leaves an empty "ghost" workspace) - and planned to verify which one live. -Neither turned out to be correct: cmux implements a **third** shape. -Verified live: `cmux close-surface --workspace <id> --surface <id>` against a workspace's LAST remaining surface **refuses outright** with a typed error, `Error: invalid_state: Cannot close the last surface`, leaving both the surface and the workspace completely untouched - no partial state, no ghost. -`cmux close-workspace --workspace <id>` against that same workspace succeeds cleanly, removing the whole workspace (surface included) in one call, only when it is not the last workspace in its window. - -Since every firstmate cmux task uses exactly one owned workspace, `close-workspace` remains the correct teardown primitive. -The next section ("Closing the last workspace in a window") owns the last-in-window exception and `fm_backend_cmux_kill`'s best-effort workaround, which still reclaims every surface in the task workspace. - -## Closing the last workspace in a window (the selected-workspace teardown fix) - -Verified live 2026-07-10 against the installed cmux 0.64.17 (build 97), macOS aarch64, socket in `automation` mode; the captain's app was not modified, relaunched, or reconfigured, and only `fm-test-` task workspaces and throwaway default workspaces were touched. - -The incident this fixes: a cmux-backed task's teardown left its workspace open because it was the currently selected workspace, and the crew closed it by hand. -`close-workspace` cleanly removes a workspace ONLY when that workspace is not the last one in its window. -cmux keeps every window at one or more workspaces, so `close-workspace` against the ONLY workspace in a window silently no-ops: it still prints `OK`, but the workspace stays. -The last workspace in a window is always the selected one, so from the outside this reads as "the selected task workspace would not close" - but being selected is not itself the trigger. -A workspace that is selected while sharing its window with another workspace closes normally (verified); being the last workspace in its window is the actual trigger. - -Evidence (workspace refs are session-relative, shown as observed): - -``` -# control: a NON-last workspace (its window holds another workspace too) closes cleanly -$ cmux close-workspace --workspace <ws-A> -OK workspace:2 -# -> <ws-A> gone from `workspace list` - -# the bug: the LAST/ONLY workspace in its window -$ cmux close-workspace --workspace <ws-B> -OK workspace:7 -# -> <ws-B> STILL PRESENT; its window still reports workspace_count=1 (silent no-op) -``` - -Neither window-closing primitive rescues it, because a window holding a live terminal session cannot be closed over the control socket: - -``` -$ cmux close-window --window <win> -OK -# -> <win> and its workspace STILL PRESENT - -$ cmux rpc window.close '{"window_id":"<win>"}' -{ "window_id" : "<win>", "window_ref" : "window:2" } -# -> STILL PRESENT -``` - -Exiting the surface's shell does not help either: cmux immediately respawns a fresh shell in a new surface (the surface id changes), so the last workspace/window is never left empty to collapse on its own. - -The reliable primitive is `close-workspace` on a workspace that is NOT the last in its window, so `fm_backend_cmux_kill` makes the target non-last first. -`fm_backend_cmux_window_of_workspace` walks `list-windows --json` and each window's own `workspace list --json --window <id>` to find the target's window and count the membership-confirming workspace-list response. -When the count is one (last in window), kill creates a throwaway sibling in that same window - `new-workspace --window <win> --focus false --id-format uuids`, an unnamed default that never carries an `fm-<home>-` title, so recovery and `list_live` ignore it - and only then closes the target. -When the count is greater than one, kill closes the target directly, exactly as before, with no sibling. - -``` -# the fix, end to end: real fm_backend_cmux_kill on a SELECTED, last-in-window fm-test task workspace -$ fm_backend_cmux_window_of_workspace <ws-B> -<win> 1 -$ cmux new-workspace --window <win> --focus false --id-format uuids -OK workspace:9 -$ cmux close-workspace --workspace <ws-B> -OK workspace:8 -# -> <ws-B> GONE; <win> survives with a fresh default workspace (title "zsh"/"~", never fm-*) +```text +backend=cmux +window=<workspace-uuid>:<surface-uuid> +cmux_workspace_id=<workspace-uuid> +cmux_surface_id=<surface-uuid> ``` -The window keeping a fresh default workspace is cmux's own "closed the last tab" outcome, not extra firstmate state; it is the closest reachable result to removing the task's workspace, given a window cannot be socket-closed. -The helper and both branches are pinned in `tests/fm-backend-cmux.test.sh` by `test_window_of_workspace_finds_window_and_count`, `test_window_of_workspace_empty_when_not_found`, `test_kill_closes_workspace_directly_when_not_last`, and `test_kill_adds_sibling_when_last_in_window`. -The live `window_of_workspace` window/count detection is pinned in `tests/fm-backend-cmux-smoke.test.sh`. -The last-in-window path is not driven end to end in the automated smoke suite because closing the last workspace inherently leaves a window cmux cannot close over the socket, so a live end-to-end run cannot self-clean; the manual run recorded above is its empirical proof instead. +The UUID pair is the active endpoint authority within one app run. +Workspace UUIDs are not stable across an app relaunch, so recovery searches by the scoped title and then resolves the current surface id. -Related current-window scoping, observed during this work and left out of scope for this fix: `workspace list --json` WITHOUT `--window` is scoped to the CURRENT window only (verified live). -`fm_backend_cmux_window_of_workspace` passes `--window` per window and is unaffected, but `fm_backend_cmux_workspace_id_for_label` and `fm_backend_cmux_list_live` see only the current window's workspaces, and `fm_backend_cmux_target_ready`'s label recovery inherits that scope. -That is correct for the selected-workspace teardown case (a selected workspace is in the current window) but is a known limitation for a task workspace parked in a non-current window. +## Current operation and safety -## Workspace ids do not survive a relaunch (verified from source, not a live restart) +A genuinely fresh surface returns an internal error from `read-screen` until something has been written. +Target readiness therefore uses the structural `list-panes` response instead of a content read. +Capture remains bounded and locally trimmed after `read-screen` becomes available. -Per this task's explicit instruction NOT to relaunch the captain's app just to test this, this was verified by reading the actual shipped Swift source instead (`Sources/Workspace.swift`, cloned read-only from `github.com/manaflow-ai/cmux` at the commit current on 2026-07-04): -`Workspace`'s only initializer unconditionally sets `self.id = UUID()`, with no restored-id parameter at all. -This differs from surfaces, whose analogous initializer DOES accept `restoredSurfaceId: UUID? = nil` and use `restoredSurfaceId ?? UUID()` - but tracing every call site of that parameter showed it is used only for same-run object-identity reuse (e.g. moving/splitting an already-live surface within the current app session), never threaded through any session-restore/relaunch code path. -No `Workspace(...)` construction anywhere in the source passes a persisted id back in. +`current_directory` follows a top-level shell `cd` but not the foreground subshell opened by `treehouse get`. +Spawn-time worktree discovery sends begin and end markers around `pwd`, captures the marked block, and joins wrapped path lines. -Conclusion: workspace ids should be treated as NOT surviving an app relaunch or session restore, the same posture as herdr's/zellij's own id-instability caveats (for different underlying reasons in each case). -`fm_backend_cmux_list_live` therefore does recovery/orphan discovery strictly by **title**, never by trusting a stored uuid, mirroring both prior adapters' recovery posture. -Because cmux has one shared app namespace, the title lookup is scoped to this firstmate installation's `fm-<home-label>-` prefix and reported back to firstmate as the plain `fm-<id>` label. -No live app restart was performed to empirically confirm this beyond the source read - the two live app restarts that did occur during this build (documented in the "Socket control modes" section above) were solely to apply the one-time `socketControlMode`/password configuration change, not to test id persistence, and the app held no captain-owned workspaces at either restart (verified: it had just been freshly launched moments before, with only the default auto-created workspace present). +Literal send and Enter are separate calls. +Enter, Escape, and Ctrl-C are supported. +The composer verifier locates the last bordered composer row and delegates the content decision to `bin/fm-composer-lib.sh`. +A bare shell prompt is `unknown`, and a slash-popup placeholder remains `pending`, so only Enter is retried and text is never retyped. +cmux exposes no native generic agent busy signal, so supervision uses the shared capture/hash and busy-regex path. -## CLI is not on PATH by default (unanticipated finding) +A task workspace's last surface cannot be closed directly. +Cleanup owns the whole workspace and uses `close-workspace`. +cmux also refuses to remove the only workspace in a macOS window while returning a misleading success response. +When the task is last in its window, Firstmate creates one unfocused unnamed sibling workspace in that same window, closes the task workspace, and leaves the window with cmux's fresh default workspace. +The sibling never carries an `fm-` title and is ignored by recovery. -Unlike a typical Homebrew-installed CLI tool, the `cmux` binary is not symlinked onto `PATH` after a plain app install - `command -v cmux` returns nothing on a fresh install. -The app source (`Sources/App/CmuxCLIPathInstaller.swift`) reveals cmux ships an OPTIONAL "install CLI" action (symlinking `/usr/local/bin/cmux` to the bundled `Contents/Resources/bin/cmux`), analogous to VS Code's "Install 'code' command in PATH" - it is opt-in, not automatic. -`fm_backend_cmux_bin` handles this without requiring that step: it prefers `command -v cmux` (respecting an operator's own PATH setup, including after running that install action) and falls back to the well-known bundle path `/Applications/cmux.app/Contents/Resources/bin/cmux` otherwise. +The exact window membership is re-read before this operation. +A selected workspace that is not last closes normally; selection itself is not the trigger. +Firstmate does not attempt to close the macOS window because cmux's socket cannot close a window holding a live terminal. -## Duplicate-title behavior (verified, expected finding) +Real tests share the captain's running app rather than creating an isolated cmux session. +`tests/cmux-test-safety.sh` permits cleanup only for an exact currently listed `fm-test-` workspace and never enumerates and closes unrelated workspaces or relaunches the app. -Same as herdr's tabs and zellij's tabs, unlike tmux's own window-name uniqueness: cmux enforces no title uniqueness at all for workspaces or for surfaces/tabs within a workspace. -Verified live: two workspaces created with the identical title `fm-test-dup` both succeeded and listed simultaneously with distinct ids; two surfaces within one workspace both renamed to the identical tab title also succeeded. -`fm_backend_cmux_create_task`'s own title-based duplicate check is therefore required, mirroring both prior adapters' posture exactly. +## Active limits -## Composer verification: structural border-row classification (adapted from herdr) +- cmux is experimental, macOS-only, GUI-first, and requires the app running. +- Socket access requires a one-time manual Settings change. +- Secondmate spawns are unsupported until a per-home lifecycle design is verified. +- There is no native busy or push-event signal. +- A target can disappear after structural readiness and before the operation. +- The only-workspace cleanup path leaves a fresh default workspace and cannot close the window. +- Label lookup and recovery are currently scoped to the current cmux window, so a task moved to a non-current window is a known recovery blind spot. +- Workspace ids do not survive app relaunch and are never recovery authority. -cmux's `read-screen` gives plain-text capture with no cursor-row primitive and no ANSI style channel, unlike tmux's `#{cursor_y}` and herdr's `--format ansi` path for ANSI-aware ghost/placeholder classification. -Per this build task's explicit direction, `fm_backend_cmux_composer_state` is adapted directly from herdr's post-incident structural border-row classifier (`fm_backend_herdr_composer_state`, `docs/herdr-backend.md`) rather than zellij's content-diff approach: it locates the composer's own row as the only captured line whose trimmed content both starts and ends with the same border glyph (`│`, `┃`, or a plain ASCII `|`), scanning forward and keeping the LAST match so an earlier border-shaped line can never outrank the real bottom-anchored composer row. -After that adapter-owned row finding, cmux delegates the shared `empty`/`pending`/`unknown` decision to `bin/fm-composer-lib.sh`; a bare shell prompt with no boxed composer row reads `unknown`, not empty. -This directly defends against the same class of incident herdr hit on 2026-07-03: a slash-command popup's first Enter can close the popup and fill an argument-hint placeholder into the composer rather than submitting, which a raw pane-content-diff check (zellij's approach) would misread as "submitted". -`tests/fm-backend-cmux.test.sh` pins this exact regression shape (`test_send_text_submit_popup_autocomplete_requires_second_enter`), verifying the adapter retries a genuine second Enter rather than declaring victory after the first one closes a popup. -All implemented submit-verifying backends expose the identical caller-facing verdict vocabulary (`empty`, `pending`, `unknown`, `send-failed`), so `fm-send.sh` needs no cmux-specific branching. +## Regression entry points -## Test safety - -Unlike herdr/zellij, cmux has no isolated, throwaway SESSION a test can spin up and tear down on its own - there is just "the app", the same real running instance a captain uses day to day. -`tests/cmux-test-safety.sh`'s guard is adapted to this shape: `cmux_refuse_if_unsafe` requires a non-empty workspace id, a caller-facing label carrying the `fm-test-` prefix, and that the workspace is CURRENTLY LISTED with the scoped title derived from that label, before `cmux_safe_close_workspace` is allowed to close it. -Every real-cmux test in this document and its accompanying test files creates only `fm-test-`-prefixed task labels, never enumerates-and-closes, and never quits or relaunches the app. - -## End-to-end verification (spawn -> steer -> peek -> done -> merge -> teardown) - -Beyond the fake-CLI unit tests (`tests/fm-backend-cmux.test.sh`) and the real-CLI smoke test (`tests/fm-backend-cmux-smoke.test.sh`), the full firstmate lifecycle was driven end to end against a real `claude` crewmate through this branch's own scripts, in a scratch `FM_HOME`, a scratch `local-only` git project, and the same captain-owned real cmux app instance (there is no isolated session to spin up for cmux, unlike herdr/zellij; only firstmate-created `fm-` task workspaces were ever touched): - -1. `FM_HOME=<scratch> bin/fm-spawn.sh cmux-e2e-t1 projects/scratch-e2e-project --backend cmux claude` - spawned successfully, printing `window=<workspace_uuid>:<surface_uuid>` in the summary and writing `backend=cmux`, `cmux_workspace_id=`, `cmux_surface_id=` to the task's meta. The worktree-discovery poll correctly resolved the real treehouse worktree path using the active `pwd`-marker-probe workaround (finding #2), exactly as designed. -2. `bin/fm-peek.sh fm-cmux-e2e-t1` - showed the live claude trust dialog ("Quick safety check: Is this a project you created or one you trust?"). -3. `FM_HOME=<scratch> bin/fm-send.sh fm-cmux-e2e-t1 --key Enter` - accepted the trust dialog; `send_key`'s Escape/Enter path confirmed live against the real claude TUI, not just a plain shell. -4. `bin/fm-peek.sh fm-cmux-e2e-t1` again - showed claude actively working through the brief (confirming worktree isolation, writing `hello.txt`, committing). -5. `FM_HOME=<scratch> bin/fm-send.sh fm-cmux-e2e-t1 "captain says: proceed as planned, this is a trivial verification task"` - a plain-text steer sent after the crewmate had already finished and stopped; `fm-send` reported no `pending`/`send-failed` error, and the message was confirmed landed and acknowledged in the next peek. This is the first live proof of `fm_backend_cmux_composer_state`'s structural border-row classifier against a REAL claude TUI composer box (every Phase 1 empirical test used a plain shell prompt, which has no bordered composer at all). -6. `FM_HOME=<scratch> bin/fm-send.sh fm-cmux-e2e-t1 "/compact"` - the popup-placeholder/second-Enter regression class, tested live: `fm-send` reported success with no error, and the next peek confirmed `/compact` had genuinely EXECUTED ("Compacting conversation... 25%", later "Compacted"), not merely sat typed-but-unsubmitted in the composer. This directly confirms `fm_backend_cmux_send_text_submit` correctly retries past a popup-closing first Enter and lands a genuine second Enter against the real app, the same incident class herdr hit on 2026-07-03. -7. The crewmate's commit (`add hello.txt`, message `add hello.txt`) was confirmed present on branch `fm/cmux-e2e-t1` in the scratch project's git history, with `hello.txt` containing exactly the expected line, and the status file ending in `done: ready in branch fm/cmux-e2e-t1`. -8. `bin/fm-teardown.sh cmux-e2e-t1` **REFUSED**, exactly as required: `REFUSED: local-only worktree ... has work not yet merged into main and not on any remote.` -9. `bin/fm-merge-local.sh cmux-e2e-t1` - fast-forwarded the scratch project's local `main` to the crewmate's commit (`e99f00a -> f064d41`). -10. `bin/fm-teardown.sh cmux-e2e-t1` now succeeded: terminated the lingering worktree processes, returned the treehouse worktree, closed the cmux workspace (confirmed gone via `workspace list` - only the pre-existing default workspace remained), and removed all of the task's `state/` files. -11. Two additional trivial crewmate tasks (`cmux-e2e-t2`, `cmux-e2e-t3`) were spawned concurrently into the same scratch project via `fm-spawn.sh`'s batch dispatch form, to exercise multiple simultaneous cmux workspaces; both reached their trust-accepted, standing-by state cleanly, were peeked successfully, and were torn down (clean worktrees, no unlanded work) with the same `fm-teardown.sh` path. - -All three tasks' cmux workspaces and worktrees were confirmed fully cleaned up afterward (`workspace list` showing only the pre-existing default workspace; `treehouse destroy --all --yes` freeing the scratch project's pool); the scratch `FM_HOME` and project were removed entirely. - -**Screenshot request (best-effort, explicitly skippable):** the captain separately asked for a screenshot of the cmux window while multiple concurrent tasks were running, to be kept only if it showed a genuinely healthy fleet with no visible errors. One `screencapture -x` (full-screen) attempt was made while all three tasks above were live. It did NOT capture cmux at all: because every firstmate cmux workspace is created with `--focus false` (finding: focus verified to default off), cmux is never the frontmost/active application, so a full-screen capture on this shared machine captured a completely different, unrelated live terminal session's frontmost window instead - one that turned out to show real, sensitive operational content (a different active firstmate/herdr fleet with real secondmate names and conversation). That file was deleted immediately without being viewed further or retained anywhere. No second attempt was made: bringing cmux to the foreground to make it capturable would mean actively focusing/raising its window, which would yank focus away from whatever the captain or another live session currently has in the foreground - the exact disruption `--focus false` exists to avoid - and enumerating other windows to target a background-window capture of just cmux risks the same kind of unrelated-content exposure. Per the captain's explicit allowance, this request was skipped rather than risk either disruption or another accidental capture. - -## Known gaps left for a follow-up +```sh +tests/fm-backend-cmux.test.sh +tests/fm-backend-cmux-smoke.test.sh +``` -- **No event push at all**, not even herdr's semantic busy-state: cmux has agent-awareness elsewhere (Claude Code hooks, session-resume) but nothing exposed over the socket API for generic busy/idle classification, so `fm-watch.sh`'s existing pane-hash + `FM_BUSY_REGEX` poll loop is the ONLY event source for this backend, identical to the tmux/zellij/Orca path. -- **GUI-first, macOS-only, requires the app running** - identical posture to Orca. - Never a candidate for a headless/CI firstmate instance, because runtime auto-detection (cmux runtime signals; see "Runtime auto-detection" above) can only fire from inside a live cmux terminal in the first place. - The one-time socket-access setup remains an unavoidable manual step regardless of how the backend was selected. -- **`--secondmate` spawns are refused** (mirrors Orca's refusal) - no per-home container design (a herdr-style workspace-per-home split, or similar) has been designed or verified for cmux yet. -- **The one-time socket-access setup is a real, undocumented-by-upstream onboarding step.** A captain who selects `backend=cmux` without first switching `automation.socketControlMode` away from its `cmuxOnly` default to a viable mode (Automation mode recommended; see "Setup") will see every spawn fail with an actionable error naming the viable modes and pointing back to this document, but there is no way for firstmate to complete that GUI-only setup step on the captain's behalf. -- **A surface can still die in the brief window between `target_ready` succeeding and the operation's own call running.** That remaining race degrades to "the operation quietly did nothing" - the same class of gap firstmate already tolerates for an unverified send on any backend, caught downstream by `fm-spawn.sh`'s worktree-discovery poll timing out, `fm_backend_cmux_send_text_submit`'s retry loop (which reports `send-failed`/`pending`/`unknown` rather than a false "sent"), or the watcher's stale-pane detection. -- **Windows cannot be closed over the control socket, and label lookup is current-window scoped** - both owned by "Closing the last workspace in a window" above. Teardown of a last-in-window task workspace therefore leaves that window a fresh default workspace rather than closing it, and `fm_backend_cmux_workspace_id_for_label`/`fm_backend_cmux_list_live` only see the current window, so a task workspace parked in a non-current window is a known blind spot for label-based recovery. +[`verification/runtime-backends.md`](verification/runtime-backends.md#cmux) records the active source and live evidence, including socket modes and last-in-window cleanup. diff --git a/docs/codex-app-backend.md b/docs/codex-app-backend.md index 6a2bbd8eb22..4f01ded7e1d 100644 --- a/docs/codex-app-backend.md +++ b/docs/codex-app-backend.md @@ -1,211 +1,57 @@ -# Codex App backend contract +# Codex App backend boundary -Status: blocked for Firstmate as a selectable shell backend. -The Codex Desktop host-tool loop works, including status-file writes, but Firstmate does not yet have a supported shell-callable bridge to those host tools. +Codex App is not a selectable Firstmate runtime backend. +Codex Desktop host tools can create and supervise visible threads and those threads can write Firstmate status files when given an authorized path, but Firstmate has no supported shell-callable bridge to those host tools. +A manual thread ledger is not a backend. -This document replaces the earlier passive visible-thread ledger shape. -A manual ledger is not a backend. +## Acceptance contract -## Backend acceptance contract +A future Codex App backend must satisfy the same lifecycle contract as terminal-backed adapters: -A Codex App backend must satisfy the same lifecycle contract as the terminal-backed adapters: +1. Create a task endpoint and return a durable thread id. +2. Send the initial instructions and later operator messages to that endpoint. +3. Read enough live state or bounded transcript to supervise the task. +4. Archive, kill, or otherwise stop the exact endpoint. +5. Let the thread append Firstmate's normal lifecycle lines to `state/<id>.status`. -1. Firstmate creates the task endpoint and receives a durable thread id. -2. Firstmate sends the initial prompt and later operator messages to that endpoint. -3. Firstmate observes enough live thread state or transcript to supervise the task. -4. Firstmate can archive, kill, or otherwise stop supervising the endpoint. -5. The Codex thread can report back through Firstmate's normal `state/<id>.status` lifecycle. +The status return channel is mandatory. +A visible thread that cannot report into Firstmate's normal lifecycle is not a complete backend. -The final point is mandatory. -If a Desktop-owned thread cannot write Firstmate status files, the backend cannot be treated as complete. +## Current blocker -## Verified Desktop host-tool smoke +Firstmate backend scripts are shell entry points and can call tmux, Herdr, Zellij, Orca, and cmux directly. +Codex Desktop host tools are available to a Desktop conversation, not to arbitrary Firstmate subprocesses. +The missing component is a Codex Desktop-supported shell-callable transport, not another local ledger. -Latest verified host-tool smoke date: 2026-07-06. -Environment: Codex Desktop host tools, local host, saved project `<FIRSTMATE_HOME>/projects/sift`, Desktop-owned worktree `<CODEX_DESKTOP_WORKTREE>`, Firstmate home `<FIRSTMATE_HOME>`. -Local absolute path prefixes are redacted as `<FIRSTMATE_HOME>` and `<CODEX_DESKTOP_WORKTREE>`; file names, host-tool ids, thread ids, status lines, and report values are otherwise exact. +`codex app-server --stdio` exposes useful JSON-RPC pieces such as thread start, turn start, thread read, and thread archive. +A one-process probe could create and archive a thread record, but no supported bridge was found that lets Firstmate create, continue, read, and archive the same visible Desktop-owned endpoint over its full lifetime. +A raw Desktop control-socket proxy is not a supported transport. +These partial pieces do not authorize adding `codex-app` to the known or spawn-capable backend registries. -Codex Desktop/OpenAI local bundle metadata from the smoke machine: +## Required bridge -```text -$ /usr/libexec/PlistBuddy -c 'Print :CFBundleShortVersionString' /Applications/Codex.app/Contents/Info.plist -26.623.101652 - -$ /usr/libexec/PlistBuddy -c 'Print :CFBundleVersion' /Applications/Codex.app/Contents/Info.plist -4674 - -$ /usr/libexec/PlistBuddy -c 'Print :CFBundleIdentifier' /Applications/Codex.app/Contents/Info.plist -com.openai.codex - -$ stat -f '%Sm %N' -t '%Y-%m-%d %H:%M:%S %z' /Applications/Codex.app/Contents/Info.plist -2026-07-02 21:55:53 -0400 /Applications/Codex.app/Contents/Info.plist -``` - -Smoke target files: - -```text -<FIRSTMATE_HOME>/state/codex-app-host-smoke-20260706-live.status -<FIRSTMATE_HOME>/data/codex-app-host-smoke-20260706-live/report.md -``` +Implementation can begin after Codex Desktop exposes one supported interface: -Host-tool operation sequence: +- a CLI wrapper for create, send, read, and archive host-tool operations; +- a documented JSON-RPC or MCP transport with stable framing; or +- a maintained helper that speaks the supported transport and returns plain JSON to a shell adapter. -1. `list_projects` confirmed the saved project target. -2. `create_thread` requested a new Codex Desktop project worktree thread. -3. `list_threads` recovered the created thread id after queued worktree setup. -4. `read_thread` observed the active and completed initial turn. -5. Shell reads verified the status/report files under the Firstmate home. -6. `send_message_to_thread` delivered a follow-up to the same thread. -7. `read_thread` observed the completed follow-up turn. -8. `set_thread_archived` archived the thread. -9. A final `read_thread` still returned the transcript and showed `status.type=notLoaded`. - -Exact host-tool requests and relevant output: +The bridge must provide these semantics: ```text -list_projects: - projectId=<FIRSTMATE_HOME>/projects/sift - projectKind=local - label=sift - path=<FIRSTMATE_HOME>/projects/sift - -create_thread request: - target.type=project - target.projectId=<FIRSTMATE_HOME>/projects/sift - target.environment.type=worktree - prompt smoke_id=codex-app-host-smoke-20260706-live - prompt status_file=<FIRSTMATE_HOME>/state/codex-app-host-smoke-20260706-live.status - prompt report_file=<FIRSTMATE_HOME>/data/codex-app-host-smoke-20260706-live/report.md - prompt required status line: working: Codex Desktop thread started - prompt required sentinel: FM_CODEX_APP_HOST_TOOL_SMOKE_20260706_LIVE_OK - -create_thread response: - pendingWorktreeId=local:a4a96438-a0ed-4305-b83c-5a47336f5abf - -list_threads query=codex-app-host-smoke-20260706-live: - id=019f39ea-5cca-7031-bfb0-f8054a2b253a - hostId=local - status=active - cwd=<CODEX_DESKTOP_WORKTREE> - -read_thread initial turn while active: - thread.id=019f39ea-5cca-7031-bfb0-f8054a2b253a - thread.status.type=active - cwd=<CODEX_DESKTOP_WORKTREE> - agentMessage: Running the smoke exactly as delegated: repo identity first, then the Firstmate status/report writes, then the requested `sed` checks. - -read_thread initial turn after completion: - thread.status.type=idle - turn.status=completed - durationMs=54923 - -$ pwd -<CODEX_DESKTOP_WORKTREE> - -$ git rev-parse --show-toplevel -<CODEX_DESKTOP_WORKTREE> - -$ git branch --show-current - -$ sed -n '1,20p' <FIRSTMATE_HOME>/state/codex-app-host-smoke-20260706-live.status -working: Codex Desktop thread started - -$ sed -n '1,40p' <FIRSTMATE_HOME>/data/codex-app-host-smoke-20260706-live/report.md -smoke_id=codex-app-host-smoke-20260706-live -cwd=<CODEX_DESKTOP_WORKTREE> -git_root=<CODEX_DESKTOP_WORKTREE> -branch= -status_file=<FIRSTMATE_HOME>/state/codex-app-host-smoke-20260706-live.status -status_file_write=ok -sentinel=FM_CODEX_APP_HOST_TOOL_SMOKE_20260706_LIVE_OK - -send_message_to_thread request: - threadId=019f39ea-5cca-7031-bfb0-f8054a2b253a - prompt required status line: done: follow-up delivered through send_message_to_thread - -send_message_to_thread response: - threadId=019f39ea-5cca-7031-bfb0-f8054a2b253a - -read_thread follow-up turn: - turn.status=completed - durationMs=7118 - -$ sed -n '1,20p' <FIRSTMATE_HOME>/state/codex-app-host-smoke-20260706-live.status -working: Codex Desktop thread started -done: follow-up delivered through send_message_to_thread - -set_thread_archived request: - threadId=019f39ea-5cca-7031-bfb0-f8054a2b253a - archived=true - -set_thread_archived response: - threadId=019f39ea-5cca-7031-bfb0-f8054a2b253a - archived=true - -read_thread after archive: - thread.id=019f39ea-5cca-7031-bfb0-f8054a2b253a - thread.status.type=notLoaded - thread.cwd=<CODEX_DESKTOP_WORKTREE> - transcript still included the initial and follow-up completed turns. -``` - -Result: a Desktop-owned Codex thread can write Firstmate status files when the prompt gives it the absolute status path and the Desktop permission context can write that checkout. -The return channel is real at the Codex Desktop host-tool layer. - -## Codex Desktop API blocker - -Firstmate's backend scripts are Bash entry points. -They can call `tmux`, `herdr`, `zellij`, primitive Orca CLI surfaces, and `cmux` directly. -The Codex Desktop host tools verified above are available to the Codex Desktop conversation, not to arbitrary Firstmate subprocesses. -The missing piece is therefore a supported Codex Desktop transport that a Bash backend can call, not another Firstmate-local ledger. - -The available Codex CLI and app-server probes found useful pieces but not a supported visible-thread backend transport: - -- `codex app-server --stdio` exposes JSON-RPC methods such as `thread/start`, `turn/start`, `thread/read`, and `thread/archive`. -- A one-shot stdio probe could create a thread record, and `thread/archive` worked through that same stdio process. -- The managed daemon path was unavailable in this Desktop install. -- A raw proxy attempt against the Desktop control socket did not accept plain JSON-RPC framing. - -That is not enough to add `codex-app` to `FM_BACKEND_KNOWN` or `FM_BACKEND_SPAWN`. -A Firstmate backend must be able to create a thread, start or continue turns, read live state while turns run, and archive/stop the same endpoint through a Codex Desktop-supported shell-callable API. -Shipping a local ledger would only record intentions; it would not supervise the actual Desktop thread. - -## Required Codex Desktop bridge - -Firstmate should implement a Codex App adapter only after Codex Desktop exposes one of these supported interfaces: - -- A supported CLI wrapper around the Desktop host tools: create thread, send message, read transcript/state, archive thread. -- A documented JSON-RPC or MCP transport that Firstmate can call from Bash with stable request/response framing. -- A small maintained helper binary/script that speaks the supported transport and returns plain JSON to `bin/backends/codex-app.sh`. - -Minimum command semantics: - -```text -create: - input: task id, cwd/worktree request, initial prompt - output: thread id, Desktop-owned cwd if different, initial status - -send: - input: thread id, text - output: accepted/rejected delivery result - -capture/read: - input: thread id, bounded transcript or status cursor - output: enough text/state for fm-peek.sh, fm-watch.sh, and fm-crew-state.sh - -archive/kill: - input: thread id - output: archived/stopped result - -status return channel: - the thread must be able to append Firstmate status lines to state/<id>.status +create: task id, worktree request, initial instructions -> thread id, cwd, state +send: thread id, text -> accepted or rejected +read: thread id, bounded cursor -> transcript and live state +archive: thread id -> archived or stopped +return: thread appends state/<id>.status lifecycle lines ``` -Once that bridge exists, the implementation should add a real `bin/backends/codex-app.sh`, persist `backend=codex-app` and `codex_app_thread_id=` in `state/<id>.meta`, and wire spawn/send/peek/watch/teardown through the same dispatcher paths used by the existing adapters. +Once available, Firstmate should add a real `bin/backends/codex-app.sh`, persist `backend=codex-app` and `codex_app_thread_id=`, and route spawn, send, peek, watch, and cleanup through the shared dispatcher. -## Rollout clause +## Rollout -After a supported shell-callable Codex Desktop/OpenAI bridge exists, Firstmate should implement Codex App for ship and scout tasks first. -Secondmate support remains out of scope until ship/scout supervision, status return, send/read, and archive/teardown are proven through the normal backend dispatcher. +Ship and scout tasks come first. +Secondmate support remains out of scope until create, send, read, status return, and archive are proven through the normal backend dispatcher. +Until then, Codex App remains a blocked backend boundary with a verified host-tool capability record, not a selectable backend. -Until then, Codex App support remains a verified host-tool smoke plus this blocked backend contract, not a selectable backend. +[`verification/runtime-backends.md`](verification/runtime-backends.md#codex-app-host-tools) owns the active Desktop host-tool smoke without exposing task-specific thread ids or local paths. diff --git a/docs/configuration.md b/docs/configuration.md index d669972b0cb..94799d0e8ea 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -12,7 +12,7 @@ This section is the single owner of the top-level operational-home layout; produ The tracked code root contains the shared instruction, skill, documentation, workflow, and `bin/` surfaces, while each effective `FM_HOME` contains private operational directories. `data/` holds durable private fleet records such as the project and secondmate registries, captain preferences, optional shared captain preferences, learnings, backlog, briefs, and scout reports. `state/` holds volatile runtime records such as task metadata, append-only status events, endpoint signals, watcher and wake-queue coordination, away-mode state, generated X-mode artifacts, private secondmate config-reread generations with their retry and quarantine state, and parent-owned secondmate pending-reply records under `state/pending-replies/` (`bin/fm-pending-reply-lib.sh`). -`config/` holds local gitignored operating choices, and `projects/` holds the local project clones that Firstmate reads but changes only through the guarded exceptions in `AGENTS.md`. +`config/` holds local gitignored operating choices, and `projects/` holds the local project clones that Firstmate reads but changes only through the narrow guarded and concrete captain-approved exceptions in `AGENTS.md`. `bin/fm-spawn.sh` owns the base task-metadata fields it emits, while the runtime-backend section below owns backend-specific fields and selector interpretation. The producing PR and X helpers own the fields they append, `bin/fm-classify-lib.sh` owns status-event vocabulary, and `bin/fm-crew-state.sh` owns current-state reconciliation. @@ -53,7 +53,7 @@ For spawn-capable adapters, the runtime session-provider backend controls where Treehouse remains the worktree provider for tmux, herdr, zellij, and cmux, since herdr, zellij, and cmux are session providers only; Orca provides both the task worktree and terminal endpoint. New spawns choose the backend in this order: an explicit `--backend` flag firstmate passes when it spawns a task, then `FM_BACKEND`, then the first non-empty line of local gitignored `config/backend`, then runtime auto-detection from `$TMUX`, `HERDR_ENV=1`, or cmux runtime signals, then default `tmux`. If more than one runtime marker is present, detection resolves innermost-first: `$TMUX` is checked before `HERDR_ENV=1`, which is checked before cmux's primary `CMUX_WORKSPACE_ID` marker and its documented fallback signals - tmux or herdr started from inside a cmux terminal is the innermost, currently-executing layer, while cmux itself (a terminal application, not a nestable multiplexer) is always checked last. -See [`docs/cmux-backend.md`](cmux-backend.md#runtime-auto-detection) for why cmux can be selected when `CMUX_WORKSPACE_ID` is absent. +See [`docs/cmux-backend.md`](cmux-backend.md#runtime-detection) for why cmux can be selected when `CMUX_WORKSPACE_ID` is absent. Auto-detected herdr or cmux prints a stderr notice naming `config/backend` and `--backend tmux` as opt-outs; auto-detected tmux stays silent to preserve existing default behavior. Zellij and Orca are never auto-detected; select them by putting the name in a local `config/backend` file, by exporting `FM_BACKEND=<name>`, or by telling the first mate in chat. Any value other than `tmux`, `herdr`, `zellij`, `orca`, or `cmux` is rejected until another adapter is implemented and verified. @@ -67,6 +67,7 @@ A zellij spawn additionally version-gates against the installed `zellij` binary' A cmux spawn additionally version-gates against the installed `cmux` binary's version, requires `jq`, and requires the control socket to be reachable and accessible (see [`docs/cmux-backend.md`](cmux-backend.md) "Setup" for the one-time socket-access configuration this needs; Automation mode is the recommended socket control mode, with Password mode supported via `config/cmux-socket-password`), refusing loudly and non-retryably on a `cmuxOnly`/unauthenticated socket. A backend spawn refusal from a missing dependency, version gate, or unauthenticated socket is terminal for that selected backend; firstmate surfaces it as a blocker instead of silently retrying another backend. Task meta records `backend=` only for a non-default backend; an absent `backend=` means `tmux`, preserving existing default-path meta files. +Every new task records `endpoint_task_id=` as the cleanup binding between the metadata filename and its opaque runtime endpoint. A herdr task additionally records `herdr_session=`, `herdr_workspace_id=`, `herdr_tab_id=`, and `herdr_pane_id=`. A zellij task additionally records `zellij_session=`, `zellij_tab_id=`, and `zellij_pane_id=`. An Orca task additionally records `orca_worktree_id=` and `terminal=`, with `window=fm-<id>` kept as the shared firstmate alias. @@ -77,10 +78,12 @@ Otherwise an exact task id matching `state/<id>.meta` wins before the legacy `fm A metadata-routed selector returns the recorded backend target (`terminal=` for Orca, otherwise `window=`), and matching explicit targets can still recover the recorded backend when metadata contains the same endpoint. Only metadata-routed task selectors carry secondmate-marker and Codex-harness context; explicit endpoint escape hatches do not. These five sentences are the single owner of the task-selector vocabulary; backend guides and other documents point here instead of restating the resolution order. -`fm-teardown.sh <id>` takes a task id directly and uses the same recorded backend target fields after loading `state/<id>.meta`. +`fm-teardown.sh <id>` takes a task id directly and validates the complete metadata-only endpoint identity before any runtime dispatch or cleanup mutation. +Missing, empty, duplicate, malformed, backend-inconsistent, or task-mismatched endpoint records are preserved and refused. +Legacy tmux metadata remains cleanup-compatible when its exact window name is `fm-<id>`; opaque non-tmux endpoints require their recorded `endpoint_task_id=` binding. By default, Herdr workspaces are derived from `FM_HOME`: the primary home uses `firstmate`, and a secondmate home marked by `.fm-secondmate-home` uses `2ndmate-<secondmate-id>`. The default-container spawn, list-live, and recovery paths read that label from the active home, so a secondmate's own crewmates stay inside that secondmate home's herdr space. -The optional local `config/herdr-presentation-spaces` presence flag instead enables Herdr's default-off disposable single-task visual projection; [`docs/herdr-backend.md`](herdr-backend.md#optional-disposable-single-task-presentation-spaces) owns its behavior, safety limits, and recovery contract. +The optional local `config/herdr-presentation-spaces` presence flag instead enables Herdr's default-off disposable single-task visual projection; [Optional presentation spaces](herdr-backend.md#optional-presentation-spaces) owns its behavior, safety limits, recovery contract, and narrow locked session-start cleanup of exact restored idle-shell children. The flag is default-off and inherited into secondmate homes under the primary-authoritative contract owned by [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md). For normal herdr operations, `HERDR_SESSION` selects the named session, but destructive test cleanup must not rely on `HERDR_SESSION` alone. Use the explicit guarded cleanup path described in [`docs/herdr-backend.md`](herdr-backend.md) instead of `herdr server stop`. @@ -89,8 +92,8 @@ Zellij has no per-home workspace split: primary and secondmate tasks share that Use the guarded cleanup path described in [`docs/zellij-backend.md`](zellij-backend.md) instead of `kill-all-sessions` or `delete-all-sessions`. cmux has no session layer at all - one workspace per task, in whatever cmux window is open - and its socket password (when configured) is read from local, gitignored `config/cmux-socket-password` under the effective config directory, never committed. The caller-facing label remains `fm-<id>`, but the actual cmux workspace title is scoped by the active `FM_HOME` readable label plus a short hash of the resolved `FM_ROOT` path as `fm-<home-label>-<id>`. -Test cleanup must use the guarded path described in [`docs/cmux-backend.md`](cmux-backend.md)'s "Test safety" section, never enumerate-and-close every workspace. -The `config/backend` file is not inherited by secondmate homes. +Test cleanup must use the guarded path in [`docs/cmux-backend.md`](cmux-backend.md#current-operation-and-safety), never enumerate-and-close every workspace. +`config/backend` is inherited into secondmate homes under the primary-authoritative contract owned by [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md). ## Away-mode supervisor backend (FM_SUPERVISOR_BACKEND / FM_SUPERVISOR_TARGET) @@ -111,7 +114,7 @@ Beyond the durable `state/.subsuper-inject-wedged` marker and the tmux status-li Directives are `off` (a position-independent kill switch that disables every active alert), `auto`/`default`, `osascript` (macOS Notification Center banner), `herdr` (herdr UI notification), and `command:<cmd>` (run `<cmd>` via `sh -c`, summary on `$1` and stdin). An absent file means `auto`, i.e. default-on on macOS: the alarm exists precisely so a wedged away-mode primary is never silent, and it fires at most once per max-defer window after a genuine wedge. A missing or failing channel logs and falls through to the next, never crashing the daemon. -See [`wedge-alarm.md`](wedge-alarm.md) for the channel reference and macOS verification evidence, and [`examples/wedge-alarm`](examples/wedge-alarm) for a copyable config. +See [`wedge-alarm.md`](wedge-alarm.md) for the current channel reference, [`verification/supervision.md`](verification/supervision.md#wedge-alarm-channels) for active evidence, and [`examples/wedge-alarm`](examples/wedge-alarm) for a copyable config. ## Gate defaults (.no-mistakes.yaml) @@ -119,7 +122,7 @@ The tracked `.no-mistakes.yaml` keeps test evidence outside the repo and pins `c That evidence policy is specific to the firstmate repo: target projects may legitimately commit `.no-mistakes/evidence/` from their own no-mistakes pipeline, but firstmate keeps `.no-mistakes/` local and CI rejects tracked entries under that path. It does not set `commands.test` to a complete `tests/*.test.sh` walk. See [CONTRIBUTING.md](../CONTRIBUTING.md) for the firstmate-specific local test policy and entry points. -Portable shard evidence and coverage rules are in [fm-test-portable-shards.md](fm-test-portable-shards.md), and [herdr-backend.md](herdr-backend.md) owns the real-Herdr lane's verification and isolation rationale. +Portable shard evidence and coverage rules are in [fm-test-portable-shards.md](fm-test-portable-shards.md); [herdr-backend.md](herdr-backend.md#destructive-lab-safety) owns the real-Herdr lane's isolation boundary, and [runtime-backends.md](verification/runtime-backends.md#herdr) owns active evidence. ## Captain Preferences (data/captain.md / data/captain-shared.md) @@ -134,6 +137,20 @@ Fleet-local operational facts and gotchas live locally in `data/learnings.md`; i The file is created lazily on first learning and follows the same dated, evidence-backed, curated style as `data/captain.md`: inspect the current file first, then rewrite or prune stale entries instead of appending forever. There is no shared learnings file by captain decision. +## Startup memory budget (config/startup-memory-budget) + +`config/startup-memory-budget` is the primary-authoritative per-home allowance for the startup prompt-memory surface: `data/captain.md`, `data/captain-shared.md`, and `data/learnings.md` together. +The locked mutable bootstrap path materializes its visible default of `7500` estimated tokens in a primary home when the file is absent. +To select another allowance, replace the primary home's file with one valid positive value in the exact format below; the next locked bootstrap convergence or `bin/fm-config-push.sh` propagates it to registered secondmates. +A secondmate does not create an independent default and instead receives the primary value through the inherited-local-material contract in [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md). +The file must be one positive base-10 integer followed by exactly one newline in a regular, single-linked file beneath a non-symlinked `config/` directory. +Malformed, multi-line, symlinked, hardlinked, special, or otherwise unsafe values are rejected rather than treated as a default. +Use `bin/fm-startup-memory-budget.sh read` to validate and print the effective value, or `bin/fm-startup-memory-budget.sh report` to account for the three files. +The stable local estimate is `ceil(UTF-8 bytes / 3)` per file, a conservative portable approximation rather than a provider-exact tokenizer. +An inherited `data/captain-shared.md` counts in a secondmate's total but remains primary-owned and read-only there. +The internal `/stow` skill curates only the editable local files in that case and reports the primary-owned shared file as a concrete exception if it alone exceeds the budget. +The helper's header owns exact parsing, publication, and report output mechanics. + ## Secondmate routes (data/secondmates.md) Persistent secondmate routes live locally in `data/secondmates.md`. @@ -166,6 +183,8 @@ When it is unset, most scripts use the repo root as the home; when it is set, sc When `FM_HOME` is unset, it also behaves as the old whole-root override. `bin/fm-send.sh` is intentionally stricter than that general fallback: it requires `FM_HOME` to be set before resolving a target, so operator steers cannot silently resolve against the wrong home. `FM_STATE_OVERRIDE`, `FM_DATA_OVERRIDE`, `FM_PROJECTS_OVERRIDE`, and `FM_CONFIG_OVERRIDE` override individual operational directories for tests and specialized harness setup. +Before `fm-brief.sh`, `fm-spawn.sh`, or `fm-afk-launch.sh` persists a path or passes it to another process, it resolves each applicable relative `FM_HOME`, `FM_STATE_OVERRIDE`, or `FM_DATA_OVERRIDE` directory against the caller's working directory, preserves absolute spellings unchanged, and rejects an unresolvable relative directory with the offending variable named. +Bootstrap applies the same relative `FM_HOME` resolution only when embedding that home in the generated X-mode poll shim; other transient consumers retain their existing shell-relative behavior. For the herdr backend, `FM_HOME` also determines the workspace label used by the adapter. For the zellij backend, `FM_HOME` does not split containers, but it determines the readable home prefix embedded in visible tab titles; use `FM_ZELLIJ_SESSION` when a separate zellij session is needed. The full zellij home label also includes a short hash of the resolved `FM_ROOT` path. @@ -174,13 +193,17 @@ The full cmux home label also includes a short hash of the resolved `FM_ROOT` pa ## Harness support -claude, codex, opencode, pi, and grok are all empirically verified; new harnesses get verified through a supervised trial task before joining the set. +claude, codex, opencode, pi, pi-signed, grok, and kimi are empirically verified for crewmate and secondmate launches; [README requirements](../README.md#requirements) own the set supported for the primary session. +New harnesses get verified through a supervised trial task before joining the set. The verified adapter knowledge - busy signatures, interrupt and exit commands, skill-invocation syntax, and per-harness quirks - lives in [`.agents/skills/harness-adapters/SKILL.md`](../.agents/skills/harness-adapters/SKILL.md). Launch mechanics, including the verified command templates, live in [`bin/fm-spawn.sh`](../bin/fm-spawn.sh). -Primary-session turn-end guard integrations for verified harnesses are tracked as repo-level hook files and documented in [`docs/turnend-guard.md`](turnend-guard.md). +Enabled primary-session turn-end guard integrations are tracked as repo-level hook files and documented in [`docs/turnend-guard.md`](turnend-guard.md). +Kimi remains outside the primary turn-end guard integrations; [`docs/turnend-guard.md`](turnend-guard.md#compatibility-limits) owns its separate captain-approved crew wake hook. Primary-session watcher wake protocols are rendered at session start by [`bin/fm-supervision-instructions.sh`](../bin/fm-supervision-instructions.sh) from [`docs/supervision-protocols/`](supervision-protocols/). -Claude and Grok use background-notify cycles, Codex uses bounded foreground checkpoints, Pi uses its two tracked primary extensions, and OpenCode uses its TUI plugin. +Claude's Stop `asyncRewake` hook owns tokenless re-arm cycles, Grok uses background-notify cycles, Codex uses bounded foreground checkpoints, Pi and pi-signed use the same two tracked primary extensions, and OpenCode uses its TUI plugin. `config/crew-harness` is a local, gitignored file containing one adapter name for crewmate and scout launches. +When pi-signed is selected, Firstmate launches the executable named `pi-signed` from `PATH` with `FM_PI_HARNESS=pi-signed` and refuses the launch if it is unavailable rather than falling back to pi. +Plain Pi launches set `FM_PI_HARNESS=pi`, so a signed primary's environment cannot relabel a plain Pi worker. When it is absent or contains `default`, crewmates mirror the firstmate's own harness. `config/secondmate-harness` is a separate local, gitignored file containing the adapter the primary uses to launch secondmate agents, optionally followed by model and effort tokens on the same line. The first non-empty, non-comment line is parsed as `<harness> [<model>] [<effort>]`. @@ -194,16 +217,21 @@ The inherited-local-material contract is owned by [`secondmate-provisioning`](.. Those inherited values are defaults and rules only; `fm-spawn` still permits a consciously chosen explicit runtime outside the config. `config/secondmate-harness` is not inherited because secondmates do not launch secondmates. For grok, `fm-spawn.sh` installs one firstmate-owned global turn-end hook under `$GROK_HOME/hooks/`, or `~/.grok/hooks/` when `GROK_HOME` is unset, and drops a per-task `.fm-grok-turnend` pointer in the worktree, with teardown removing the task token and pointer. -For Pi secondmate launches, `fm-spawn.sh` starts Pi with `-e` pointed at the secondmate home's own tracked `.pi/extensions/fm-primary-pi-watch.ts` and `.pi/extensions/fm-primary-turnend-guard.ts`, both already present from the secondmate home's git worktree. +For Kimi crews, `fm-spawn.sh` runs `fm-kimi-turnend-hook.sh install`, drops a per-task `.fm-kimi-turnend` pointer in the worktree, and records the matching private registry token for teardown. +Kimi continues to use the captain's normal Kimi home, including the existing config, skills, and memory; Firstmate does not create an isolated Kimi home. +The Kimi installer requires an existing regular non-symlink `~/.kimi-code/config.toml`, `python3` with `tomllib`, and `jq`; it validates but never serializes the captain's TOML and refuses before writing when the config is missing, malformed, or surprising or when either tool requirement is unavailable. +Its `remove` action excises only the marker-delimited Firstmate region and removes Firstmate's hook files. +For Pi and pi-signed secondmate launches, `fm-spawn.sh` starts the selected executable with `-e` pointed at the secondmate home's own tracked `.pi/extensions/fm-primary-pi-watch.ts` and `.pi/extensions/fm-primary-turnend-guard.ts`, both already present from the secondmate home's git worktree. ## Crew dispatch profiles (config/crew-dispatch.json) `config/crew-dispatch.json` is an optional local, gitignored file containing natural-language rules that firstmate reads before dispatching a crewmate or scout. -The shell scripts do not match those rules; firstmate chooses the best matching rule with judgment, resolves that rule directly or through a supported selector, and passes only concrete `--harness`, `--model`, and `--effort` flags to `fm-spawn.sh`. +The shell scripts do not match those rules; firstmate chooses the best matching rule with judgment, resolves its profile object or array under the operating contract in `AGENTS.md` section 4 and `quota-array-dispatch`, and passes only concrete `--harness`, `--model`, and `--effort` flags to `fm-spawn.sh`. When the file exists, `fm-spawn.sh` enforces that contract by refusing crewmate and scout spawns that lack an explicit harness (`--harness`, a positional adapter, or a raw launch command). Batch spawns satisfy the same requirement with a shared `--harness`. Secondmate spawns are exempt and still resolve through `config/secondmate-harness` and its optional model and effort tokens. -This section is the single owner of the canonical schema and its per-field semantics; `AGENTS.md` section 4 keeps only the dispatch procedure and points here. +This section is the single owner of the canonical schema and its per-field semantics. +`AGENTS.md` section 4 owns the always-loaded dispatch intake boundary, and `quota-array-dispatch` owns the pace-aware profile-array selection procedure. ```json { @@ -213,7 +241,6 @@ This section is the single owner of the canonical schema and its per-field seman "use": [ { "harness": "<adapter>", "model": "<optional model>", "effort": "<low|medium|high|xhigh|max, optional>" } ], - "select": "<optional strategy>", "why": "<optional rationale that helps firstmate choose>" } ], @@ -228,17 +255,14 @@ Both `use` and the optional top-level `default` accept either one profile object The single-object form stays fully backward-compatible, and every profile needs `harness`. Profile `model` and `effort` fields and rule `why` are optional. An omitted model or effort means the selected harness uses its own default for that axis. -Every profile array is an implicit quota-aware choice and does not need a selector property. -`select: "quota-balanced"` remains accepted on rules for compatibility and has the same behavior as an implicit array choice. -If no dispatch rule fits, firstmate resolves `default` through the same object-or-array selection path before falling back to `config/crew-harness`. +Every profile array is an implicit quota-aware choice resolved through `quota-array-dispatch`. +If no dispatch rule fits, firstmate resolves `default` through the same object-or-array path before falling back to `config/crew-harness`. If a selected profile carries an effort value the chosen harness does not accept, `fm-spawn.sh` records the requested `effort=` in task meta for traceability but omits the launch flag, and bootstrap reports the invalid harness/effort pair as a `CREW_DISPATCH` diagnostic when it is visible in the file. -Quota-aware selection is implemented by `bin/fm-dispatch-select.sh`, whose header owns provider and product mapping, relevant-window scoring, the stale-clear freshness margin, random tie-breaking, OS-backed random operational fallback, and safe selection-basis diagnostics. -Quota-data trouble never blocks dispatch, but malformed profile configuration remains an actionable validation error. See [`docs/examples/crew-dispatch.json`](examples/crew-dispatch.json) for a starting point to copy into local `config/crew-dispatch.json`. When the file exists, bootstrap validates it with `jq`. Valid files stay silent by default; with `FM_BOOTSTRAP_VERBOSE_FACTS=1`, bootstrap emits `BOOTSTRAP_INFO: crew dispatch active config/crew-dispatch.json`, one `BOOTSTRAP_INFO:` fact per rule, and one fact for the optional default profile set. -Malformed JSON, an empty or malformed rule/default array, an unverified harness, an unknown `select`, or an effort value unsupported by that harness is reported as `CREW_DISPATCH: invalid config/crew-dispatch.json - ...`; missing `jq` is reported through the normal `MISSING: jq` install-consent flow. -Because the spawn backstop is gated by file presence, any fallback path after a missing match, validation error, or missing `jq` still passes a resolved harness explicitly until the file is fixed or removed. +Malformed JSON, an empty or malformed rule/default array, an unverified harness, or an effort value unsupported by that harness is reported as `CREW_DISPATCH: invalid config/crew-dispatch.json - ...`; missing `jq` is reported through the normal `MISSING: jq` install-consent flow. +While the file remains present, no crewmate or scout spawn may proceed without an explicit resolved harness; malformed configuration must be reported and corrected rather than selected around. Secondmate homes inherit this file from the primary, so a secondmate's own crewmates apply the same dispatch profile behavior. ## Toolchain @@ -260,7 +284,7 @@ When `config/crew-dispatch.json` exists, bootstrap also requires `jq` for dispat When X mode is opted in, bootstrap also requires `curl` and `jq` before arming the relay poll shim. `tasks-axi` and `quota-axi` are required bootstrap tools in every profile, the same class as `lavish-axi`. An absent or incompatible `tasks-axi` reports `MISSING: tasks-axi (install: npm install -g tasks-axi)`; when `config/backlog-backend` is not `manual` and compatible `tasks-axi` is on `PATH`, bootstrap stays silent and firstmate uses its verbs for routine backlog mutations, otherwise it hand-edits `data/backlog.md` until installation is approved and completed. -An absent `quota-axi` reports `MISSING: quota-axi (install: npm install -g quota-axi)`; `bin/fm-dispatch-select.sh` still selects uniformly from the valid candidate array with an OS-backed random source when quota data is unavailable. +An absent `quota-axi` reports `MISSING: quota-axi (install: npm install -g quota-axi)`; firstmate cannot resolve a profile array until current quota output is available for every candidate. Bootstrap also reports a `TANGLE:` line when `FM_ROOT` is on a named non-default branch; follow the printed checkout remediation rather than treating it as an installable tool problem. In a read-only session that did not get the fleet lock, the same line is advisory and omits the checkout command. The locked session-start bootstrap step also runs a best-effort project clone refresh through `fm-fleet-sync.sh`. @@ -276,7 +300,7 @@ When a running home advances and its loaded instruction surface (`AGENTS.md`, `b If that send fails, bootstrap keeps an idempotent retry marker and emits `NUDGE_SECONDMATES:` with the failure reason. The same bootstrap run emits `SECONDMATE_LIVENESS:` only when a registered secondmate is skipped or its relaunch fails; already-live and successfully relaunched secondmates are handled silently. For a mid-session inherited local-material edit where tracked-file sync is not needed, run `bin/fm-config-push.sh`. -It uses the same live secondmate discovery and propagation helper as bootstrap, prints each live home's `crew-dispatch.json`, `crew-harness`, `backlog-backend`, `herdr-presentation-spaces`, and `data/captain-shared.md` result as `pushed`, `unchanged`, `skipped`, or `error`, and exits non-zero for real propagation errors or config-reread send failures. +It uses the same live secondmate discovery and propagation helper as bootstrap, prints each live home's `crew-dispatch.json`, `crew-harness`, `backlog-backend`, `backend`, `herdr-presentation-spaces`, `startup-memory-budget`, and `data/captain-shared.md` result as `pushed`, `unchanged`, `skipped`, or `error`, and exits non-zero for real propagation errors or config-reread send failures. When an allowlisted config item changes for an already-running home, it sends the literal-content reread pointer described in [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md); unchanged allowlisted config sends no pointer unless a previous delivery is pending. The locked bootstrap inheritance pass uses the same per-home changed-set and reread path for already-running homes; see `secondmate-provisioning` for the single contract owner. That live discovery starts from `state/*.meta` records with `kind=secondmate`; `data/secondmates.md` only backfills `home=` for older or incomplete meta records. @@ -372,9 +396,9 @@ FM_BACKEND= # optional runtime backend override for new spawns; tmux HERDR_SESSION=default # herdr-only: named session for normal backend ops; not enough for destructive cleanup (docs/herdr-backend.md) FM_BACKEND_HERDR_COMPOSER_LINES=20 # herdr-only: tail lines scanned by composer-state guard/fallback paths; idle-baseline submit confirmation uses agent-state FM_BACKEND_HERDR_IDLE_RE='^Type a message\.\.\.$' # herdr-only: empty-composer placeholder regex after shared ghost extraction plus border and prompt stripping -FM_BACKEND_HERDR_BARE_PROMPT_RE='^[❯›]' # herdr-only: verified agent glyphs recognized as an UNBORDERED (bare) composer row, e.g. claude's ❯ or codex's ›; shell glyphs remain unknown rather than empty, and de-emphasised ghost/placeholder text (dim or dark-truecolor) after an agent prompt reads empty via the shared fm_composer_strip_ghost (docs/herdr-backend.md "Incident (2026-07-08)", "Incident (2026-07-10)") -FM_BACKEND_HERDR_PI_COMPOSER_MAX_LINES=8 # herdr-only: maximum rows admitted between Pi's native-identity-corroborated separator pair; taller or ambiguous candidates stay unknown (docs/herdr-backend.md "Incident (2026-07-14)") -FM_BACKEND_HERDR_SUBMIT_POLLS=6 # herdr-only: agent-state samples spread across each Enter attempt's budget when confirming a submit (docs/herdr-backend.md "Native agent-state submit confirmation") +FM_BACKEND_HERDR_BARE_PROMPT_RE='^(❯|›)' # herdr-only: verified agent glyphs recognized as an UNBORDERED (bare) composer row, e.g. Claude's ❯ or Codex's ›; an alternation, not a `[...]` bracket expression, so a C-locale byte-decomposed match can never misfire on an unrelated multibyte glyph; shell glyphs remain unknown rather than empty, and de-emphasised ghost/placeholder text reads empty through shared fm_composer_strip_ghost (docs/herdr-backend.md "Composer and injection safety") +FM_BACKEND_HERDR_PI_COMPOSER_MAX_LINES=8 # herdr-only: maximum rows admitted between Pi's native-identity-corroborated separator pair; taller or ambiguous candidates stay unknown (docs/herdr-backend.md "Composer and injection safety") +FM_BACKEND_HERDR_SUBMIT_POLLS=6 # herdr-only: agent-state samples spread across each Enter attempt's budget when confirming a submit (docs/herdr-backend.md "Current transport behavior") FM_BACKEND_HERDR_SUBMIT_MIN_SLEEP=0.6 # herdr-only: minimum per-Enter confirmation budget before polling agent-state after an idle baseline FM_BACKEND_ORCA_COMPOSER_LINES=200 # orca-only: terminal-read lines scanned to locate the composer row for submit verification FM_BACKEND_ORCA_IDLE_RE='^Type a message\.\.\.$' # orca-only: empty-composer placeholder regex after border/prompt stripping @@ -406,10 +430,13 @@ FMX_FOLLOWUP_MAX_AGE_SECS=604800 # local window for posting X-mode completion FMX_FOLLOWUP_MAX_COUNT=3 # local cap on X-mode completion follow-ups per linked mention FM_LOCK_STALE_AFTER=2 # seconds before dead-pid lock records can be reclaimed; mid-acquire locks keep at least 2s grace FM_GUARD_GRACE=300 # seconds before guard warnings, arm health checks, and the primary turn-end guard treat a watcher beacon as stale -FM_ARM_CONFIRM_TIMEOUT=10 # seconds fm-watch-arm waits to confirm a fresh watcher before reporting FAILED +FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=800 # milliseconds the --claude turn-end guard waits for the Stop auto-arm's claim, health, or fresh rewake epoch before re-blocking +FM_CLAUDE_AUTOARM_EPOCH_FRESH=15 # seconds a recorded auto-arm rewake outcome counts as this event epoch's owned recovery +FM_CLAUDE_TURNEND_BLOCK_BUDGET=3 # consecutive --claude guard re-blocks before a degraded allow; safely below Claude Code's 8-block override +FM_ARM_CONFIRM_TIMEOUT=10 # seconds fm-watch-arm waits to confirm a fresh watcher before reporting FAILED; default 30 on Git Bash/MSYS FM_ARM_ATTACH_POLL=0.5 # seconds between checks while fm-watch-arm is attached to an existing healthy watcher cycle -FM_OPENCODE_ARM_READY_TIMEOUT_MS=12000 # milliseconds the OpenCode primary watcher plugin waits for an arm attempt to report started, healthy, wake, or failure -FM_PI_ARM_READY_TIMEOUT_MS=12000 # milliseconds the Pi watcher extension waits for a successor arm to report started or attached +FM_OPENCODE_ARM_READY_TIMEOUT_MS=12000 # milliseconds the OpenCode primary watcher plugin waits for an arm attempt to report started, healthy, wake, or failure; default 35000 on Windows to stay above the MSYS confirm budget +FM_PI_ARM_READY_TIMEOUT_MS=12000 # milliseconds the Pi watcher extension waits for a successor arm to report started or attached; default 35000 on Windows to stay above the MSYS confirm budget FM_WATCH_ARM_RETIRE_TIMEOUT_MS=1000 # milliseconds Pi/OpenCode wait for an unready successor arm to exit before abandoning retries FM_WATCH_REARM_RETRY_BASE_MS=250 # Pi/OpenCode adapter base delay for continuity restoration retries FM_WATCH_REARM_RETRY_MAX_MS=4000 # Pi/OpenCode adapter cap for exponential continuity retry delay @@ -421,6 +448,7 @@ FM_SIGNAL_GRACE=30 # seconds to coalesce nearby status and turn-end signals FM_CAPTAIN_RE='done:|needs-decision:|blocked:|failed:|PR ready|checks green|ready in branch|merged' # captain-relevant status regex; nonterminal progress verbs remain excluded even when their prose matches FM_CLASSIFY_PAUSED_VERB=paused # leading status verb for a declared external wait; excluded from FM_CAPTAIN_RE and distinct from blocked FM_STALE_ESCALATE_SECS=240 # idle seconds before a provably-working stale pane escalates; stale panes whose crew is not provably working surface immediately unless they declare the pause verb +FM_BUSY_TURN_MAX_SECS=3600 # maximum age of a busy pane's latest state/<id>.turn-ended marker, or its state/<id>.meta spawn record before any turn completes, before the same wedge escalation used for a provably-working non-busy stale takes over; inspection-only, never an automatic interrupt or restart FM_PAUSE_RESURFACE_SECS=3600 # seconds before an idle declared external wait re-surfaces for a recheck in the watcher or away-mode daemon FM_WEDGE_DEMAND_INSPECT_COUNT=3 # consecutive provably-working stale escalations on the same unchanged pane before demand-deep-inspection is added FM_WATCH_TRIAGE_LOG_MAX_BYTES=262144 # size cap for the watcher's absorbed-wake debug log @@ -433,7 +461,7 @@ FM_STALE_WORKTREE_LOCK_RETRY_WAIT_SECS= # legacy alias for FM_TREEHOUSE_RETURN FM_FLEET_SYNC_PACKED_REFS_LOCK_RETRIES=3 # fetch retries after fm-fleet-sync.sh hits the orphaned .git/packed-refs.lock signature FM_FLEET_SYNC_PACKED_REFS_LOCK_RETRY_WAIT_SECS=1 # seconds fm-fleet-sync.sh waits before each of those retries FM_FLEET_SYNC_PACKED_REFS_LOCK_AGE_SECS=30 # min mtime age before fm-fleet-sync.sh treats a leftover packed-refs.lock as provably stale -FM_BUSY_REGEX='esc (to )?interrupt|Working\.\.\.|Ctrl\+c:cancel' # busy-pane signatures, shared by watcher, fm-crew-state pane fallback, and tmux helper +FM_BUSY_REGEX= # optional global override for every harness-scoped busy-pane matcher; unset uses each recorded harness's verified signature FM_COMPOSER_IDLE_RE= # optional empty-composer regex, applied after ghost and border stripping FM_COMPOSER_GHOST_LUMA_MAX=128 # fleet-wide: max perceived luminance (0.299R+0.587G+0.114B, 0-255) for a TRUECOLOR foreground to count as de-emphasised ghost/placeholder text and be stripped; dim/faint (SGR 2) is stripped regardless. Assumes a dark terminal theme (bin/fm-composer-lib.sh's fm_composer_strip_ghost, shared by the tmux and herdr composer readers) GROK_HOME= # optional Grok config home for firstmate's global grok turn-end hook; defaults to ~/.grok diff --git a/docs/documentation-audiences.json b/docs/documentation-audiences.json new file mode 100644 index 00000000000..c9dc1dcb5e9 --- /dev/null +++ b/docs/documentation-audiences.json @@ -0,0 +1,375 @@ +{ + "version": 1, + "scope": { + "trackedPatterns": [ + "*.md", + "*.mdx", + "*.rst", + "*.txt", + "docs/examples/*" + ], + "description": "Tracked Markdown-like prose, including symlinked aliases, plus copyable documentation examples. Script headers, executable configuration, generated JSON evidence, and legal text keep their existing code/config/legal owners." + }, + "allowedAudiences": [ + "public-product", + "operator-current", + "operator-example", + "maintainer-architecture", + "maintainer-verification", + "agent-runtime" + ], + "setupAudiences": [ + "public-product", + "operator-current", + "operator-example" + ], + "readmeSetupTargets": [ + "docs/configuration.md", + "docs/wedge-alarm.md", + "docs/tmux-backend.md", + "docs/herdr-backend.md", + "docs/zellij-backend.md", + "docs/orca-backend.md", + "docs/cmux-backend.md" + ], + "requiredOwnerPointers": [ + { + "source": ".agents/skills/firstmate-coding-guidelines/SKILL.md", + "target": "docs/documentation-audiences.md" + }, + { + "source": ".no-mistakes.yaml", + "target": "docs/documentation-audiences.md" + }, + { + "source": "CONTRIBUTING.md", + "target": "docs/documentation-audiences.md" + }, + { + "source": "README.md", + "target": "docs/documentation-audiences.md" + }, + { + "source": "docs/documentation-audiences.md", + "target": "docs/documentation-audiences.json" + }, + { + "source": "docs/calm.md", + "target": "docs/calm-mode-feasibility.md" + }, + { + "source": "docs/calm-mode-feasibility.md", + "target": "docs/calm.md" + }, + { + "source": "docs/sessionstart-nudge.md", + "target": "docs/verification/supervision.md" + }, + { + "source": "docs/turnend-guard.md", + "target": "docs/verification/supervision.md" + }, + { + "source": "docs/watcher-continuity.md", + "target": "docs/verification/supervision.md" + }, + { + "source": "docs/wedge-alarm.md", + "target": "docs/verification/supervision.md" + }, + { + "source": "docs/tmux-backend.md", + "target": "docs/verification/runtime-backends.md" + }, + { + "source": "docs/herdr-backend.md", + "target": "docs/verification/runtime-backends.md" + }, + { + "source": "docs/zellij-backend.md", + "target": "docs/verification/runtime-backends.md" + }, + { + "source": "docs/orca-backend.md", + "target": "docs/verification/runtime-backends.md" + }, + { + "source": "docs/cmux-backend.md", + "target": "docs/verification/runtime-backends.md" + }, + { + "source": "docs/codex-app-backend.md", + "target": "docs/verification/runtime-backends.md" + } + ], + "surfaces": [ + { + "path": ".agents/skills/afk/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/ahoy/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/ask-user-authority/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/bearings/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/bootstrap-diagnostics/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/decision-hold-lifecycle/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/diagnostic-reasoning/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/firstmate-codexapp/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/firstmate-coding-guidelines/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/firstmate-orca/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/fmx-respond/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/harness-adapters/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/i-have-adhd/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/mobile-mode/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/project-management/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/quota-array-dispatch/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/secondmate-provisioning/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/secrets-management/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/stow/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/stuck-crewmate-recovery/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/updatefirstmate/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": "AGENTS.md", + "audience": "agent-runtime" + }, + { + "path": "CLAUDE.md", + "audience": "agent-runtime" + }, + { + "path": "CONTRIBUTING.md", + "audience": "maintainer-architecture" + }, + { + "path": "README.md", + "audience": "public-product" + }, + { + "path": "docs/architecture.md", + "audience": "maintainer-architecture" + }, + { + "path": "docs/arm-pretool-check.md", + "audience": "maintainer-architecture" + }, + { + "path": "docs/calm-mode-feasibility.md", + "audience": "maintainer-verification" + }, + { + "path": "docs/calm.md", + "audience": "operator-current" + }, + { + "path": "docs/cd-guard.md", + "audience": "maintainer-architecture" + }, + { + "path": "docs/cmux-backend.md", + "audience": "operator-current" + }, + { + "path": "docs/codex-app-backend.md", + "audience": "operator-current" + }, + { + "path": "docs/configuration.md", + "audience": "operator-current" + }, + { + "path": "docs/decision-hold-lifecycle.md", + "audience": "maintainer-architecture" + }, + { + "path": "docs/documentation-audiences.md", + "audience": "maintainer-architecture" + }, + { + "path": "docs/examples/crew-dispatch.json", + "audience": "operator-example" + }, + { + "path": "docs/examples/doppler-oidc-job.yml", + "audience": "operator-example" + }, + { + "path": "docs/examples/doppler-service-token-job.yml", + "audience": "operator-example" + }, + { + "path": "docs/examples/project-secrets-policy.json", + "audience": "operator-example" + }, + { + "path": "docs/examples/wedge-alarm", + "audience": "operator-example" + }, + { + "path": "docs/fm-test-isolation-proof.md", + "audience": "maintainer-verification" + }, + { + "path": "docs/fm-test-portable-shards.md", + "audience": "maintainer-verification" + }, + { + "path": "docs/gitlab-merge-watch.md", + "audience": "maintainer-verification" + }, + { + "path": "docs/herdr-backend.md", + "audience": "operator-current" + }, + { + "path": "docs/moshi-mobile-review.md", + "audience": "operator-current" + }, + { + "path": "docs/orca-backend.md", + "audience": "operator-current" + }, + { + "path": "docs/promotion-ladder.md", + "audience": "operator-current" + }, + { + "path": "docs/scripts.md", + "audience": "operator-current" + }, + { + "path": "docs/sessionstart-nudge.md", + "audience": "operator-current" + }, + { + "path": "docs/subagent-guard.md", + "audience": "maintainer-architecture" + }, + { + "path": "docs/supervision-protocols/claude.md", + "audience": "agent-runtime" + }, + { + "path": "docs/supervision-protocols/codex.md", + "audience": "agent-runtime" + }, + { + "path": "docs/supervision-protocols/grok.md", + "audience": "agent-runtime" + }, + { + "path": "docs/supervision-protocols/opencode.md", + "audience": "agent-runtime" + }, + { + "path": "docs/supervision-protocols/pi.md", + "audience": "agent-runtime" + }, + { + "path": "docs/supervision-protocols/unknown.md", + "audience": "agent-runtime" + }, + { + "path": "docs/tmux-backend.md", + "audience": "operator-current" + }, + { + "path": "docs/turnend-guard.md", + "audience": "operator-current" + }, + { + "path": "docs/toolchain-versions.md", + "audience": "maintainer-verification" + }, + { + "path": "docs/verification/runtime-backends.md", + "audience": "maintainer-verification" + }, + { + "path": "docs/verification/moshi-mobile-review.md", + "audience": "maintainer-verification" + }, + { + "path": "docs/verification/stow-memory.md", + "audience": "maintainer-verification" + }, + { + "path": "docs/verification/supervision.md", + "audience": "maintainer-verification" + }, + { + "path": "docs/watcher-continuity.md", + "audience": "operator-current" + }, + { + "path": "docs/wedge-alarm.md", + "audience": "operator-current" + }, + { + "path": "docs/zellij-backend.md", + "audience": "operator-current" + }, + { + "path": "skills/stow/SKILL.md", + "audience": "public-product" + } + ] +} diff --git a/docs/documentation-audiences.md b/docs/documentation-audiences.md new file mode 100644 index 00000000000..ca569041a04 --- /dev/null +++ b/docs/documentation-audiences.md @@ -0,0 +1,28 @@ +# Documentation audiences + +[`documentation-audiences.json`](documentation-audiences.json) is the machine-consumed classification owner for every maintained prose surface. +`bin/fm-doc-audience-check.sh` validates exact inventory coverage, README setup routing, required owner pointers, and local link targets. +Audience metadata is centralized there rather than copied into front matter on every page. + +The audience classes have one placement purpose each: + +- `public-product` introduces the product or provides standalone public material. +- `operator-current` explains current behavior, setup, supported limits, stable invariants, concise rationale, and current verification entry points. +- `operator-example` is copyable current setup material. +- `maintainer-architecture` explains stable ownership, extension points, mechanism boundaries, and safety rationale for contributors. +- `maintainer-verification` records repeatable evidence for an active guarantee and may include dates, versions, exact commands, and exact output. +- `agent-runtime` is loaded or rendered as an operating contract for Firstmate agents rather than read as product documentation. + +The knowledge-placement policy is owned by [`firstmate-coding-guidelines`](../.agents/skills/firstmate-coding-guidelines/SKILL.md). +Task-specific chronology, delivery transcripts, temporary paths, branches, failed hypotheses, and one-off process identifiers stay in private task reports or PR evidence by default. +Before removing that evidence from a tracked page, distill every unique current fact into its classified owner and retain a focused regression pointer. + +Run the structural check directly with: + +```sh +bin/fm-doc-audience-check.sh +``` + +The check intentionally does not lint dates, versions, commands, paths, incident language, or transcript-like prose. +Those forms are legitimate in maintainer verification and require semantic review rather than keyword heuristics. +For every changed prose surface, review its audience, authoritative owner, current relevance, evidence destination, and unique safety facts, then repeat that review over the complete branch diff after all fixes. diff --git a/docs/examples/crew-dispatch.json b/docs/examples/crew-dispatch.json index 886557f95a6..23a5391d20a 100644 --- a/docs/examples/crew-dispatch.json +++ b/docs/examples/crew-dispatch.json @@ -16,7 +16,7 @@ { "harness": "claude", "model": "claude-sonnet-5", "effort": "high" }, { "harness": "codex", "model": "gpt-5.5", "effort": "high" } ], - "why": "Arrays are quota-aware automatically, so use a strong coding profile with the most available relevant quota." + "why": "Firstmate compares every candidate with current relevant quota and pace before dispatch, so use a strong coding profile." } ], "default": [ diff --git a/docs/fm-test-isolation-proof.json b/docs/fm-test-isolation-proof.json index 7ffe660250b..ec605bf10f2 100644 --- a/docs/fm-test-isolation-proof.json +++ b/docs/fm-test-isolation-proof.json @@ -1,196 +1,36 @@ { "concurrency": 4, - "finished_at": "2026-07-22T04:08:06Z", + "finished_at": "2026-07-29T23:21:46Z", "fm_test_run_jobs_enabled": false, "kind": "isolation-proof", "production_sharding_enabled": false, - "run_id": "fm-isolation-1784693155237-99474", + "run_id": "fm-isolation-1785367157179-18165", "scripts": [ - { - "duration_ms": 29102, - "exit": 0, - "path": "tests/fm-arm-pretool-check.test.sh", - "worker": 1 - }, - { - "duration_ms": 35417, - "exit": 0, - "path": "tests/fm-backend-herdr.test.sh", - "worker": 2 - }, - { - "duration_ms": 897, - "exit": 0, - "path": "tests/fm-brief.test.sh", - "worker": 3 - }, - { - "duration_ms": 90, - "exit": 0, - "path": "tests/fm-captain-translation-contract.test.sh", - "worker": 4 - }, - { - "duration_ms": 18610, - "exit": 0, - "path": "tests/fm-cd-pretool-check.test.sh", - "worker": 5 - }, - { - "duration_ms": 2803, - "exit": 0, - "path": "tests/fm-composer-ghost.test.sh", - "worker": 6 - }, - { - "duration_ms": 68, - "exit": 0, - "path": "tests/fm-composer-lib.test.sh", - "worker": 7 - }, - { - "duration_ms": 19896, - "exit": 0, - "path": "tests/fm-crew-state.test.sh", - "worker": 8 - }, - { - "duration_ms": 21133, - "exit": 0, - "path": "tests/fm-decision-hold-lifecycle.test.sh", - "worker": 9 - }, - { - "duration_ms": 874, - "exit": 0, - "path": "tests/fm-dispatch-select.test.sh", - "worker": 10 - }, - { - "duration_ms": 348, - "exit": 0, - "path": "tests/fm-ensure-agents-md.test.sh", - "worker": 11 - }, - { - "duration_ms": 5963, - "exit": 0, - "path": "tests/fm-grok-harness.test.sh", - "worker": 12 - }, - { - "duration_ms": 12517, - "exit": 0, - "path": "tests/fm-herdr-lab.test.sh", - "worker": 13 - }, - { - "duration_ms": 232, - "exit": 0, - "path": "tests/fm-instruction-owners.test.sh", - "worker": 14 - }, - { - "duration_ms": 1274, - "exit": 0, - "path": "tests/fm-lint.test.sh", - "worker": 15 - }, - { - "duration_ms": 201, - "exit": 0, - "path": "tests/fm-nm-test-contract.test.sh", - "worker": 16 - }, - { - "duration_ms": 36, - "exit": 0, - "path": "tests/fm-no-mistakes-ownership.test.sh", - "worker": 17 - }, - { - "duration_ms": 1056, - "exit": 0, - "path": "tests/fm-pi-primary-types.test.sh", - "worker": 18 - }, - { - "duration_ms": 8939, - "exit": 0, - "path": "tests/fm-pr-merge.test.sh", - "worker": 19 - }, - { - "duration_ms": 2549, - "exit": 0, - "path": "tests/fm-review-diff.test.sh", - "worker": 20 - }, - { - "duration_ms": 6953, - "exit": 0, - "path": "tests/fm-send-popup-settle.test.sh", - "worker": 21 - }, - { - "duration_ms": 3524, - "exit": 0, - "path": "tests/fm-send-settle.test.sh", - "worker": 22 - }, - { - "duration_ms": 1551, - "exit": 0, - "path": "tests/fm-send-strict.test.sh", - "worker": 23 - }, - { - "duration_ms": 684, - "exit": 0, - "path": "tests/fm-spawn-batch.test.sh", - "worker": 24 - }, - { - "duration_ms": 57, - "exit": 0, - "path": "tests/fm-stow-contract.test.sh", - "worker": 25 - }, - { - "duration_ms": 283, - "exit": 0, - "path": "tests/fm-supervision-instructions.test.sh", - "worker": 26 - }, - { - "duration_ms": 4645, - "exit": 0, - "path": "tests/fm-test-run.test.sh", - "worker": 27 - }, - { - "duration_ms": 2552, - "exit": 0, - "path": "tests/fm-tmux-submit-busy.test.sh", - "worker": 28 - }, - { - "duration_ms": 104, - "exit": 0, - "path": "tests/fm-transition-lib.test.sh", - "worker": 29 - }, - { - "duration_ms": 38449, - "exit": 0, - "path": "tests/fm-x-mode.test.sh", - "worker": 30 - } + {"duration_ms": 46788, "exit": 0, "path": "tests/fm-arm-pretool-check.test.sh", "worker": 1}, + {"duration_ms": 48294, "exit": 0, "path": "tests/fm-backend-herdr.test.sh", "worker": 2}, + {"duration_ms": 2224, "exit": 0, "path": "tests/fm-brief.test.sh", "worker": 3}, + {"duration_ms": 34207, "exit": 0, "path": "tests/fm-cd-pretool-check.test.sh", "worker": 4}, + {"duration_ms": 9065, "exit": 0, "path": "tests/fm-composer-ghost.test.sh", "worker": 5}, + {"duration_ms": 64, "exit": 0, "path": "tests/fm-composer-lib.test.sh", "worker": 6}, + {"duration_ms": 25365, "exit": 0, "path": "tests/fm-crew-state.test.sh", "worker": 7}, + {"duration_ms": 30771, "exit": 0, "path": "tests/fm-decision-hold-lifecycle.test.sh", "worker": 8}, + {"duration_ms": 581, "exit": 0, "path": "tests/fm-ensure-agents-md.test.sh", "worker": 9}, + {"duration_ms": 6251, "exit": 0, "path": "tests/fm-grok-harness.test.sh", "worker": 10}, + {"duration_ms": 15422, "exit": 0, "path": "tests/fm-herdr-lab.test.sh", "worker": 11}, + {"duration_ms": 5237, "exit": 0, "path": "tests/fm-lint.test.sh", "worker": 12}, + {"duration_ms": 2945, "exit": 0, "path": "tests/fm-pi-primary-types.test.sh", "worker": 13}, + {"duration_ms": 8564, "exit": 0, "path": "tests/fm-pr-merge.test.sh", "worker": 14}, + {"duration_ms": 2875, "exit": 0, "path": "tests/fm-review-diff.test.sh", "worker": 15}, + {"duration_ms": 5644, "exit": 0, "path": "tests/fm-send-popup-settle.test.sh", "worker": 16}, + {"duration_ms": 2911, "exit": 0, "path": "tests/fm-send-settle.test.sh", "worker": 17}, + {"duration_ms": 2747, "exit": 0, "path": "tests/fm-send-strict.test.sh", "worker": 18}, + {"duration_ms": 855, "exit": 0, "path": "tests/fm-spawn-batch.test.sh", "worker": 19}, + {"duration_ms": 703, "exit": 0, "path": "tests/fm-supervision-instructions.test.sh", "worker": 20}, + {"duration_ms": 15674, "exit": 0, "path": "tests/fm-test-run.test.sh", "worker": 21}, + {"duration_ms": 4816, "exit": 0, "path": "tests/fm-tmux-submit-busy.test.sh", "worker": 22}, + {"duration_ms": 248, "exit": 0, "path": "tests/fm-transition-lib.test.sh", "worker": 23}, + {"duration_ms": 52939, "exit": 0, "path": "tests/fm-x-mode.test.sh", "worker": 24} ], - "started_at": "2026-07-22T04:05:55Z", - "summary": { - "duration_ms": 131001, - "failed": 0, - "total": 30 - } + "started_at": "2026-07-29T23:19:17Z", + "summary": {"duration_ms": 149010, "failed": 0, "total": 24} } diff --git a/docs/fm-test-isolation-proof.md b/docs/fm-test-isolation-proof.md index 9bfe19e40be..716dca73a56 100644 --- a/docs/fm-test-isolation-proof.md +++ b/docs/fm-test-isolation-proof.md @@ -1,61 +1,39 @@ -# Firstmate test isolation proof (Phase 2) +# Firstmate test isolation proof -This document is the archived concurrent isolation proof for the portable parallel candidate set. -It is the human-readable companion to `bin/fm-test-isolation-proof.sh`. -Phase 4 production portable shards and bounded local `fm-test-run.sh --jobs` for this exact set are owned by `bin/fm-test-run.sh` and documented in [fm-test-portable-shards.md](fm-test-portable-shards.md). -The archived proof JSON below still records the Phase 2 proof-time flags (`production_sharding_enabled` / `fm_test_run_jobs_enabled` false at proof time). +This record is the concurrent isolation proof for the portable parallel candidate set. +`bin/fm-test-isolation-proof.sh` is the authoritative harness and `docs/fm-test-isolation-proof.json` is the machine-readable result. +`bin/fm-test-run.sh` owns the production lane partition. -## Owner +## Verification -- Harness: `bin/fm-test-isolation-proof.sh` -- Contract tests: `tests/fm-test-isolation-proof.test.sh` -- Family labels (Phase 1): `bin/fm-test-run.sh` -- Timing evidence used for planning: CI artifact `fm-test-timing` from Phase 1 PR #825 - -## Proof posture +- Date: 2026-07-29 +- Command: `bin/fm-test-isolation-proof.sh --jobs 4 --json /tmp/fm-source-content-test-cleanup-r1-isolation.json` +- Result: `FM_ISOLATION_SUMMARY total=24 failed=0 concurrency=4 duration_ms=149010` | Field | Value | |---|---| -| `run_id` | `fm-isolation-1784693155237-99474` | -| `started_at` | `2026-07-22T04:05:55Z` | -| `finished_at` | `2026-07-22T04:08:06Z` | -| concurrency | **4** | -| candidates | **30** | -| failed | **0** | -| wall duration_ms | **131001** (~131.0s) | -| `production_sharding_enabled` | `False` | -| `fm_test_run_jobs_enabled` | `False` | -| host proof date | 2026-07-22 (UTC day of archive write) | - -Isolation checks that passed with this run: - -- Distinct mode-`0700` temporary roots per worker under a proof-owned parent -- Per-worker `TMPDIR`/`TMP` so `mktemp` / `fm_test_tmproot` stay private -- Ambient `FM_HOME` / `FM_*_OVERRIDE` cleared for each worker -- `git config --global` snapshot unchanged before/after the matrix -- Aggregate failure reporting (any non-zero candidate fails the harness; no retry-until-green) +| `run_id` | `fm-isolation-1785367157179-18165` | +| `started_at` | `2026-07-29T23:19:17Z` | +| `finished_at` | `2026-07-29T23:21:46Z` | +| concurrency | 4 | +| candidates | 24 | +| failed | 0 | +| wall duration | 149010 ms | -## Exact candidate set - -Sorted paths as selected by `bin/fm-test-isolation-proof.sh --list` at proof time: +## Candidate set - `tests/fm-arm-pretool-check.test.sh` - `tests/fm-backend-herdr.test.sh` - `tests/fm-brief.test.sh` -- `tests/fm-captain-translation-contract.test.sh` - `tests/fm-cd-pretool-check.test.sh` - `tests/fm-composer-ghost.test.sh` - `tests/fm-composer-lib.test.sh` - `tests/fm-crew-state.test.sh` - `tests/fm-decision-hold-lifecycle.test.sh` -- `tests/fm-dispatch-select.test.sh` - `tests/fm-ensure-agents-md.test.sh` - `tests/fm-grok-harness.test.sh` - `tests/fm-herdr-lab.test.sh` -- `tests/fm-instruction-owners.test.sh` - `tests/fm-lint.test.sh` -- `tests/fm-nm-test-contract.test.sh` -- `tests/fm-no-mistakes-ownership.test.sh` - `tests/fm-pi-primary-types.test.sh` - `tests/fm-pr-merge.test.sh` - `tests/fm-review-diff.test.sh` @@ -63,111 +41,51 @@ Sorted paths as selected by `bin/fm-test-isolation-proof.sh --list` at proof tim - `tests/fm-send-settle.test.sh` - `tests/fm-send-strict.test.sh` - `tests/fm-spawn-batch.test.sh` -- `tests/fm-stow-contract.test.sh` - `tests/fm-supervision-instructions.test.sh` - `tests/fm-test-run.test.sh` - `tests/fm-tmux-submit-busy.test.sh` - `tests/fm-transition-lib.test.sh` - `tests/fm-x-mode.test.sh` -## Per-candidate durations (concurrent run) +## Durations | duration_ms | exit | worker | script | |---:|---:|---:|---| -| 38449 | 0 | 30 | `tests/fm-x-mode.test.sh` | -| 35417 | 0 | 2 | `tests/fm-backend-herdr.test.sh` | -| 29102 | 0 | 1 | `tests/fm-arm-pretool-check.test.sh` | -| 21133 | 0 | 9 | `tests/fm-decision-hold-lifecycle.test.sh` | -| 19896 | 0 | 8 | `tests/fm-crew-state.test.sh` | -| 18610 | 0 | 5 | `tests/fm-cd-pretool-check.test.sh` | -| 12517 | 0 | 13 | `tests/fm-herdr-lab.test.sh` | -| 8939 | 0 | 19 | `tests/fm-pr-merge.test.sh` | -| 6953 | 0 | 21 | `tests/fm-send-popup-settle.test.sh` | -| 5963 | 0 | 12 | `tests/fm-grok-harness.test.sh` | -| 4645 | 0 | 27 | `tests/fm-test-run.test.sh` | -| 3524 | 0 | 22 | `tests/fm-send-settle.test.sh` | -| 2803 | 0 | 6 | `tests/fm-composer-ghost.test.sh` | -| 2552 | 0 | 28 | `tests/fm-tmux-submit-busy.test.sh` | -| 2549 | 0 | 20 | `tests/fm-review-diff.test.sh` | -| 1551 | 0 | 23 | `tests/fm-send-strict.test.sh` | -| 1274 | 0 | 15 | `tests/fm-lint.test.sh` | -| 1056 | 0 | 18 | `tests/fm-pi-primary-types.test.sh` | -| 897 | 0 | 3 | `tests/fm-brief.test.sh` | -| 874 | 0 | 10 | `tests/fm-dispatch-select.test.sh` | -| 684 | 0 | 24 | `tests/fm-spawn-batch.test.sh` | -| 348 | 0 | 11 | `tests/fm-ensure-agents-md.test.sh` | -| 283 | 0 | 26 | `tests/fm-supervision-instructions.test.sh` | -| 232 | 0 | 14 | `tests/fm-instruction-owners.test.sh` | -| 201 | 0 | 16 | `tests/fm-nm-test-contract.test.sh` | -| 104 | 0 | 29 | `tests/fm-transition-lib.test.sh` | -| 90 | 0 | 4 | `tests/fm-captain-translation-contract.test.sh` | -| 68 | 0 | 7 | `tests/fm-composer-lib.test.sh` | -| 57 | 0 | 25 | `tests/fm-stow-contract.test.sh` | -| 36 | 0 | 17 | `tests/fm-no-mistakes-ownership.test.sh` | - -## Audit notes (why this set) - -Source families from the Phase 1 manifest and scout report §3.1: - -1. **pure-contract-unit** candidates audited from the Phase 1 family manifest, minus deliberate serial exclusions -2. **Extra hermetic candidates** after static audit: fake backend, private git fixtures, stubbed network - -The harness pins this exact archived set and does not automatically admit later family additions. -A candidate-set change requires a new audit and concurrent proof archive. - -### Included extras (beyond pure-contract-unit) - -| Script | Why included | -|---|---| -| `tests/fm-backend-herdr.test.sh` | Fake Herdr CLI + private temps; no real Herdr binary | -| `tests/fm-send-strict.test.sh` | Fake tmux PATH shim; private `FM_HOME` | -| `tests/fm-spawn-batch.test.sh` | Argument routing only; no real windows/worktrees | -| `tests/fm-pr-merge.test.sh` | Fake `gh`/`gh-axi`; private state | -| `tests/fm-review-diff.test.sh` | Local git fixtures via `fm_git_*`; no live forge | -| `tests/fm-x-mode.test.sh` | Fake `curl`; inert without token | - -### Deliberately serial (kept out of this pool) - -Run `bin/fm-test-isolation-proof.sh --list-exclusions` for the machine-readable list. -High-signal classes: - -| Class | Examples | Reason | -|---|---|---| -| Process-holder pure unit | `fm-continuity-pretool-check` | Background `sleep 300` lock-holder process | -| Watcher / wake / locks | `fm-watcher-lock`, `fm-wake-queue`, ... | Intentional process locks and daemon races | -| AFK | `fm-afk-inject-e2e`, ... | Daemon lifecycle and inject path | -| Real Herdr | `fm-backend-herdr-smoke`, presentation e2e, ... | Named labs, session-global locks; Herdr lane is Phase 3+ | -| Real tmux smoke | `fm-backend-tmux-smoke` | Real multiplexer server (even on private socket) | -| Live harness opt-in | `fm-*-live-e2e` | Real interactive agents | -| GUI backends | cmux smoke | Shared GUI app | -| Gray-zone git/spawn | `fm-backend`, spawn settle/profile, teardown | Heavier worktree or lock-race matrices | -| Watcher-adjacent forge security | `fm-pr-check-security` | `.watch.lock` / poll security surface | -| Self | `fm-test-isolation-proof.test.sh` | Must not re-enter the concurrent matrix | - -### Small isolation fix landed with this phase - -`tests/fm-arm-pretool-check.test.sh` no longer writes Claude deny stderr to a fixed `/tmp/fm-arm-pretool-check-claude-stderr.$$` path. -It uses `mktemp` under `TMPDIR` so concurrent workers cannot collide on a global temp name pattern. - -## Failures - -None. -Every candidate exited 0 under concurrency=4. - -Policy: a script that fails only under concurrency is **removed** from the candidate set and investigated. -It is never retried into green, skipped more broadly, or weakened in assertions. - -## What this phase did not do (Phase 2 scope) - -- Did not land production CI Behavior matrix / shard jobs (Phase 4) -- Did not add general `bin/fm-test-run.sh --jobs` (Phase 4 enables it only for this proven set) -- Did not land the Herdr install lane (Phase 3) -- Did not re-run the complete local suite as part of this proof (focused matrix only) - -## How to re-run +| 52939 | 0 | 24 | `tests/fm-x-mode.test.sh` | +| 48294 | 0 | 2 | `tests/fm-backend-herdr.test.sh` | +| 46788 | 0 | 1 | `tests/fm-arm-pretool-check.test.sh` | +| 34207 | 0 | 4 | `tests/fm-cd-pretool-check.test.sh` | +| 30771 | 0 | 8 | `tests/fm-decision-hold-lifecycle.test.sh` | +| 25365 | 0 | 7 | `tests/fm-crew-state.test.sh` | +| 15674 | 0 | 21 | `tests/fm-test-run.test.sh` | +| 15422 | 0 | 11 | `tests/fm-herdr-lab.test.sh` | +| 9065 | 0 | 5 | `tests/fm-composer-ghost.test.sh` | +| 8564 | 0 | 14 | `tests/fm-pr-merge.test.sh` | +| 6251 | 0 | 10 | `tests/fm-grok-harness.test.sh` | +| 5644 | 0 | 16 | `tests/fm-send-popup-settle.test.sh` | +| 5237 | 0 | 12 | `tests/fm-lint.test.sh` | +| 4816 | 0 | 22 | `tests/fm-tmux-submit-busy.test.sh` | +| 2945 | 0 | 13 | `tests/fm-pi-primary-types.test.sh` | +| 2911 | 0 | 17 | `tests/fm-send-settle.test.sh` | +| 2875 | 0 | 15 | `tests/fm-review-diff.test.sh` | +| 2747 | 0 | 18 | `tests/fm-send-strict.test.sh` | +| 2224 | 0 | 3 | `tests/fm-brief.test.sh` | +| 855 | 0 | 19 | `tests/fm-spawn-batch.test.sh` | +| 703 | 0 | 20 | `tests/fm-supervision-instructions.test.sh` | +| 581 | 0 | 9 | `tests/fm-ensure-agents-md.test.sh` | +| 248 | 0 | 23 | `tests/fm-transition-lib.test.sh` | +| 64 | 0 | 6 | `tests/fm-composer-lib.test.sh` | + +## Scope + +Each worker used a separate mode-`0700` temporary root and private `TMPDIR` and `TMP`. +The harness cleared ambient `FM_HOME` and `FM_*_OVERRIDE` values for every worker and verified that global Git configuration was unchanged. +A candidate failure fails the aggregate run and requires investigation rather than a retry. + +## Re-run ```sh bin/fm-test-isolation-proof.sh --list bin/fm-test-isolation-proof.sh --jobs 4 --json /tmp/fm-isolation-proof.json -bash tests/fm-test-isolation-proof.test.sh +bin/fm-test-run.sh --check-coverage ``` diff --git a/docs/fm-test-portable-shards.md b/docs/fm-test-portable-shards.md index 9afa8887506..0bfa5e6bee4 100644 --- a/docs/fm-test-portable-shards.md +++ b/docs/fm-test-portable-shards.md @@ -1,88 +1,67 @@ -# Firstmate portable test shards (Phase 4) +# Firstmate portable test shards -This document records how the two portable parallel CI shards were balanced from measured evidence. -Composition and execution are owned by `bin/fm-test-run.sh` (`--lane portable-parallel-1` / `portable-parallel-2` / `portable-serial`). -The proven-isolated candidate set remains owned by `bin/fm-test-isolation-proof.sh`. +`bin/fm-test-run.sh` owns portable lane composition and execution. +`bin/fm-test-isolation-proof.sh` owns the proven-isolated candidate set. -## Inputs +## Verification inputs -| Input | Owner / source | -|---|---| -| Proven-isolated set (30 scripts) | `bin/fm-test-isolation-proof.sh --list` and `docs/fm-test-isolation-proof.md` | -| Phase 1 serial durations | CI timing artifacts `fm-test-timing` from main after #825 / #832 / #834 | -| Real-Herdr family | `bin/fm-test-run.sh --family real-herdr-gated` (dedicated required CI lane) | +The current candidate timings came from the 2026-07-29 concurrent proof recorded in [fm-test-isolation-proof.md](fm-test-isolation-proof.md). +The proof ran 24 candidates with four workers and no failures. -Phase 1 averages used for balance (mean of available serial `duration_ms` across those artifacts): - -| duration_ms (avg) | script | +| duration_ms | script | |---:|---| -| 29639 | `tests/fm-arm-pretool-check.test.sh` | -| 25402 | `tests/fm-decision-hold-lifecycle.test.sh` | -| 19428 | `tests/fm-x-mode.test.sh` | -| 14979 | `tests/fm-cd-pretool-check.test.sh` | -| 9339 | `tests/fm-backend-herdr.test.sh` | -| 6885 | `tests/fm-herdr-lab.test.sh` | -| 5127 | `tests/fm-crew-state.test.sh` | -| 4044 | `tests/fm-pr-merge.test.sh` | -| 3922 | `tests/fm-grok-harness.test.sh` | -| 2492 | `tests/fm-test-run.test.sh` | -| 1901 | `tests/fm-send-popup-settle.test.sh` | -| 1234 | `tests/fm-spawn-batch.test.sh` | -| 851 | `tests/fm-send-strict.test.sh` | -| 791 | `tests/fm-review-diff.test.sh` | -| 627 | `tests/fm-tmux-submit-busy.test.sh` | -| 525 | `tests/fm-brief.test.sh` | -| 321 | `tests/fm-composer-ghost.test.sh` | -| 283 | `tests/fm-dispatch-select.test.sh` | -| 276 | `tests/fm-send-settle.test.sh` | -| 189 | `tests/fm-ensure-agents-md.test.sh` | -| 175 | `tests/fm-supervision-instructions.test.sh` | -| 138 | `tests/fm-instruction-owners.test.sh` | -| 133 | `tests/fm-lint.test.sh` | -| 108 | `tests/fm-pi-primary-types.test.sh` | -| 106 | `tests/fm-nm-test-contract.test.sh` | -| 67 | `tests/fm-transition-lib.test.sh` | -| 64 | `tests/fm-captain-translation-contract.test.sh` | -| 48 | `tests/fm-composer-lib.test.sh` | -| 36 | `tests/fm-stow-contract.test.sh` | -| 28 | `tests/fm-no-mistakes-ownership.test.sh` | - -## Balancing method - -Longest-processing-time (LPT) assignment onto two workers using the Phase 1 averages above. -Do not rebalance alphabetically or by family intuition. -Shard execution order is longest-first so wall-clock tracks the balanced sum. - -| Lane | Script count | Sum of Phase 1 averages | +| 52939 | `tests/fm-x-mode.test.sh` | +| 48294 | `tests/fm-backend-herdr.test.sh` | +| 46788 | `tests/fm-arm-pretool-check.test.sh` | +| 34207 | `tests/fm-cd-pretool-check.test.sh` | +| 30771 | `tests/fm-decision-hold-lifecycle.test.sh` | +| 25365 | `tests/fm-crew-state.test.sh` | +| 15674 | `tests/fm-test-run.test.sh` | +| 15422 | `tests/fm-herdr-lab.test.sh` | +| 9065 | `tests/fm-composer-ghost.test.sh` | +| 8564 | `tests/fm-pr-merge.test.sh` | +| 6251 | `tests/fm-grok-harness.test.sh` | +| 5644 | `tests/fm-send-popup-settle.test.sh` | +| 5237 | `tests/fm-lint.test.sh` | +| 4816 | `tests/fm-tmux-submit-busy.test.sh` | +| 2945 | `tests/fm-pi-primary-types.test.sh` | +| 2911 | `tests/fm-send-settle.test.sh` | +| 2875 | `tests/fm-review-diff.test.sh` | +| 2747 | `tests/fm-send-strict.test.sh` | +| 2224 | `tests/fm-brief.test.sh` | +| 855 | `tests/fm-spawn-batch.test.sh` | +| 703 | `tests/fm-supervision-instructions.test.sh` | +| 581 | `tests/fm-ensure-agents-md.test.sh` | +| 248 | `tests/fm-transition-lib.test.sh` | +| 64 | `tests/fm-composer-lib.test.sh` | + +## Parallel lanes + +The two parallel lanes use longest-processing-time assignment from those measured durations. + +| Lane | Script count | Estimated duration | |---|---:|---:| -| `portable-parallel-1` | 15 | 64579 ms (~64.6 s) | -| `portable-parallel-2` | 15 | 64579 ms (~64.6 s) | -| imbalance | | 0 ms | +| `portable-parallel-1` | 11 | 162436 ms (~162.4 s) | +| `portable-parallel-2` | 13 | 162754 ms (~162.8 s) | +| imbalance | | 318 ms | -Exact ordered membership is the heredoc lists in `bin/fm-test-run.sh` (`list_portable_parallel_1` / `list_portable_parallel_2`). +`bin/fm-test-run.sh` contains the exact ordered memberships in `list_portable_parallel_1` and `list_portable_parallel_2`. ## Portable serial remainder -`portable-serial` is every `tests/*.test.sh` that is neither proven-isolated nor `real-herdr-gated`. -That keeps watcher, lock, AFK, real tmux, daemon, secondmate lifecycle, bootstrap, live-harness opt-in (default skip), GUI backends, and other stateful or unproven work serial. -Measured serial remainder wall (from the same Phase 1 artifacts, excluding Herdr) is about **13 minutes**. +`portable-serial` includes every `tests/*.test.sh` that is neither proven-isolated nor `real-herdr-gated`. +It keeps watcher, lock, AFK, real tmux, daemon, secondmate lifecycle, bootstrap, live-harness opt-in, GUI-backend, and other unproven work serial. ## Coverage guard -`bin/fm-test-run.sh --check-coverage` proves: - -1. The two portable parallel shards are a partition of the proven-isolated set. -2. Proven-isolated embeds match `bin/fm-test-isolation-proof.sh --list`. -3. Union of portable parallel shards + portable serial + real-Herdr family equals the complete `tests/*.test.sh` inventory. -4. Those four partitions are pairwise disjoint (no missing scripts, no duplicates). - -CI runs that guard as a required job (`test-coverage`). +`bin/fm-test-run.sh --check-coverage` verifies that both parallel lanes partition the proven-isolated set. +It also verifies that the parallel lanes, portable serial lane, and real-Herdr family are disjoint and cover every `tests/*.test.sh` script. ## Timing artifacts -Every portable shard, the portable serial lane, and the Herdr lane upload their runner-generated timing JSON even when the behavior run reports failures. -The dependent aggregate job runs after all four lanes, combines every available lane JSON through `bin/fm-test-run.sh --aggregate-json`, and uploads one summary artifact for critical-path review. -The workflow in `.github/workflows/ci.yml` owns the exact artifact names and aggregation wiring. +Portable shards, the portable serial lane, and the Herdr lane upload runner-generated timing JSON. +`bin/fm-test-run.sh --aggregate-json` creates the combined summary artifact. +`.github/workflows/ci.yml` owns the exact artifact names and aggregation wiring. ## Local entry points @@ -93,15 +72,8 @@ The workflow in `.github/workflows/ci.yml` owns the exact artifact names and agg | Job | timeout-minutes | Rationale | |---|---:|---| -| portable parallel 1/2 | 10 | Measured shard sum ~1 min; hang tripwire with margin | -| portable serial | 20 | Measured ~13 min remainder; reduced from interim 25m full-portable slack after sharding | -| Herdr | 40 | Unchanged hang tripwire for the real-Herdr lane | - -Timeouts remain hang tripwires, not expected healthy ends of green suites. -Do not raise them as a substitute for green results, retries, or weaker assertions. - -## What this phase does not do +| portable parallel 1/2 | 10 | The measured shard sums are about three minutes and the timeout is a hang tripwire. | +| portable serial | 20 | The serial remainder needs a larger hang tripwire. | +| Herdr | 40 | The real-Herdr lane keeps its dedicated timeout. | -- Does not expand the proven-isolated set without a new concurrent isolation proof. -- Does not parallelize watcher, AFK, real Herdr, real tmux, or other stateful families. -- Does not start rollout verification; that waits until this PR is green and merged. +Timeouts are hang tripwires rather than expected healthy durations. diff --git a/docs/herdr-backend.md b/docs/herdr-backend.md index b10579755b9..91047bcc6f3 100644 --- a/docs/herdr-backend.md +++ b/docs/herdr-backend.md @@ -1,1040 +1,286 @@ -# Herdr runtime backend (experimental) +# Herdr runtime backend -This document records the empirical verification behind `bin/backends/herdr.sh`, the herdr session-provider adapter added in P2 of the runtime-backend abstraction. -It is the herdr equivalent of the tmux facts recorded in the `harness-adapters` skill and `docs/architecture.md`'s "Runtime session backends" section. - -Herdr is [an agent-native terminal multiplexer](https://herdr.dev) with a socket API, CLI wrappers, and native per-pane agent-state detection. -Originally verified against herdr 0.7.1, protocol 14, on macOS aarch64; the latest dated evidence below uses herdr 0.7.5, protocol 16. -Current real-herdr verification uses isolated named sessions plus the guarded `bin/fm-herdr-lab.sh` lifecycle helper, either directly or through the compatibility wrappers in `tests/herdr-test-safety.sh`. -A 2026-07-02 cleanup bug proved that `HERDR_SESSION` alone is not a safe way to target destructive session cleanup; see "Session targeting: the `--session` flag, not `HERDR_SESSION` alone" below. -All real-herdr verification in this document uses isolated sessions and guarded cleanup; the captain's default herdr session and live tmux fleet were never intended targets. +Herdr is an experimental agent-native terminal backend with native per-pane agent state and push events. +Firstmate requires Herdr protocol 14 or newer; versions 0.7.1, 0.7.3, 0.7.4, and 0.7.5 are verified, with protocol-16 features enabled only when available. +Herdr provides the terminal session while Treehouse continues to provide task worktrees. +[`configuration.md`](configuration.md#runtime-backend-configbackend--fm_backend) owns shared backend selection and metadata semantics. ## Setup -Pick herdr when you want native per-pane agent-state detection (busy/idle/blocked) instead of tmux's regex-based guessing, and you are comfortable running an experimental backend. - -Herdr is dual-licensed AGPL-3.0-or-later / commercial - see its LICENSE file (github.com/ogulcancelik/herdr) or https://herdr.dev. -Firstmate only drives the `herdr` CLI as a separate process, which carries no AGPL obligations for firstmate users. +Pick Herdr when you want native busy, idle, and blocked state and accept the experimental limits below. Prerequisites: -- `herdr` itself, protocol 14 or newer (0.7.1, 0.7.3, and 0.7.4 verified) - see [herdr.dev](https://herdr.dev) for install instructions. -- `jq`, required to parse herdr's JSON output: `brew install jq` (or your platform's package manager). -- The universal firstmate prerequisites - a verified crew harness plus the required toolchain, owned by [`docs/configuration.md`](configuration.md) ("Harness support", "Toolchain"); treehouse still provides the worktree, herdr only provides the session. - -### CI pin and required real-Herdr lane - -The required GitHub Actions Herdr Behavior job uses the suite-verified releases pinned by `bin/fm-install-herdr.sh` and `bin/fm-install-treehouse.sh`, never a floating package-manager latest. -Those installer headers own the exact versions, release assets, checksums, download bounds, and post-install gates. -The workflow owns lane composition, while `bin/fm-test-run.sh --help` owns the exact family-selection and required gate-skip mechanics that prevent a missing Herdr binary from passing silently. -Live harness credential tests stay outside that family and outside default CI. -CI cleanup stays inside the guarded, non-default Herdr lab contract and preserves the default-session tripwire; `bin/fm-herdr-ci-cleanup.sh` owns the exact snapshot and teardown rules. -The first required lane targets Linux x86_64; if a genuine unsupported platform invariant appears (focus, cleanup, or default-session tripwire), keep the failure evidence and move the job to macOS rather than skip or weaken the assertion. - -Select herdr by putting `herdr` in a local `config/backend` file - the durable way to pick it - or by exporting `FM_BACKEND=herdr` when you launch your harness for a one-off session; telling the first mate in chat to use herdr also works. -It can also be auto-detected: when firstmate itself is running natively inside herdr (`HERDR_ENV=1`) and no explicit backend is set, firstmate auto-selects herdr and prints a one-time opt-out notice; running inside tmux nested in herdr always resolves to tmux instead. -A herdr spawn refuses loudly before creating a session container or acquiring a ship/scout worktree if `herdr` or `jq` is missing or the installed herdr's protocol is older than verified. -For `--secondmate` launches, secondmate home sync and inherited local-material propagation happen before this spawn-time backend gate. - -No first-run provisioning is needed beyond having `herdr` and `jq` on `PATH`; firstmate creates the workspace and tab it needs on first spawn. - -Watching and attaching: by default, each firstmate home gets its own herdr workspace (the primary uses `firstmate`; each secondmate uses `2ndmate-<secondmate-id>`), with one tab per task inside it, named `fm-<id>`. -With the optional projection disabled, attach to the selected `HERDR_SESSION` and switch to the workspace for the home you want to watch to see every one of that home's tasks as tabs in one tab bar. -You do not need to attach for routine supervision: from an active firstmate session, `bin/fm-peek.sh fm-<id>` reads a task's pane without attaching, and `FM_HOME=<this-firstmate-home> bin/fm-send.sh fm-<id> "<text>"` steers it unless `FM_HOME` is already set to the active firstmate home. - -An optional local `config/herdr-presentation-spaces` presence flag gives a clean new task a disposable one-task workspace instead. -The flag is absent by default, is inherited into secondmate homes through the primary-authoritative inheritable-config owner, and the feature is presentation-only and best-effort rather than durable grouping. -Every newly projected child created by a primary or secondmate home is inserted as a top-level space immediately after its owning parent (`firstmate` or `2ndmate-<id>`) contiguous child block when Herdr protocol 16 `workspace.move` and `python3` are available. -Unavailable or failed ordering warns and leaves the successfully created worker running in Herdr's current order. -See "Optional disposable single-task presentation spaces" below before enabling it. - -Verify it works by spawning a trivial task with `--backend herdr` and confirming the task's meta records `backend=herdr` plus `herdr_session=`, `herdr_workspace_id=`, `herdr_tab_id=`, and `herdr_pane_id=`; the selected Herdr workspace should show the new `fm-<id>` tab. - -Limitations: herdr is experimental and still carries the open gaps documented below. -Resolved backend evidence, including the 2026-07-06 symlinked-project-prefix isolation fix, is kept in the same follow-up log for auditability. - -## Status: experimental - -Herdr is experimental, exactly like every non-tmux backend in this design. -Select it by putting `herdr` in a local `config/backend` file, by exporting `FM_BACKEND=herdr`, or by telling the first mate in chat to use herdr. -It can also be selected by runtime auto-detection when firstmate itself is running inside herdr and no explicit backend setting exists. -Absent those three explicit settings, firstmate falls through to runtime auto-detection. -When nothing is explicitly configured, `bin/fm-backend.sh`'s `fm_backend_detect` checks the runtime firstmate itself is executing inside: `$TMUX` (set inside every tmux pane, including a tmux pane nested inside a herdr pane) selects tmux and wins when present, `HERDR_ENV=1` (injected into every process herdr manages a pane for) selects herdr when `$TMUX` is absent, and cmux runtime signals select cmux only after those multiplexer markers are absent. -See [`docs/cmux-backend.md`](cmux-backend.md#runtime-auto-detection) for cmux's primary `CMUX_WORKSPACE_ID` marker and macOS-only fallback signals. -An auto-detected herdr spawn prints one loud stderr notice (set `config/backend` or pass `--backend tmux` to opt out). -Auto-detecting tmux stays silent, since that reproduces today's unconfigured default byte-for-byte. -Only when none of that resolves anything does firstmate fall back to the hard default, tmux. -Absent `backend=` in a task's meta always means `tmux`; a herdr task carries an explicit `backend=herdr` line, while other experimental adapters carry their own backend values. -A herdr spawn refuses loudly if `herdr` or `jq` is missing, or if the installed herdr's protocol is older than the verified minimum (`fm_backend_herdr_version_check`). - -## Worktree provider stays treehouse - -Herdr is a session provider only. -Treehouse remains the worktree provider, exactly as it is for tmux. -Herdr's own `worktree.*` operations (branch-based, pooling/lease-free) are never used by this adapter. - -## Default task container shape: tab-per-task in one workspace PER FIRSTMATE HOME - -Firstmate creates one herdr workspace PER FIRSTMATE HOME - the primary gets `firstmate`, each secondmate gets its own `2ndmate-<secondmate-id>` - and one TAB per task inside that home's own workspace. -This is the same "one container, one endpoint per task" shape tmux uses (one session, one window per task), refined one level: the container is now scoped per home, not shared machine-wide. - -This refines, but does not reverse, P2's original authoritative container decision (AGENTS.md task herdr-sm-spaces-k4). -P2 established workspace-per-TASK vs. tab-per-task-in-one-shared-workspace and picked tab-per-task for the durable default. -The optional disposable projection described below does not change that ownership model because Firstmate never discovers or adopts a projection by label, token, or workspace shape, and it never directly closes a workspace. -What changed is the container's OWNER: P2 assumed a single firstmate instance per herdr session, so one shared `firstmate` workspace was enough. -With secondmates now spawning their own herdr tasks, jamming every home's tabs into that one shared workspace made a captain's tab bar an unlabeled mix of primary and secondmate work with no visual way to tell them apart. -Workspace-per-HOME fixes that while keeping tab-per-task's original human-watching win intact **within** each home: attaching to a home's own workspace (`herdr`, then switching to its space) still shows every one of *that home's* tasks as a tab in one tab bar, switchable with `ctrl+b <n>`; the ADDITIONAL win is that a captain juggling several homes on one herdr session now sees them as clearly labeled, separate spaces in herdr's spaces sidebar instead of one undifferentiated pile. - -### Label derivation (stable, derived from the home itself) - -`fm_backend_herdr_workspace_label` (`bin/backends/herdr.sh`) resolves the label from `$FM_HOME`, read fresh on every call rather than cached or threaded through env plumbing: - -- The PRIMARY home (no `.fm-secondmate-home` marker at its root) resolves to the constant `firstmate` - byte-identical to every pre-P3 task's recorded label. -- A SECONDMATE home (carrying `.fm-secondmate-home`, written by `bin/fm-home-seed.sh` at seed time and containing exactly that secondmate's id) resolves to `2ndmate-<secondmate-id>`, e.g. `2ndmate-sshhip-h7`. - -Because the label is derived from the home's own durable identity - the marker file lives at the home's root, not in an environment variable passed down a call chain - it is automatically stable across every respawn, recovery, and firstmate restart for the life of that home, with no extra bookkeeping required. -Two different secondmate homes always get two different, non-colliding labels because their marker ids are unique (verified: `tests/fm-backend-herdr.test.sh`'s `test_workspace_label_different_secondmates_get_different_labels`). - -Every workspace-scoped adapter path reads this SAME resolution: find/ensure (`fm_backend_herdr_workspace_find`/`_ensure`), tab create and its duplicate-label check (`fm_backend_herdr_create_task`), list-live recovery (`fm_backend_herdr_list_live`), and pane-for-tab (`fm_backend_herdr_pane_for_tab`, via the workspace id these resolve). -So a secondmate's own recovery/duplicate-check calls are automatically scoped to its own space and never see (or collide with) the primary's or a sibling secondmate's tabs. - -### The one wrinkle: a `--secondmate` spawn is launched BY the primary - -For every other spawn kind, `$FM_HOME` at spawn time already names the right home: the primary spawning its own crewmate/scout, or a secondmate spawning a crewmate/scout FROM ITS OWN `fm-spawn.sh` process (its own `$FM_HOME` already IS that secondmate's home). -The one exception is `bin/fm-spawn.sh <id> <secondmate-home> --secondmate`: this command runs IN THE PRIMARY's own process, so the primary's OWN `$FM_HOME` is what the label-resolution helpers would see by default, even though the tab being created belongs to the SECONDMATE. -`fm-spawn.sh`'s herdr case arm handles this with a narrow, targeted shadow: it computes `HERDR_LABEL_HOME` (the secondmate's own home, `PROJ_ABS`, for `KIND = secondmate`; the process's own `$FM_HOME` otherwise) and passes it as a bash temporary-assignment prefix - `FM_HOME="$HERDR_LABEL_HOME" fm_backend_herdr_container_ensure ...` and `FM_HOME="$HERDR_LABEL_HOME" fm_backend_herdr_create_task ...` - which scopes the override to exactly those two calls and is automatically restored afterward (verified: bash's temporary-assignment-before-a-simple-command form applies for the duration of a shell FUNCTION call too, not only external commands). -Nothing else in `fm-spawn.sh` reads `$FM_HOME` again after this point, so no explicit restore is needed. - -Every other backend-scoped call site needs no such glue: it already runs inside a process whose own `$FM_HOME` correctly names the home doing the work. -This includes the previously-unexercised path of a crewmate spawned FROM a secondmate's own `fm-spawn.sh` - proven end to end in `tests/fm-backend-herdr-workspace-per-home-e2e.test.sh`, not merely by code inspection (see "End-to-end verification" below). - -### Focus behavior: never steals the captain's attention - -Verified empirically against the real binary, in an isolated session: - -- `herdr workspace create` and `herdr tab create` do NOT focus by default once at least one workspace already exists in the session - matching (and no worse than) the pre-P3 adapter's already-flagless calls. -- The ONE exception: the very first workspace ever created in a brand-new, empty herdr session focuses regardless, because herdr always needs something focused to attach a client to - there is nothing to "not steal focus from" at that point. -- `--focus` reliably DOES focus (both the workspace and, for a tab, the pane within it) - confirming the flag has real effect and isn't a no-op, so its absence is meaningful. - -Both `fm_backend_herdr_workspace_ensure`'s workspace create and `fm_backend_herdr_create_task`'s tab create now pass `--no-focus` unconditionally. -This is defense in depth rather than a behavior change in the already-safe steady state: it guards workspace and tab creation after the session already has a focused workspace, but it cannot prevent herdr's unavoidable first-workspace focus in a brand-new empty session. -Once a workspace exists, spawning - primary or secondmate, workspace or tab - should not switch whatever space the captain is actively watching. - -### Label collisions: adopt-don't-duplicate, unchanged in spirit - -Herdr enforces NO label uniqueness at all for either workspaces or tabs (re-verified for workspaces specifically in this pass: creating a second workspace with an already-used label succeeds and produces two workspaces sharing that label). -`fm_backend_herdr_workspace_find` therefore adopts the FIRST matching workspace `jq` returns for a home's own label - in practice list order, normally creation order / the oldest - rather than attempting to disambiguate; this mirrors the pre-existing tab duplicate-label check in `fm_backend_herdr_create_task` (which still refuses an exact duplicate TAB label within the adopted workspace). -Practical consequence: if a user manually creates their own herdr workspace that happens to share a firstmate home's label (`firstmate`, or `2ndmate-<some-id>`), firstmate's next spawn silently ADOPTS that pre-existing workspace as if it were its own, rather than creating a second one or refusing. -This is a pre-existing characteristic of the adapter's find-before-create pattern, not a new risk introduced by the per-home refinement; avoid naming a personal herdr workspace `firstmate` or `2ndmate-<secondmate-id>` if you want to keep it separate from firstmate's own space. - -### No forced migration - -Existing live tasks are unaffected by this change: a task's meta already records its own `window=`/`herdr_pane_id=` target, which every backend-scoped operation (send/capture/kill/busy-state) resolves directly and never re-derives from a workspace label. -So a task spawned before this pass keeps working exactly as before, from whatever workspace it already lives in (the old shared `firstmate` workspace, or a pre-rename `firstmate-<secondmate-id>` workspace if that is where its home's tasks previously landed). -New workspace lookup does not adopt old secondmate labels: for new spawns, recovery, and list-live, the adapter exact-matches the current label derived from `FM_HOME` (`2ndmate-<secondmate-id>`). -If an older live workspace is still labeled `firstmate-<secondmate-id>`, rename it with `herdr workspace rename <workspace_id> 2ndmate-<secondmate-id>` before expecting new tasks or recovery/list-live to use that workspace. - -Tab-per-task within each home's own workspace remains the durable default for the reason P2 originally found: attaching once shows every one of that home's tasks as a tab in one tab bar, switchable with `ctrl+b <n>`, matching how a captain already watches a tmux-backed fleet. -Durable workspace-per-task remains rejected. -The optional projection accepts a top-level space per clean new task as a disposable visual aid, with exact same-identity restart replacement and explicit flat fallback for every ambiguous case. - -## Default workspace lifecycle: one per-home workspace, reused - -Each home's own workspace (`firstmate` for the primary, `2ndmate-<secondmate-id>` for a secondmate - see "Label derivation" above) is created as needed and reused by each subsequent default-container spawn while it exists: `fm_backend_herdr_workspace_ensure` calls `fm_backend_herdr_workspace_find` first and creates a workspace only when none labelled for that home exists yet. -Teardown (`fm_backend_herdr_kill`) closes only the task's pane/tab, never the workspace. - -## Optional disposable single-task presentation spaces - -Create the local, gitignored `config/herdr-presentation-spaces` file on the primary home to enable the presentation projection. -The primary's literal presence or absence converges to registered secondmate homes through the same launch, bootstrap, and config-push inheritance owner as the other declared inheritable config items. -An absent file is off, and the off path runs the existing home-workspace and `fm-<id>`-tab command sequence unchanged. -A home that has not yet converged stays flat rather than gaining partial projection authority. -This is a visual convenience, not a task container authority, lifecycle foundation, or durable grouping guarantee. -The `kind=secondmate` agent itself always uses its ordinary `2ndmate-<id>` parent workspace and never receives a corner projection; only eligible crewmates and scouts launched by that home project beneath it. - -Only a Herdr task with neither `state/<id>.meta` nor `state/<id>.herdr-presentation` is eligible for a projected create. -Firstmate generates 128 random bits, encodes them as a 22-character base64url `projection_id`, and atomically publishes `state/<id>.herdr-presentation` before asking Herdr to create anything. -The initial three-line version 1 journal contains only `version=1`, `task_id=<id>`, and `projection_id=<token>`. -After the exact new workspace is nested under one unambiguous parent and converges to one exact task tab and pane, Firstmate atomically upgrades the journal to version 2. -Version 2 has exactly 12 fields: the version, task id, token, physical home, named session, workspace id, tab id, pane id, parent workspace id, parent label, workspace label, and task label. -The journal never selects or authorizes send, capture, Treehouse return, or general task-ownership decisions. - -The new workspace is created with the normal project cwd, `--no-focus`, and a visible label such as `└ release-notes · p:AbCdEfGhIjKlMnOpQrStUv`. -Every newly created child uses the literal U+2514 `└`, one space, the concise task label with redundant `firstmate/`, `2ndmate-<id>/`, and presentation-level `fm-` owner prefixes removed, then the unchanged ` · p:<full-22-character-token>` suffix. -The ordinary task tab remains `fm-<id>` and is unchanged. -The full token is intentionally visible because Herdr has no verified persistent hidden field suitable for this non-adversarial correlator. -The create response's exact workspace, seeded tab, and root pane IDs stay in the spawning process while it verifies the projection. -Only the verified workspace, task tab, task pane, and exact parent identities are persisted in the version 2 restart binding. -The normal `fm-<id>` tab is created in that exact workspace, and only the exact seeded tab from the same workspace-create response is eligible for pruning. -The projected create refuses success unless the workspace converges to exactly one tab and one pane, both matching the new task response. -There is no log or placeholder tab because retaining one would keep the workspace alive after the task pane closes. -Immediately before and after projected workspace create, task-tab create, seeded-tab prune, workspace move, abort cleanup, and normal cleanup, Firstmate verifies one exact active workspace id and active tab id. -The snapshot comes only from the named session's response and is cross-checked against that workspace's focused tab. -An ambiguous pre-operation snapshot refuses the focus-sensitive mutation rather than guessing from a label, order, or ambient client. - -For every eligible projected create from a primary or secondmate home, Firstmate makes one presentation-only ordering attempt after that exact workspace has converged. -One bounded lock per live named Herdr session/socket serializes projected creates, ordering, exact restart replacements, abort cleanup, and projected normal cleanup across every Firstmate home that shares the session. -The lock key is derived from the verified session name and canonical socket path and lives in a machine-private shared runtime namespace, never inside any one home's `state/`. -An unverified or ambiguous socket or an insecure shared-lock namespace fails closed for presentation mutation, warns, and leaves the task on the ordinary flat path. -The new response-derived workspace id is inserted immediately after its owning parent (`firstmate` or `2ndmate-<id>`) contiguous child block and before the next parent. -New-format `└ ... · p:<token>` children define that block; already-adjacent old-format `firstmate/... · p:<token>` or `2ndmate-<id>/... · p:<token>` projections may extend it read-only for compatibility and are never renamed or migrated. -An ambiguous, foreign, or detached presentation child makes the ordering shape unverifiable, so Firstmate warns and skips the move instead of assigning ownership by guesswork. -Only the exact workspace id returned by the current projected create is ever a move target. -After a successful move, the sequence of every pre-existing workspace id excluding the new id must be byte-identical to the pre-move sequence. -Labels and tokens remain non-authoritative correlators only; by themselves they never authorize adoption, close, delete, rename, task routing, Treehouse return, or recovery. - -Herdr 0.7.4 protocol 16 exposes `workspace.move` in `herdr api schema`, with exact parameters `workspace_id` and zero-based `insert_index`, but does not expose it as a CLI subcommand. -`bin/backends/herdr-workspace-move.py` therefore sends that one whitelisted method over the exact named session's Unix socket and accepts only its matching `workspace_list` response. -The returned order is checked against the full pre-existing workspace-id sequence and the owning-parent insertion point. -The installed move does not focus its target, but Firstmate still compares the exact pre-operation workspace and tab afterward and restores that exact tab if a future or failed move changes focus. -Focus restoration is not an ordering retry and grants no authority over the moved workspace. - -Ordering is best-effort and never becomes task or lifecycle authority. -An unavailable protocol, missing method schema, missing `python3`, ambiguous socket or workspace layout, busy shared lock, explicit move error, lost response, or failed verification prints a warning and does not fail the spawn. -Firstmate performs no ordering retry, adoption, reuse, close, delete, rename, or cleanup in response. -If a move response is lost after Herdr applied it, the current order may already have changed, but the worker remains safely running and no ambiguous response grants additional authority. - -After creation, the ordinary task metadata remains the sole operational endpoint record. -Its `window=`, `herdr_session=`, `herdr_workspace_id=`, `herdr_tab_id=`, and `herdr_pane_id=` fields have exactly the same shape as the flag-off path. -No projection ownership flag is added. -The existing `treehouse get`, cwd polling, worktree validation, harness launch, and teardown return sequence is unchanged. - -If the same spawning process fails after both creates returned complete exact IDs, its abort trap may close only the exact task and seeded panes returned by those calls. -An ambiguous create result grants no cleanup authority, so Firstmate performs no lookup, adoption, reuse, or cleanup and leaves the journal quarantined. -Normal teardown still calls only the existing exact recorded task-pane close and never calls `workspace close`. -When that pane was the workspace's last pane, Herdr removes the empty tab and workspace through its existing last-pane behavior. -Herdr 0.7.4 has a focus bug in that last-pane path: closing a non-focused projected workspace can move the session's active workspace and tab to a neighbor even though the closed workspace was not active. -The exact reproduction moved focus from `2ndmate-bravo`'s active tab to `2ndmate-alpha` at `herdr pane close <projected-task-pane>`; workspace create, task-tab create, seeded-pane prune, and `workspace.move` all preserved both ids. -Projected cleanup therefore runs under the same shared presentation lock, captures the exact active workspace and tab immediately before close, and uses one exact `tab focus <captured-tab-id>` to restore both after Herdr moves them. -If the projection pane belongs to the active tab, cleanup refuses the close because deleting that tab cannot preserve it exactly. -If the lock, snapshot, or exact pane verification is ambiguous, cleanup warns, leaves the journal quarantined, and refuses the close. -If exact-tab restoration fails after the pane close has already succeeded, cleanup warns, and the ordinary exact-pane confirmation still decides whether to retire the journal. -The journal is retired only when one exact token-bearing workspace correlates with the recorded endpoint before close and the exact pane is confirmed gone afterward. -An unconfirmed close, renamed label, duplicate token, flat fallback, or unreadable state retains the journal and attempts no workspace cleanup. - -Recovery is deliberately conservative and presentation-only. -An existing journal suppresses another projected create for that task id. -Before any recovery mutation, Firstmate holds both the task-id spawn lock and the named session's presentation lock. -The existing metadata must contain one exact Herdr session, workspace, tab, and pane endpoint, and that exact pane plus every token-matched pane must be positively dead or agent-free before flat fallback is safe. -Only an agent-free pane is eligible for in-place replacement; a missing pane falls back flat because it no longer proves the bound workspace shape. -A version 2 reclaim additionally requires the same physical home, named session, metadata endpoint, unique token match, exact workspace label, exact single tab and pane, exact unique parent workspace and label, current placement inside that parent's contiguous child block, and one unambiguous non-target focus snapshot. -The replacement tab is created first in the exact bound workspace with `--no-focus`, its response-derived tab and pane identities are verified, and the old agent-free pane is checked again immediately before an exact focus-preserving close. -After the old pane is confirmed gone and the workspace converges to exactly the replacement tab and pane, the version 2 journal advances atomically to the new endpoint before metadata publication and launch handoff. -A replacement failure rolls back only the exact response-derived new pane when focus-safe verification permits it. -The reclaim path never calls workspace move, close, delete, or rename, and it never touches a parent, sibling, captain, or foreign pane. -Version 1 journals, dead panes, duplicate tokens, renamed spaces, detached or foreign spaces, cross-home mismatches, inconsistent metadata, ambiguous workspace/tab/pane snapshots, active target tabs, and uncertain focus all refuse in-place replacement and retain the existing flat fallback when duplicate-agent risk is positively absent. -A live or unknown endpoint or token-matched pane refuses the launch entirely. -Zero token matches, including a label whose token was removed by a human rename, degrade to flat and leave every old workspace untouched. - -The user-visible compromises are intentional: - -- Grouping remains best-effort rather than guaranteed; only an exact same-identity version 2 binding survives a Herdr restart in place. -- Clean projected creates form one stable contiguous child block immediately after their owning parent (`firstmate` or `2ndmate-<id>`); existing ambiguous or manually interleaved layouts degrade with a warning instead of being rewritten. -- Existing live or ambiguous projected spaces are never force-renamed, moved, or promoted from tabs into the new topology. -- A same-identity Herdr restart retains its projected space only when every exact version 2 binding and agent-absence check agrees. -- Any missing or ambiguous binding degrades to the ordinary flat home workspace without rewriting the old space. -- Crashes, response loss, failed exact-pane close, or human renames can leave stale empty-looking spaces that Firstmate never auto-deletes. -- Spaces have no cross-home cleanup, and a secondmate child can reclaim only under its exact bound home and parent. -- Manual cleanup happens in Herdr's UI after human inspection. -- Regaining a dedicated space after ambiguous degradation requires stopping or retiring the flat task, manually verifying the stale projection is harmless, clearing its quarantined journal, and starting a genuinely fresh task. -- The visible 22-character token is only a restart-stable correlator and never substitutes for the exact binding. - -The projection, its ordering follow-up, and exact restart replacement make no Herdr provider/API change, no Treehouse lease or return change, no ownership registry, and no cross-home cleanup path. -It is intentionally separate from any future Treehouse hardening work. - -### Isolated E2E evidence (2026-07-24) - -The mandatory projection suite, including same-identity restart reclaim and multi-home secondmate-child topology, ran against Herdr 0.7.5, protocol 16, on macOS aarch64 through the guarded named-session lab contract. -The default-session fleet-state tripwire was identical before and after teardown. - -Exact command: - -```sh -HERDR_LAB_HELPER='/Users/kunchen/.treehouse/firstmate-b8697d/3/firstmate/bin/fm-herdr-lab.sh' \ - bash tests/fm-backend-herdr-presentation-e2e.test.sh -``` - -Exact result was exit 0. -The abridged output below covers the new restart cases plus the existing ambiguity and focus boundaries. - -```text -ok - real Herdr lab: every projected create, task-tab create, seeded prune, and move preserves active workspace and tab -ok - real Herdr lab: Hi Bit and Wheelhouse-style same-identity restarts reclaim one nested space with exact focus and idempotence -ok - real Herdr lab: secondmate restart binding and reclaim stay isolated to the exact child home and parent -ok - real Herdr lab: concurrent cross-home recoveries replace exact husks under one session lock with no focus drift -ok - real Herdr lab: legacy projection labels and flat secondmate tabs are left unmigrated -ok - real Herdr lab: multi-home exact-pane teardowns restore captain focus without workspace close authority -ok - real Herdr lab: missing, renamed, and duplicate tokens trigger zero destructive or adoptive calls, and live duplicate risk refuses launch -ok - real Herdr lab validation completed on Herdr 0.7.5 with the default-session tripwire intact -``` - -Reserved-keyword guard: never name a `jq --arg`/`--argjson` after a `jq` keyword (`label`, `and`, `or`, `not`, `if`, `then`, `else`, `end`, `reduce`, `foreach`, `import`, `def`, `as`, `__loc__`). -jq <= 1.6 rejects a keyword-named `$`-variable as a compile error, and this adapter pipes `jq`'s stderr to `/dev/null`, so on jq <= 1.6 the error silently becomes an empty result rather than a visible failure. -Use a distinct name such as `$want` instead; `tests/fm-backend-herdr.test.sh` greps `bin/` for this pattern so a new violation fails loudly rather than silently. +- Herdr protocol 14 or newer, installed from [herdr.dev](https://herdr.dev). +- `jq` for JSON responses. +- The universal harness and toolchain requirements in [`configuration.md`](configuration.md#toolchain). +- `python3` only for optional protocol-16 presentation-space ordering and native event subscription. -### Default-tab prune +Herdr is dual-licensed AGPL-3.0-or-later or commercial. +Firstmate invokes its CLI as a separate process. -`herdr workspace create` seeds the new workspace with one auto-created default tab (label `1`) that firstmate never uses. -`fm_backend_herdr_create_task` prunes it (best-effort, via `fm_backend_herdr_workspace_prune_seeded_default_tab`) right after creating the first real task tab in a freshly created workspace, never earlier: closing a workspace's LAST tab deletes the whole workspace on real herdr, and immediately after creation the default tab is the only one present. +Select Herdr with local `config/backend` containing `herdr`, `FM_BACKEND=herdr` for one launch, or an explicit request to Firstmate. +It is also auto-detected when the primary runs natively under `HERDR_ENV=1` and is not inside tmux. +A tmux pane nested inside Herdr resolves to tmux because the innermost multiplexer wins. +An auto-detected Herdr spawn prints an opt-out notice. -**The prune target is identified structurally (created-vs-adopted), never by label pattern.** -`fm_backend_herdr_workspace_ensure` captures the seeded default tab's `tab_id` straight from its OWN `workspace create` response (`.result.tab.tab_id`, verified empirically to be present on the same response as `.result.workspace.workspace_id` - no follow-up `tab list` call is needed) ONLY when that call itself just created the workspace. -`fm_backend_herdr_container_ensure` threads that id through to its caller as a second field: it echoes `"<session>:<workspace_id>\t<seeded_default_tab_id>"`, the second field empty whenever the workspace was ADOPTED (`fm_backend_herdr_workspace_find` matched a pre-existing workspace by label) rather than created fresh. -`fm_backend_herdr_create_task` accepts that value as an explicit 4th argument and is the ONLY place allowed to act on it; it never re-derives "prunable" from a tab's label or the workspace's tab count. -An adopted workspace's caller always passes an empty 4th argument, so create_task never even looks for a prune candidate in that case - it is structurally impossible for an adopted workspace's tabs to be pruned, regardless of how they are labeled. +Spawn stops before creating a Herdr container or acquiring a task worktree when `herdr`, `jq`, or the protocol floor is unavailable. +No separate first-run provisioning is required. -Defense in depth on top of that gate (not the primary safety mechanism): before closing the seeded tab, `fm_backend_herdr_workspace_prune_seeded_default_tab` re-verifies the tab is still present, re-checks it is still labeled `1`, and refuses if its pane's `agent get` reports `agent_status: working` (herdr's own native agent-state detection) - belt-and-suspenders against a live agent having landed there through some other path. +The required CI lane uses the pinned installers in `bin/fm-install-herdr.sh` and `bin/fm-install-treehouse.sh`. +Those script headers own release assets, checksums, download bounds, and post-install gates. +Real harness credential tests remain opt-in rather than part of default CI. -#### Incident: the 2026-07-02 self-kill +## Watching and task containers -The previous implementation derived "prunable" at `create_task` time from a pure label heuristic run against whatever workspace `workspace_find` had just resolved: exactly one tab, labeled `1`. -Herdr enforces no label uniqueness (see "Label collisions" above) and derives an unlabeled workspace's DISPLAYED label from its pane cwd's basename. -A captain who launches herdr directly inside a directory named `firstmate` therefore gets a workspace whose label is `firstmate` - byte-identical, by coincidence, to the primary firstmate home's own derived label - with a single auto-created tab, also labeled `1`. -`fm_backend_herdr_workspace_find` adopted that pre-existing, captain-owned, LIVE workspace by the label match (a label match can never distinguish an explicitly `--label`-created workspace from one whose label only coincidentally matches); the old heuristic matched too, since it looked only at the adopted workspace's own tab shape, not at whether THIS spawn had actually created it. -The very next crewmate spawn's `create_task` call closed the captain's own live pane roughly 27ms after creating its own task tab, killing the primary firstmate agent and its watcher mid-turn. -Log evidence: `~/.config/herdr/herdr-server.log` showed `cli:tab:create` (the new task tab) immediately followed by `cli:pane:close` on the captain's pane (pid 36335, launched ~8 minutes earlier); `~/.config/herdr/session.json` showed the adopted workspace's `custom_name: null` with `identity_cwd` pointing at the firstmate repo. +Each Firstmate home gets one durable workspace with one task tab per endpoint. +The primary workspace is `firstmate`. +A secondmate home uses `2ndmate-<secondmate-id>`, derived from its validated `.fm-secondmate-home` marker. +The secondmate process and every child it launches resolve the same home label; a secondmate launched by the primary receives a narrowly scoped home override during container creation. -The fix is structural, not another heuristic, and is unit- and E2E-tested: see `tests/fm-backend-herdr.test.sh`'s `test_adopted_workspace_never_prunes_default_tab` and `test_label_collision_startup_workspace_leaves_live_tab_alone`, and `tests/fm-backend-herdr-prune-safety-e2e.test.sh`'s isolated real-herdr reproduction of the exact incident shape. +Attach to the selected named Herdr session and switch to the relevant home workspace to watch its task tabs. +Routine supervision uses `bin/fm-peek.sh <id>` and `FM_HOME=<home> bin/fm-send.sh <id> '<text>'` without attaching. -Because closing a workspace's last tab deletes it, a home's workspace does not outlive a fully idle fleet (zero live tasks for that home) - the next spawn's `workspace_find` simply finds nothing and recreates it. Reuse holds across concurrent and sequential tasks; it is not a guarantee that the workspace itself survives the whole session unconditionally. +Workspace and tab creation use `--no-focus`. +The first workspace in a completely empty Herdr session must become focused because no prior target exists, but later task creation does not intentionally steal focus. -A workspace whose label this adapter did not derive (see "Label derivation" above) is never adopted, reused, or torn down by firstmate - `fm_backend_herdr_workspace_find` and `fm_backend_herdr_list_live` only ever match a home's own derived label. +Herdr does not enforce workspace or tab label uniqueness. +Firstmate adopts the first workspace matching its derived home label and refuses duplicate task tabs inside it. +Avoid naming a personal workspace `firstmate` or `2ndmate-<id>` because the adapter cannot distinguish that label collision from its own container. +An older secondmate workspace using `firstmate-<id>` is not migrated automatically; rename it manually before expecting new tasks or recovery to use it. -## Target string and meta fields +Existing task operations use recorded endpoint ids and do not move a live task when labels change. +The per-home workspace is reused while it has task tabs. +Closing its last tab can remove the workspace, and the next spawn recreates it. -A herdr task's `window=` meta field holds `<herdr-session>:<pane-id>`, for example `default:w1:p2`. -The pane id itself contains a colon, so the adapter splits on the FIRST colon only, never on every colon. -This mirrors tmux's `session:window` target shape closely enough that `fm_backend_resolve_selector` (in `bin/fm-backend.sh`) needed no backend-specific logic at all - it already just returns a task's recorded `window=` value verbatim. -Task-selector resolution is the shared contract owned by [`docs/configuration.md`](configuration.md) ("Runtime backend"). -For a bare unknown non-`fm-` name, Herdr retains the legacy tmux live-window fallback. +## Optional presentation spaces -Herdr tasks additionally record: +Create local gitignored `config/herdr-presentation-spaces` to request a disposable one-task workspace for each new crewmate or scout. +The setting is inherited into secondmate homes through the normal configuration-convergence owner. +A secondmate agent itself always stays in its ordinary parent workspace; only children launched by that home are eligible. +An absent or unconverged setting keeps the flat default. -- `herdr_session=` - the named herdr session this task's server lives in. -- `herdr_workspace_id=` - the id of the exact workspace containing this task's endpoint, ordinarily the primary's `firstmate` workspace or a secondmate's own `2ndmate-<id>` workspace, and a disposable task workspace when the optional projection succeeds; for reference only, since day-to-day operations use the recorded pane target. -- `herdr_tab_id=` - the task's tab id. -- `herdr_pane_id=` - the task's pane id, the fast-path operational target. +Presentation is a best-effort visual projection, never task ownership or lifecycle authority. +Only a fresh task with neither metadata nor an existing presentation journal is eligible for projected creation. +Firstmate atomically publishes a three-field version 1 journal containing a random 128-bit base64url token before asking Herdr to create anything. +After the new workspace converges to one exact task endpoint beneath one exact parent, the journal advances to a version 2 binding that records the physical home, named session, endpoint, parent, and immutable expected labels. +The token is visible in the workspace title because Herdr exposes no verified hidden persistent field, but neither token, title, nor journal authorizes send, capture, task ownership, Treehouse return, or general recovery. -## Verified CLI facts +The normal `fm-<id>` task tab is created in the exact new workspace returned by Herdr. +Only the exact seeded default tab returned by the same workspace-create response can be pruned. +Before and after create, prune, order, abort cleanup, and normal cleanup, Firstmate verifies exact workspace, tab, pane, and active-focus ids. +An ambiguous response grants no mutation or cleanup authority. -| Operation | Verified herdr call | What was verified | -|---|---|---| -| Version/protocol gate | `herdr status --json` -> `.client.protocol` | Session-independent; `.server.*` fields ARE session-dependent. | -| Headless server start | `HERDR_SESSION=<name> herdr server --session <name>` (backgrounded) | A bare socket call does NOT auto-start the server; the adapter always starts-then-polls before any workspace/tab/pane call. This fact is for start only, not cleanup, and the explicit `--session` flag is intentional because `HERDR_SESSION` alone is not safe session targeting. | -| Duplicate task check | `herdr tab list --workspace <id>`, match by `.label` | Herdr does NOT enforce tab-label uniqueness itself; two tabs can share a label. The adapter's own duplicate check is required. | -| Send literal (unsubmitted) | `herdr pane send-text <pane> <text>` | Does NOT auto-submit, contrary to the original design addendum's guess. Verified directly: a unique marker sent this way sits unexecuted in the composer until a separate Enter. Behaves exactly like tmux's `send-keys -l`. | -| Send + submit atomically | `herdr pane run <pane> <command>` | Runs and submits a command in one call; used for the two fixed spawn-time commands (`treehouse get`, the `GOTMPDIR` export) exactly where tmux used one `send-keys ... Enter` call. | -| Send key | `herdr pane send-keys <pane> <key>` | Verified names: `enter`, `escape` (alias `esc`), `ctrl+c` (aliases `C-c`, `c-c`). `ctrl+c` verified to interrupt a running foreground process immediately. | -| Submit confirmation (idle baseline) | `herdr agent get <pane>` -> `.result.agent.agent_status` after Enter | `fm_backend_herdr_send_text_submit` records the pre-Enter status and, when it is idle/done, confirms delivery by polling for `working`/`blocked` across the Enter attempt's confirmation budget. Composer-state reads remain the affirmative-empty pre-injection guard and the conservative fallback for preexisting submit-active or unreadable baselines; see "Native agent-state submit confirmation". | -| Bounded capture | `herdr pane read <pane> --source recent --lines N` | See "Verified bug" below - N is never passed through directly. | -| ANSI capture | `herdr pane read <pane> --source recent --lines N --format ansi` | Herdr 0.7.3 preserves composer de-emphasis styling, letting the shared `fm_composer_strip_ghost` extractor treat dim/faint and dark-TRUECOLOR ghost/placeholder text as empty while retaining real typed input. The same small-`--lines` workaround applies. | -| Busy state | `herdr agent get <pane>` -> `.result.agent.agent_status` | Verified live against an interactive `claude` session: reports `working` while generating, `done` once idle. Mapped: `working` -> busy; `idle`/`done` -> idle; `blocked` -> idle (surfaced like a stale pane, not suppressed as busy - a blocked agent is stuck waiting on the human, not grinding); anything else -> unknown (the cue for the shared tail-regex fallback). | -| Kill | `herdr pane close <pane>` | Closing a tab's only (root) pane also closes the tab - no separate tab-close call needed for this adapter's one-pane-per-tab shape. Best-effort: closing an already-closed pane exits non-zero, matching tmux's `kill-window \|\| true` contract. Teardown itself only ever closes the task's own pane/tab, never the workspace - but closing a workspace's LAST tab (verified real-herdr behavior) deletes the workspace as a side effect, so a home's own workspace persists only while at least one task tab remains; see "Default workspace lifecycle" above. | -| Default-tab prune (create_task, first task in a fresh workspace only) | `herdr workspace create`'s own response (`.result.tab.tab_id`) identifies the seeded tab; `herdr tab list` + `herdr agent get <pane>` re-verify it; `herdr pane close <pane>` closes exactly that tab id | `herdr workspace create` seeds the new workspace with one auto-created default tab (label `1`, id captured straight from the create response) firstmate never uses. `fm_backend_herdr_create_task` closes EXACTLY that captured tab id right after creating the first real task tab in a freshly created workspace - never right after `workspace create` itself (see Kill row), and never re-derived from a tab's label or the workspace's tab count at create_task time (see "Default-tab prune" above for the created-vs-adopted safety gate and the 2026-07-02 incident it fixes). Best-effort; an ADOPTED workspace (not freshly created by this same call) is never a prune candidate at all. | -| Presentation workspace ordering | Raw protocol-16 `workspace.move` with `{workspace_id, insert_index}` over the exact named session socket | Herdr 0.7.4 exposes the method and zero-based `WorkspaceMoveParams.insert_index` in `herdr api schema` but has no `herdr workspace move` CLI subcommand, while moving the exact newly created workspace returns the full `workspace_list`, preserves focus and every other workspace's relative order, and is never used for recovery, ownership, adoption, or cleanup. The surrounding projection guard captures and verifies the exact active workspace and tab anyway. | -| Presentation cleanup focus | `herdr pane close <exact-projection-pane>`, followed only when needed by `herdr tab focus <exact-prior-tab>` | Herdr 0.7.4 can move focus to a neighboring workspace when closing a non-focused workspace's last pane. Firstmate serializes projected cleanup, refuses to close the active tab, and restores only the exact response-derived pre-close tab id. No label, order, or projection token is restoration authority. | -| Presentation restart reclaim | Exact version 2 journal plus metadata and live named-session snapshots, then `herdr tab create` before `herdr pane close <exact-old-husk>` | Herdr 0.7.5 restores exact workspace/tab/pane identities but no registered agent after a session restart; Firstmate replaces only one fully bound agent-free husk under the session lock, advances the journal to the exact replacement endpoint, and falls back flat without mutation for every ambiguous binding or focus snapshot. | -| Recovery / list-live | `herdr tab list --workspace <id>`, filter labels starting with `fm-` | Label-based, never trusts a stored id blindly - see "ID stability" below. `<id>` is always THIS home's own workspace (`fm_backend_herdr_workspace_find`), so recovery never sees a sibling home's tabs. | -| Workspace create / tab create (focus) | `herdr workspace create --no-focus`, `herdr tab create --no-focus` | Verified: neither focuses by default once a workspace already exists in the session, matching pre-P3 (flagless) behavior; `--no-focus` is passed anyway for defense in depth, since the very first workspace ever created in a brand-new session focuses regardless of the flag. `--focus` was separately verified to reliably focus, confirming the flag has real effect. | -| Session targeting for DESTRUCTIVE calls | `herdr session stop <name> --session <name> --json`, then `herdr session delete <name> --session <name> --json`; never `herdr server stop` | Owned by `bin/fm-herdr-lab.sh` (which `tests/herdr-test-safety.sh` sources), re-querying `herdr session list --json` before every destructive call. See "Session targeting" below - `HERDR_SESSION` alone is not reliably honored once another herdr server is already running on the machine. | +Protocol 16 exposes `workspace.move` over the named session socket but no CLI subcommand. +`bin/backends/herdr-workspace-move.py` sends only that whitelisted method and verifies the complete returned workspace order. +Projected children are placed in one contiguous block immediately after their owning home when the session layout, protocol, socket, `python3`, and machine-private per-session lock are all verifiable. +Existing legacy child labels may extend an already adjacent block read-only but are never renamed or migrated. +A foreign, ambiguous, detached, or manually interleaved child makes ordering skip with a warning rather than rewriting the layout. -## Incident (2026-07-13): the ASCII request separator erased the secondmate marker +Ordering failure never fails the task spawn. +Firstmate does not retry, adopt, reuse, close, delete, or rename anything in response to an unavailable method, lock contention, ambiguous socket, lost response, failed move, or verification mismatch. +The worker remains on the ordinary flat or Herdr-current-order path. -A routed request reached a Pi/Herdr secondmate without the visible `[fm-from-firstmate]` label, so the secondmate correctly treated it as direct captain conversation and returned nothing to the parent status path. -The initial suspicion was selector classification, but a real isolated reproduction disproved that: exact-id lookup found the right metadata, read `kind=secondmate`, selected the recorded Herdr endpoint, and still delivered an unmarked Pi prompt. +Normal task metadata remains the sole endpoint authority after creation. +Cleanup closes only the exact recorded task pane and never calls `workspace close`. +Herdr can move focus when closing the last pane of a non-focused projected workspace, so projected cleanup runs under the same session lock, captures the exact active tab, refuses to delete the active tab, closes the exact task pane, and restores only the exact prior tab when needed. +If lock, snapshot, pane identity, or restoration is ambiguous, cleanup warns and preserves the journal for manual inspection. -The reproduction used Herdr 0.7.3 (protocol 16), Pi 0.80.6, a task-local sender home, a real `fm-spawn.sh --secondmate --harness pi --backend herdr` endpoint, and a generated non-`default` session from `bin/fm-herdr-lab.sh`. -Every adapter call was routed through the lab helper, and teardown verified the default-session fleet-state tripwire. -The end-user command was run with normal `FM_SEND_SETTLE`: - -```sh -FM_HOME=<isolated-sender-home> bin/fm-send.sh marker-pi-sm \ - 'FM_MARKER_E2E_CURRENT exact-id request' -``` - -Immediately before submission, the authoritative selector helpers reported: +Recovery is deliberately conservative and presentation-only. +An existing journal suppresses another projected create. +Before any recovery mutation, Firstmate holds both the task spawn lock and the named-session presentation lock. +A same-identity version 2 binding may replace one exact agent-free restart husk in place only when the physical home, session, metadata endpoint, unique token match, workspace shape and labels, parent identity and placement, and non-target focus snapshot all agree. +The replacement tab and pane are created and verified before the old pane is rechecked and closed, then the journal advances atomically to the replacement endpoint before metadata publication. +The reclaim path never moves, closes, deletes, or renames a workspace and never touches a parent, sibling, captain, or foreign pane. +A failed replacement rolls back only the exact response-derived new pane when focus-safe verification permits it. +Version 1 journals, dead or missing panes, duplicate or absent tokens, renamed or detached spaces, cross-home mismatches, inconsistent endpoint bindings, active target tabs, and ambiguous identity or focus fall back flat without mutating the old projection when duplicate-agent risk is positively absent. +A live or unknown recorded or token-matched endpoint refuses duplicate launch. + +Locked session start has one narrower cleanup for a restored projected child that is no longer current task state. +It runs only when the current home has at least one ordinary presentation journal and considers only that home; a primary never recursively sweeps a secondmate home. +Discovery starts from the exact current `└ <concise-task> · p:<22-character-token>` grammar, but a title or token alone is never mutation authority. +The title must contain exactly one token occurrence across the named-session snapshot and must equal the title derived from exactly one valid presentation journal in this home's own `state/`; a version 2 journal additionally must bind this exact physical home, named session, workspace, tab, and pane. +The task's ordinary metadata must be absent, and the candidate must have exactly one tab and exactly one pane. +Before cleanup, Firstmate acquires the existing task-id spawn lock and then the shared named-session presentation lock. +Inside both locks it takes one exact snapshot, requires one unambiguous non-target focus and the exact title, token, tab, and pane shape, positively confirms no registered agent, and reads Herdr's process information for the exact named-session pane. +The process proof requires one recognized idle shell as both the shell process and the sole foreground process-group member, an operating-system process-table row for that shell, no child process, and a sleeping or idle shell state. +Any foreground command, child process, active shell job, unknown shell, unreadable process table, missing field, or API error preserves the pane. +Firstmate immediately revalidates the same journal, metadata absence, workspace title and token uniqueness, one-tab and one-pane topology, exact pane relationship, absent agent, process proof, and non-target focus before calling the existing exact-pane focus-preserving close helper. +It closes only that pane, never a workspace. +The matching journal is retired only after the exact pane is positively confirmed gone; an unconfirmed close retains the journal, while a confirmed close may retire it even when focus restoration reported an error after the close. +A second run finds no matching title or journal and is a no-op. +A malformed or missing title or token, duplicate token, zero or multiple journal matches, cross-home version 2 binding, current metadata, registered or unknown agent, extra tab or pane, active target, busy lock, changed revalidation, unreadable check, or any error preserves the candidate and lets session startup continue with at most a concise warning. + +Operational compromises: + +- Grouping is best-effort; only an exact same-identity version 2 binding survives a Herdr restart in place. +- Existing layouts are not force-renamed or rearranged. +- Missing or ambiguous restart bindings fall back to the ordinary home workspace while the old projection remains untouched. +- Crashes, lost responses, failed exact-pane cleanup, or human renames can leave quarantined spaces; session start removes only the exact home-local, uniquely journal-correlated, childless idle-shell shape above. +- Spaces have no cross-home cleanup path, and a secondmate child can clean up only from its exact home. +- Every stale-looking space outside that narrow startup proof still requires manual cleanup in Herdr's UI after human inspection. +- Regaining a dedicated space after degradation requires stopping the flat task, manually checking the stale projection, and clearing its journal before a genuinely fresh launch. +- The visible token is only a restart-stable correlator and never substitutes for the exact binding. + +`tests/fm-backend-herdr-presentation-e2e.test.sh` covers multi-home ordering, concurrency, lock contention, legacy coexistence, focus preservation, exact same-identity restart replacement, ambiguous bindings and tokens, and exact-pane cleanup through the guarded lab path. +`tests/fm-herdr-session-cleanup.test.sh` covers every discovery, ownership, topology, process, locking, revalidation, focus, retirement, and continue-on-error boundary. +`tests/fm-herdr-session-cleanup-e2e.test.sh` covers the restored-shell cleanup in a guarded non-default named lab; [`verification/runtime-backends.md`](verification/runtime-backends.md#per-home-and-presentation-topology) owns the active versioned evidence. + +## Default-tab prune safety + +`herdr workspace create` seeds one default tab. +Firstmate prunes it only after a real task tab exists and only when the same create response supplied the seeded tab id. +An adopted workspace never supplies that id and can never enter the prune path, regardless of labels or tab count. +Immediately before close, Firstmate rechecks the exact tab, expected seed label, and native agent state. +A working seed pane is never closed. + +This created-versus-adopted gate is a destructive safety boundary. +A prior label heuristic could adopt a captain-owned workspace named `firstmate` and close its live seed-shaped tab. +The current structural gate removes label inference from cleanup authority. +`tests/fm-backend-herdr-prune-safety-e2e.test.sh` reproduces the collision in an isolated named session and proves the adopted pane remains untouched. + +## Endpoint metadata ```text -resolved-meta=<isolated-sender-home>/state/marker-pi-sm.meta -kind=secondmate -target=<generated-lab-session>:w1:p2 backend=herdr -expected-label=fm-marker-pi-sm -``` - -At the time, Pi's separator-only idle composer was outside the Herdr structural classifier's recognized bordered/bare shapes, so composer state was conservatively `unknown` both before and after the send. -The endpoint's native agent state was idle before submission, and the normal idle-to-working confirmation made `fm-send.sh` return successfully. -A task-local Pi `before_agent_start` hook then captured the exact received prompt and UTF-8 bytes: - -```json -{"prompt":"FM_MARKER_E2E_CURRENT exact-id request","hex":"464d5f4d41524b45525f4532455f43555252454e542065786163742d69642072657175657374"} +window=<session>:<pane-id> +herdr_session=<session> +herdr_workspace_id=<workspace-id> +herdr_tab_id=<tab-id> +herdr_pane_id=<pane-id> ``` -The old marker should instead have started with label bytes `5b666d2d66726f6d2d66697273746d6174655d`, followed by ASCII `1f` and then those request bytes. -The Pi transcript independently rendered only `FM_MARKER_E2E_CURRENT exact-id request`, and the agent answered it conversationally as captain input. - -The failure was in marker transport, not backend selection or metadata classification. -`fm-send.sh` correctly passed `[fm-from-firstmate]`, ASCII unit separator `0x1f`, and the request to `herdr pane send-text`. -Herdr's terminal input path treated the C0 byte as a control action rather than text, removing the preceding label before Pi submitted the remaining request. -A tmux-stub unit test could not expose this because it logged the string argument without driving a real terminal editor. - -The single marker owner, `bin/fm-marker-lib.sh`, now uses U+2063 INVISIBLE SEPARATOR (UTF-8 `e2 81 a3`) after the visible label. -U+2063 has no normal keyboard keystroke but travels through terminal input as text rather than a C0 control byte. -The same owner now provides the idempotent marker transformation, so an already-marked request is not prefixed twice. -No Herdr-specific injection or classification branch was added. - -The opt-in regression command is: - -```sh -FM_SEND_MARKER_HERDR_E2E=1 tests/fm-send-secondmate-marker-herdr-e2e.test.sh -``` - -The real post-fix Pi capture reported exactly one marker followed by the request: - -```text -evidence: exact-id received-hex=5b666d2d66726f6d2d66697273746d6174655de281a3464d5f4d41524b45525f48455244525f4532452065786163742d69642072657175657374 -``` - -The same run injected direct terminal text without `fm-send.sh` and captured it byte-exact with no marker: - -```text -evidence: direct-input received-hex=464d5f4d41524b45525f48455244525f444952454354206361707461696e20696e707574 -``` - -Unit coverage in `tests/fm-send-secondmate-marker.test.sh` pins exact-id and stable-label secondmates, exact-id and stable-label ordinary crewmates, explicit endpoints with and without local metadata, key-only sends, direct unmarked input, exact U+2063 bytes, and idempotence. -Strict unresolved-selector behavior remains covered by `tests/fm-send-strict.test.sh`. - -## Incident (2026-07-14): Pi-on-Herdr away escalation stayed non-injectable for 4555 seconds - -A guarded reproduction used Herdr 0.7.3, Pi 0.80.7, a generated non-`default` session from `bin/fm-herdr-lab.sh`, an isolated Firstmate home, a real Pi primary pane, the public `bin/fm-afk-launch.sh start` entrypoint, a live synthetic child pane, the real daemon and wake queue, `bin/fm-afk-launch.sh stop`, `bin/fm-wake-drain.sh`, and `bin/fm-bearings-snapshot.sh`. -Every production-adapter and explicit Herdr command was routed through the lab helper, and teardown verified that the running `default` session was unchanged. -The synthetic child appended `blocked [key=synthetic-dependency]: firstmate can refresh the synthetic token` while away mode was active. -The daemon classified the status as captain-relevant, retained it in `state/.subsuper-escalations`, and the watcher retained the matching `signal` in `state/.wake-queue`. -The oldest-escalation sidecar was backdated by 4555 seconds to reproduce the observed interval without waiting in wall-clock time. -The alarm then recorded `fm away-mode inject WEDGED: 4556s undelivered`, the active notifier fired exactly once, the buffer remained intact, and Pi captured zero injected prompts. - -The causal probes against the exact recorded supervisor target were: - -```text -target_exists=yes -busy_state=idle -composer_state=unknown -``` - -The plain Pi capture showed a blank content row between two horizontal separators: - -```text -───────────────────────────────────────────────────── - -───────────────────────────────────────────────────── -<project and model footer> -``` - -The ANSI capture showed the same two blue separator rows with a reverse-video cursor in the blank content row. -Real pending text occupied that same middle row and native `agent get` remained idle, which proved native agent state alone could not distinguish an empty Pi composer from an unsubmitted Pi draft. -The exact target was correct, the agent was not busy, and submit verification was never reached because the affirmative-empty pre-injection guard rejected the unrecognized structure. -This rules out target resolution, busy detection, submit acknowledgement, and shutdown ordering as the 4555-second cause. -The root cause was solely that the Herdr structural classifier recognized bordered composers and bare `❯` or `›` prompt rows, while Pi renders a separator-only composer. - -`fm_backend_herdr_composer_state` remains the single backend owner of structural row recognition. -It now accepts content between the bottom-most complete pair of Pi separator rows only when Herdr's native identity says the target agent is exactly `pi` and its status is `idle`, `done`, or `blocked`. -A working Pi, a pending middle row, a missing or non-Pi identity, an incomplete pair, or an over-tall candidate remains `pending` or `unknown`, so dead shells and ambiguous panes are still non-injectable. -The extracted content still routes through the shared `bin/fm-composer-lib.sh` decision owner. - -Making the Pi composer injectable exposed the already-proven terminal-control hazard from the 2026-07-13 incident: Herdr consumed a leading ASCII `0x1f`, so Pi received `Supervisor escalate...` without the away marker. -At the time of this reproduction, `FM_INJECT_MARK` used a bare leading U+2063 INVISIBLE SEPARATOR, while the from-firstmate marker remained a visible label followed by U+2063, so their full prefixes stayed distinct. -`bin/fm-operational-input.sh` owns current operational-input construction, while the `/afk` skill owns the daemon's stay-away handling and legacy bare-marker compatibility. -U+2063 has no normal keyboard keystroke and the real post-fix Pi prompt capture retained its `e281a3` prefix byte-exact. - -The return half of the same reproduction showed that separate `stop` and `wake-drain` calls left policy ownership to the operator and allowed an ordinary Bearings request to begin while the live blocker remained open. -Bearings' authoritative structured projection was already correct: - -```json -{"in_flight":[{"id":"synthetic-child","state":"blocked"}],"decisions_open":[{"id":"synthetic-child","verb":"blocked"}],"gates":[]} -``` - -The live blocker was never structured as queued work, so no return policy was duplicated into Bearings wording or its projection. -`bin/fm-afk-return.sh` now owns deterministic stop, drain, durable evidence, and the fail-closed return gate. -It refuses ordinary work until each live open `blocked:` key is resolved after immediate remediation or explicitly reclassified with a durable reason. -The `/afk` skill owns the situation-specific procedure, and the always-loaded `AGENTS.md` away stub contains only the safety-critical trigger to run that owner before processing the return message. -`bin/fm-bearings-snapshot.sh` consults the owner's read-only guard and contains no copy of the policy. - -The guarded post-fix command was: - -```sh -FM_AFK_PI_HERDR_E2E=1 HERDR_LAB_HELPER=bin/fm-herdr-lab.sh tests/fm-afk-pi-herdr-return-e2e.test.sh -``` - -The real result was: - -```text -ok - real Pi/Herdr pending composer refuses injection without forced submit and raises one observable fallback -ok - real idle Pi/Herdr accepts one marked escalation promptly, verifies submit, clears wedge state, and emits no duplicate alert -ok - real unmarked Pi return opens catch-up and blocks Bearings before the unresolved blocker can be deferred -ok - resolved return catch-up allows Bearings and a clean idempotent away re-entry -evidence: herdr=herdr 0.7.3 pi=0.80.7 inject-hex-prefix=e281a3 notifier-count=1 -``` - -Unit coverage in `tests/fm-backend-herdr.test.sh` pins the exact idle and pending Pi captures plus working, non-Pi, unreadable, and over-tall refusal. -`tests/fm-afk-return.test.sh` pins durable catch-up evidence, blocker ownership, Bearings precedence, explicit reclassification, re-entry, and tmux/Herdr parity. -`tests/fm-bearings-snapshot.test.sh` pins that live blocked work remains live structured state and never becomes a queued gate. -The existing tmux injection E2E remains the transport-parity proof for type-once, verified-submit behavior. -The wedge alarm remains defense in depth and is not the primary delivery path. - -## Verified bug: `pane read --lines N` returns empty for small N - -This was the most significant finding of this verification pass. - -`herdr pane read <pane> --source recent --lines N` returns **completely empty output** when `N` is smaller than the pane's current viewport height, instead of clamping to the last `N` lines. -Reproduced deterministically by binary search against a 23-row pane: `--lines 5/6/8/15` all returned zero bytes; `--lines 20` returned a partial read; `--lines 24` and above returned the full expected content, correctly clamping down even at `--lines 1000`. - -This silently broke exactly the small bounded reads the adapter needs most - the composer-state guard/fallback reads around submit and injection, and would have affected any small `fm-peek.sh` line count too. -Before the workaround, an early version of the real-herdr smoke test flaked intermittently for exactly this reason. - -**Workaround:** `fm_backend_herdr_capture` never passes a caller's small requested line count straight through to herdr's own `--lines` flag. -It always requests a generous floor (>= 200 lines, comfortably above any realistic pane viewport) from herdr, then trims to the caller's actual requested bound locally with `tail -n N`. -Verified this eliminates the flake across repeated full smoke-test runs. - -## Verified gap: `agent.get` reads idle during a long foreground tool call - -`herdr agent get <pane>` -> `.result.agent.agent_status` was verified against a short interactive `claude` exchange (see "Busy state" above): `working` while the model streams a turn, `done` once it stops. -That verification did not cover a crew blocked on its OWN long-running foreground tool call - e.g. `no-mistakes axi run` without `--yes`, which blocks synchronously for the whole pipeline (minutes to tens of minutes) until a gate or outcome, per `AGENTS.md` section 7. -For that entire span the model is not generating - it already finished the turn that invoked the tool and is waiting on the tool's result - so `agent_status` reads `idle` (or `blocked`, which the adapter also maps to `idle`), even though the pane's own rendered text keeps showing the harness's busy banner (`BUSY_REGEX`, e.g. `esc to interrupt`) the whole time, exactly as it would in a plain tmux pane. - -This surfaced as a real fleet incident (2026-07-02): `bin/fm-watch.sh`'s absorb-only-when-provably-working stale path (`AGENTS.md` section 8) treated a herdr `idle` verdict from `crew_pane_is_busy` as final, so it skipped the shared tail-regex corroboration that `unknown` already got. -At the same time, an independent no-mistakes run-step attribution fallback could miss this crew's run when `axi status` reported another branch; current `bin/fm-crew-state.sh` falls back to top-level `no-mistakes runs --limit ${FM_CREW_STATE_RUNS_LIMIT:-200}` and applies its authoritative current-code matching rules before accepting a coarse verdict. -Together, those gaps let a genuinely still-working herdr crew read as not provably working, triggering an immediate stale wake instead of the intended absorb-then-escalate behavior. - -**Fix:** `bin/fm-crew-state.sh`'s `crew_pane_is_busy` now corroborates BOTH `idle` and unknown/unparseable native verdicts with the shared tail-regex before concluding "not busy" - only a bare `busy` verdict is trusted outright. -The cross-branch attribution fallback now uses the real `no-mistakes runs` command with current-code matching, and the watcher checks provably-working evidence before a stale status-log verb can make a stale pane terminal. -This does not mask a genuinely human-blocked agent (a permission dialog, not mid-tool-call): that pane does not render the busy banner, so the corroboration still correctly reports not-busy for it. - -## Slash/`$` autocomplete popup hazard (confirmed, same mitigation as tmux) - -Typing `/mem` into a live `claude` composer inside a herdr pane and reading the pane back within 0.1 seconds already shows the full autocomplete popup. -This confirms the same hazard tmux already mitigates: submitting immediately after a `/`- or `$`-prefixed send risks Enter landing on a popup selection instead of the literal typed command. -`fm_backend_herdr_send_text_submit` takes the same settle-before-first-Enter parameter tmux's submit core does; the settle-duration DECISION itself lives in `fm-send.sh` (harness-aware, backend-independent), so neither adapter needs its own settle policy. - -`escape` was verified to dismiss the popup while leaving the typed text in the composer, not a full clear. - -## Incident (2026-07-03): a slash command left fully typed but unsubmitted, silently - -Two grok/herdr crewmates were each sent `/no-mistakes` via `fm-send.sh`. -In both panes the command sat fully typed in the composer, unsubmitted (footer still read `Enter:send`), for minutes, until a manual `FM_HOME=<home> fm-send.sh <target> --key Enter` landed it instantly. -`fm-send.sh` had exited 0 both times - no failure surfaced to the caller. - -Root cause, reproduced live against real grok 0.2.82 on an isolated herdr session: the send-text-submit verification at the time used the old delta-based strategy and declared success whenever the captured pane content changed AT ALL between before and after an Enter. -For an argument-taking slash command, the FIRST Enter does not submit - it closes the completion popup and, for a command like `/compact [context]`, EXPANDS the composer text into an argument-hint placeholder (`/compact` -> `/compact compaction instructions`). -The popup disappearing and the composer text changing is a real, visible content change, so the old delta check declared "submitted" after exactly one Enter, even though the composer still held real, unsubmitted text and the footer still read `Enter:send`. -A genuine second Enter was required to actually submit - exactly the manual recovery that worked both times in the incident. -Plain (non-argument) commands like `/new` did submit on the first Enter in the same live test, so the false-positive was specific to commands whose popup selection fills an argument placeholder rather than submitting outright - `/no-mistakes` (optional task-first argument) is exactly that shape. - -The tmux backend was NOT affected by this incident: `fm_tmux_composer_state` reads the actual cursor row and classifies it as pending whenever real text remains, so its retry loop correctly issued the second Enter and landed the same live repro; this was confirmed side-by-side against the same real grok pane. - -**Fix:** `fm_backend_herdr_composer_state` replaced the delta-based check with a structural read of the composer's OWN row, mirroring what the cursor-row read gives tmux. -Herdr's CLI exposes no cursor-row primitive, so the composer row is located by shape instead of position. -For bordered composers, the row is the only line in a generous tail capture whose trimmed content both starts and ends with the same border glyph (`│`, `┃`, or a plain `|`) - the box's own top/bottom rows use rounded corners and never match, popup item rows and separator rows carry no border glyph at all, and the footer help line uses `│` only as an interior separator (never as the first/last character), so none of those can be mistaken for the composer. -For unbordered live composers, added after the 2026-07-07 incident below, the row is a bottom-most trimmed line starting with a verified agent prompt glyph (`❯` for claude or `›` for codex); decorative bordered boxes above it lose to that bottom-most match. -For Pi, added after the 2026-07-14 incident above, the candidate is the content between the bottom-most complete separator pair, admitted only when native Herdr identity reports exactly `pi` with status `idle`, `done`, or `blocked` and the bounded structure is unambiguous. -A popup-close-with-placeholder-fill still reads as real content on that row, so composer fallback correctly classifies it as pending; on the normal idle-baseline path, the same first Enter also fails to start a turn, so native agent-state confirmation likewise retries instead of stopping early. -Known ghost/placeholder composer text (`Type a message...`, verified grok 0.2.82's empty-composer hint) is recognized and still reads as empty. -When ANSI capture is available, the shared `fm_composer_strip_ghost` extractor removes de-emphasised ghost/placeholder runs before classification while retaining real typed input. -The full dim/faint and dark-TRUECOLOR contract is recorded in the 2026-07-10 incident below. -`FM_BACKEND_HERDR_IDLE_RE` extends that placeholder match, `FM_BACKEND_HERDR_BARE_PROMPT_RE` controls the recognized unbordered prompt glyphs, `FM_BACKEND_HERDR_COMPOSER_LINES` controls the tail-window scan depth, and `FM_BACKEND_HERDR_PI_COMPOSER_MAX_LINES` bounds a Pi separator candidate; all four are documented in [`docs/configuration.md`](configuration.md). -See `fm_backend_herdr_composer_state`, `fm_backend_herdr_wait_for_working`, and `fm_backend_herdr_send_text_submit` in `bin/backends/herdr.sh` for the implementation, and `tests/fm-backend-herdr.test.sh`'s composer-state, wait-for-working, and send-text-submit sections for the fake-harness coverage. - -## Composer-state classifier: structural row read, not delta-based - -The herdr adapter no longer diffs raw pane content before/after Enter (see the incident above for why that was unsafe). -It keeps `fm_backend_herdr_composer_state` as a structural classifier for the composer's own content - located as the bottom-most bordered row, verified bare prompt row, or identity-corroborated Pi separator region described above - and reports `empty`, `pending`, or `unknown`. -When ANSI capture is available, the classifier keeps the raw styled row long enough to route it through the shared `fm_composer_strip_ghost` extractor before classification. -The 2026-07-10 incident below records the supported dim/faint and dark-TRUECOLOR ghost/placeholder styling. -That classifier is still the away-mode daemon's affirmative-empty pre-injection guard and the conservative fallback when `fm_backend_herdr_send_text_submit` cannot use an idle/done native agent-state baseline. -Normal idle-baseline submit confirmation now uses herdr's native agent-state instead; see "Native agent-state submit confirmation" for the current submit path. -A dedicated composer-state or cursor-row/style primitive is still a candidate upstream Herdr feature request; it would let the guard/fallback classifier eventually reach tmux's cursor-row precision instead of relying on a structural approximation over captured tail rows and ANSI style. - -All implemented backends expose the identical caller-facing verdict vocabulary (`empty`, `pending`, `unknown`, `send-failed`), so `fm-send.sh` needs no backend-specific branching at all. - -## Session targeting: the `--session` flag, not `HERDR_SESSION` alone - -`HERDR_SESSION=<name>` is the adapter's normal way to select a named herdr session for NON-destructive operations: start, workspace, tab, pane, capture, send, and busy-state calls all still use it (via `fm_backend_herdr_cli`, below). - -Destructive session cleanup is different, and this distinction was learned the hard way. -Verified empirically: on the installed herdr 0.7.1 client, neither an exported `HERDR_SESSION` nor an inline `HERDR_SESSION="$name"` prefix reliably targets a CLI subcommand once ANOTHER herdr server (e.g. the captain's live default session) is already bound on the machine - the client silently falls back to whatever server IS running instead of the requested one. -This is not a hypothetical: it killed the captain's live default herdr server, twice, from real-herdr test cleanup that relied on exactly this assumption (2026-07-02; this section is the full account, and `bin/fm-herdr-lab.sh` now owns the guard that prevents a recurrence). -`herdr server stop` is the sharpest edge of this, because it takes NO target argument at all - it always acts on "whatever server is running," resolved ambiently, with no positional name to catch a misroute. - -The fix, verified against the real binary in an isolated session (both a genuinely separate isolated session and the default session's untouched state confirmed before and after): - -- The `--session <name>` GLOBAL FLAG reliably routes every herdr subcommand tried (`status`, `workspace *`, `tab *`, `pane *`, `agent *`, `server`, `session stop`/`delete`) to the named session, in either leading (`herdr --session <name> <subcommand>`) or trailing (`herdr <subcommand> ... --session <name>`) position - both verified to work identically. -- `bin/backends/herdr.sh`'s `fm_backend_herdr_cli` helper wraps every herdr invocation in the adapter: it sets `HERDR_SESSION` (kept for cosmetic/forward-compat reasons - harmless, and it is what the client's own JSON echoes back) AND appends a trailing `--session <name>`, so every adapter call is correctly scoped regardless of what else is running on the machine. -- For destructive session cleanup specifically, use `herdr session stop <name>` / `herdr session delete <name>` (the explicit-by-name forms - `<name>` is a REQUIRED positional argument, so herdr cannot resolve it ambiguously; herdr's own help text requires literally typing `default` to affect the default session), never the ambient `herdr server stop`. `bin/fm-herdr-lab.sh` now owns this guard as the single source of truth: `fm_herdr_lab_teardown` does the stop-then-delete, gated by a read-only hard guard (`fm_herdr_lab_refuse_if_default`, re-querying `herdr session list --json` immediately before EVERY stop/delete call, refusing on a literal `default` name, a not-found name, or `default:true`) as a second, independent layer that fails closed on any ambiguity. `tests/herdr-test-safety.sh` now sources that helper, so its `herdr_safe_stop_and_delete`/`herdr_refuse_if_default` names are thin delegating wrappers over the same owner. - -The same guard is now a first-class production helper, `bin/fm-herdr-lab.sh`, not just test scaffolding. -It provisions an isolated never-`default` lab session (names must start with `fm-lab-`), runs every task command through `run <session> ...` with a mandatory trailing `--session` appended, and refuses caller-supplied `--session`, any leading option before the subcommand, and every server or session-lifecycle subcommand. -Destructive teardown goes only through `teardown <session>` (or a deliberate mid-run `stop <session>`), each re-running the refuse-default check immediately before every stop and delete. -It also adds a before/after fleet-state tripwire: `provision` records the live `default` session before creating the lab session, and `teardown` verifies that recorded state is byte-identical afterward before clearing it, treating any missing, stopped, or changed default session as a hard failure rather than a warning. -Crewmate briefs for tasks that drive Herdr lifecycle get this exact contract embedded by scaffolding with `bin/fm-brief.sh --herdr-lab`; every crewmate brief scaffolded without the flag instead carries a loud not-enabled gate, because the scaffold cannot detect from the caller-supplied repo string whether the task will touch Herdr lifecycle. - -## ID stability across a server restart - -The original design addendum flagged this as an open risk to verify. -It turned out better than feared. - -`herdr session stop <name>` followed by a fresh `herdr server --session <name>` - the realistic "firstmate restarted, herdr server needs reattaching" recovery scenario - preserves workspace id, tab id, pane id, and every label exactly. -Herdr persists this metadata to disk per named session, independent of the live server process. -What does NOT survive is the underlying shell/agent process inside each pane (a fresh shell starts in its place) and each pane's live `agent_status` (resets to unknown). - -P2 verified this in the single-workspace shape only. -Re-verified here in the MULTI-workspace shape (P3, workspace-per-home): with two coexisting workspaces (a `firstmate` and a `2ndmate-<secondmate-id>`, each with its own tab/pane) in one isolated session, a `session stop` + fresh server restart preserved BOTH workspaces' ids and labels, and BOTH tasks' pane ids, exactly - automated in `tests/fm-backend-herdr-smoke.test.sh`'s restart-stability section. +A Herdr pane id contains a colon, so the adapter splits `window=` on the first colon only. +The recorded pane is the operational fast path. +Workspace and tab ids support verification and cleanup but are not inferred from mutable labels during normal operation. -Practical consequence: a stored `herdr_pane_id=` remains a valid, fast-path operational target across an ordinary server restart within the same named session, regardless of how many other homes' workspaces coexist in that session. -The adapter still implements label-based recovery (`fm_backend_herdr_list_live`), both for a differently-configured or freshly-created session where old ids would not exist at all, and as the more defensive default in general. +## Current transport behavior -## Respawn idempotency: a restored task tab is a husk, not a duplicate +The adapter starts and polls a named server before workspace, tab, pane, or agent calls. +Every Herdr invocation goes through `fm_backend_herdr_cli`, which sets the environment and passes an explicit trailing `--session <name>`. +An environment variable alone is not reliable when another Herdr server is running. -A restart's other consequence (the previous section's "what does NOT survive") used to make every fleet respawn after it a manual chore: a restored `fm-<id>` tab comes back alive but with a fresh shell process and no registered agent (`agent_status` reset to unknown, `agent get` reporting `agent_not_found`) - or, if the pane's own process failed to restart at all, structurally gone (`pane get` reporting `pane_not_found`). -Before this fix, `fm_backend_herdr_create_task`'s duplicate-label guard treated either shape identically to a genuinely live duplicate and refused unconditionally, so recovering a fleet after a real herdr server restart (or, worse, a full reboot) meant closing every husk pane by hand before firstmate could spawn into it again - this reproduced in production on 2026-07-03. +Literal text and Enter are separate operations for ordinary steers. +Spawn-time fixed commands may use Herdr's atomic run primitive. +Enter, Escape, and Ctrl-C are supported. +Slash and dollar-prefixed input uses the shared harness-aware settle before the first Enter so a completion popup cannot consume it. +Text is typed once; only Enter is retried. -The guard is now husk-aware. -`fm_backend_herdr_pane_agent_state` classifies an existing same-labeled tab's pane as one of `dead` (`pane get` -> `pane_not_found`), `no-agent` (the pane exists but `agent get` -> `agent_not_found` - the restored-plain-shell shape, and also what a future `resume_agents_on_restore = false` herdr config would produce unconditionally), `live` (a real registered `agent_status`, including idle/blocked - never just "working"), or `unknown` (anything unparseable or unexpected). -Only `dead` and `no-agent` are treated as a husk; `live` and `unknown` both refuse exactly as before, fail-safe toward refusal whenever the state cannot be classified with confidence. -A confirmed husk is closed and replaced instead of refused: `fm_backend_herdr_create_task` always creates the REPLACEMENT tab first, closes the preexisting husk tab by id only after that succeeds, and verifies no same-labeled tab except the replacement remains before returning success. -It never closes the husk first, because closing a workspace's last remaining tab deletes the whole workspace on real herdr (see "Default workspace lifecycle" above) and a session-restore husk can legitimately be that workspace's only tab. -This is the identical create-before-close safety argument `fm_backend_herdr_workspace_prune_seeded_default_tab` already established for the seeded default tab. +On an idle or done native baseline, submit confirmation waits for `working` or `blocked` across a bounded polling window. +On an already active or unreadable baseline, it falls back to conservative composer clearance. +A fully unreadable target stops retrying and reports unknown. +The poll density bounds the residual possibility of an extremely fast complete turn; a missed transition can cause only a redundant Enter on an empty composer, never duplicate message text. -Verified against the real binary (`tests/fm-backend-herdr-respawn-idem-e2e.test.sh`, an isolated non-default session): a real `session stop` + fresh `herdr server` restart, followed by a same-labeled `fm_backend_herdr_create_task` call, closes and replaces the restored no-agent husk for both a crewmate/scout-shaped and a `--secondmate`-shaped task (the same function serves both spawn paths), while a pane carrying a genuinely registered agent (via herdr's own `pane report-agent`) still refuses. -The `dead` (`pane_not_found`) classification is covered at the unit level (`tests/fm-backend-herdr.test.sh`, canned-response fake) but not end-to-end against the real binary: killing a pane's underlying process on a live server was observed to make herdr immediately reap both the pane AND its tab together (so the tab never lingers in `tab list` for the duplicate check to even find), and a session restart was never observed to produce a structurally-dead-but-still-listed pane either - only a live, agent-less one. -The `dead` branch remains a conservative, defensively-coded path for a herdr failure mode (e.g. a restored process that fails to start) that has not been reproduced against the real binary. +`pane read --lines N` can return empty output when N is below the viewport height. +The capture owner requests at least 200 lines from Herdr and trims locally to the caller's bound. +This generous floor is required for small composer and peek reads. -## Agent liveness probe reuses the husk classifier +Herdr's native agent state can read idle while a harness waits on its own long foreground tool. +The shared crew-state path therefore corroborates every native non-busy or unreadable result with the recorded harness's rendered busy signature before concluding that a pane is not working. +A human-blocked permission dialog has no busy banner and still surfaces. -`bin/fm-bootstrap.sh`'s session-start secondmate-liveness sweep needs the same underlying question the husk check above already answers with confidence: is this pane's agent actually alive, or is it a bare shell / gone pane pretending to be a live endpoint? -Rather than add a second Herdr classifier, `fm_backend_herdr_agent_state` (`bin/backends/herdr.sh`) is a thin wrapper around the already-verified `fm_backend_herdr_pane_agent_state`: a structurally gone pane becomes `missing`, a restored agent-less shell becomes `dead`, a registered agent becomes `alive`, and an unexpected read becomes `unreadable`. -`fm_backend_herdr_agent_alive` preserves the older three-state compatibility view for callers that do not need to distinguish a missing endpoint from an existing husk. -No new empirical verification was needed for the mapping itself because `fm_backend_herdr_pane_agent_state`'s four states are already verified above, both at the unit level and, for `no-agent`, against the real binary via the respawn-idempotency end-to-end test. -Unlike tmux's probe, Herdr's has no equivalent "which harness is running under a generic interpreter name" ambiguity because the classification comes from Herdr's own registered-agent state, so it correctly resolves every verified harness including Pi. +## Composer and injection safety -## End-to-end verification (spawn -> steer -> peek -> done -> merge -> teardown) +Herdr has no direct cursor-row primitive. +The adapter locates the bottom-most recognized bordered row, Claude `❯` row, Codex `›` row, or a Pi separator region admitted only when native identity is exactly Pi and state is idle, done, or blocked. +A working Pi, pending middle row, missing identity, incomplete separator pair, or over-tall candidate remains pending or unknown. -Beyond the fake-CLI unit tests (`tests/fm-backend-herdr.test.sh`) and the real-CLI smoke tests (`tests/fm-backend-herdr-smoke.test.sh` and `tests/fm-backend-autodetect-smoke.test.sh`), the full firstmate lifecycle was driven end to end against a real `claude` crewmate through this branch's own scripts, in a scratch `FM_HOME`, a scratch `local-only` git project, and an isolated `HERDR_SESSION`: +ANSI capture preserves de-emphasized placeholder style. +`bin/fm-composer-lib.sh` is the fleet-wide owner that strips dim or faint runs and dark truecolor placeholders while retaining bright typed input. +If a future Herdr version strips ANSI style, ghost suggestions become pending rather than empty, which safely defers injection and eventually raises the wedge alarm. -1. `FM_HOME=<scratch> FM_BACKEND=herdr HERDR_SESSION=<isolated> bin/fm-spawn.sh herdr-e2e-t1 projects/scratch-e2e-project claude` - spawned successfully, printing `backend=herdr` in the summary and writing `herdr_session=`/`herdr_workspace_id=`/`herdr_tab_id=`/`herdr_pane_id=` to the task's meta. -2. `FM_HOME=<scratch> HERDR_SESSION=<isolated> bin/fm-peek.sh fm-herdr-e2e-t1` - showed the live claude trust dialog. -3. `FM_HOME=<scratch> HERDR_SESSION=<isolated> bin/fm-send.sh fm-herdr-e2e-t1 --key Enter` - accepted the trust dialog. -4. `FM_HOME=<scratch> HERDR_SESSION=<isolated> bin/fm-peek.sh fm-herdr-e2e-t1` again - showed claude actively working through the brief (creating the branch, writing the file). -5. `FM_HOME=<scratch> HERDR_SESSION=<isolated> bin/fm-send.sh fm-herdr-e2e-t1 "captain says: proceed as planned"` - a plain-text steer, exercising the send-and-verify path; the text appeared correctly in the pane. -6. The crewmate appended `done: hello.txt committed on fm/herdr-e2e-t1` to its status file, and its commit (`add hello.txt` on branch `fm/herdr-e2e-t1`) was confirmed present in the project's git history. -7. `bin/fm-teardown.sh herdr-e2e-t1` **REFUSED**, exactly as required: `REFUSED: local-only worktree ... has work not yet merged into main and not on any remote.` -8. `bin/fm-merge-local.sh herdr-e2e-t1` - fast-forwarded local `main` to the crewmate's commit. -9. `bin/fm-teardown.sh herdr-e2e-t1` now succeeded: returned the treehouse worktree, closed the herdr pane (verified gone via `herdr pane get`), and removed all of the task's `state/` files. +A bare shell prompt is never an empty agent composer. +Away-mode injection proceeds only on an affirmative `empty` result, never on unknown. +This prevents a dead agent pane from receiving and possibly executing an escalation as shell input. -Two real, non-obvious bugs were caught and fixed by this pass alone, both already reflected above and in `bin/backends/herdr.sh`: +The current operational envelope starts with U+2063 and `FIRSTMATE_OP: `. +The separate routed-request carrier uses `[fm-from-firstmate]` plus U+2063. +U+2063 survives Herdr terminal input as text, unlike the legacy ASCII control separator that could erase the visible routing label. +`bin/fm-operational-input.sh` owns current operational construction and parsing, and the AFK skill owns legacy away-input compatibility. +No Herdr-specific copy of that protocol exists. -- The `pane read --lines N` small-N bug (see above) - without the fix, this E2E run flaked intermittently on the very first `send_text_line` call. -- `pane get`'s `.result.pane.cwd` field is frozen at pane-creation time and never updates; `fm_backend_herdr_current_path` originally read it and would have made `fm-spawn.sh`'s worktree-discovery poll misresolve the acquired treehouse worktree path (it would see the pane's ORIGINAL directory, not where `treehouse get`'s subshell actually landed) - fixed by reading `.result.pane.foreground_cwd` instead, which tracks the live running process. +## Restart and liveness behavior -The isolated herdr session, the treehouse pool worktree, and the scratch `FM_HOME` were all stopped/deleted/removed after this run, using the guarded teardown described in "Session targeting" above; the captain's default herdr session and the live tmux fleet were never touched at any point. +Stopping and restarting a named Herdr server preserves workspace, tab, pane, and label ids, but the underlying harness processes and live agent registrations do not survive. +A restored same-labeled tab with a missing pane or no registered agent is a husk. +Create replaces only a confidently dead or no-agent husk, creates the replacement before closing the old tab, and refuses live or unknown states. +This prevents closing the workspace's last tab before a replacement exists. -## End-to-end verification: workspace-per-home (P3) +The generic Herdr agent-liveness probe reuses the same classifier. +A structurally gone pane becomes `missing`, a restored agent-less shell becomes `dead`, a registered agent becomes `alive`, and an unexpected read becomes `unreadable`. +Unlike tmux process-name inspection, native registration can classify Pi without guessing from a generic interpreter name. -`tests/fm-backend-herdr-workspace-per-home-e2e.test.sh` drives `bin/fm-spawn.sh` and `bin/fm-teardown.sh` for real, in a scratch `TMP_ROOT` holding two scratch firstmate homes (a primary-shaped one with no marker, and a secondmate-shaped one carrying `.fm-secondmate-home`) and two scratch local-only projects, on one isolated `HERDR_SESSION` (never the captain's default), with the same `herdr_safe_stop_and_delete` guarded cleanup. -This exercises the fm-spawn.sh-level behavior the adapter-primitive smoke test cannot reach: the label-resolution home-shadowing for a `--secondmate` spawn, and - the one path that had never run before this test - a crewmate spawned FROM a secondmate's own `fm-spawn.sh` process. +The session-start sweep uses this probe. +Mid-session secondmate liveness is not implemented because idle secondmates are deliberately exempt from stale-pane escalation and need a separate periodic identity signal. -1. A primary-shaped home spawns an ordinary crewmate (`cm1`) on the herdr backend: its tab lands in a workspace herdr itself labels `firstmate`. -2. The PRIMARY spawns a `--secondmate` task (`e2esm1`, home = the secondmate-shaped scratch home): its tab lands in a DIFFERENT workspace than `cm1`'s, labeled `2ndmate-e2esm1` by herdr - proving the `fm-spawn.sh` FM_HOME-shadow glue for this one launched-by-the-primary case. -3. A crewmate (`cm2`) is spawned by running `bin/fm-spawn.sh` again, this time with `FM_HOME` set to the SECONDMATE's own home (simulating the secondmate running its own spawn, exactly as it would live) - no special-casing needed. Its tab lands in the SAME workspace as `e2esm1`'s (`2ndmate-e2esm1`), never the primary's - confirming per-home resolution "falls out" naturally for this path, as the design predicted, now proven rather than merely inspected. -4. `fm_backend_herdr_list_live`, called with `FM_HOME` set to each home in turn, sees only that home's own tab(s): the primary's list shows only `cm1`; the secondmate's list shows both `e2esm1` and `cm2`, and neither list leaks into the other. -5. `bin/fm-teardown.sh cm1` closes only `cm1`'s pane - the secondmate's own pane and `cm2`'s pane, both confirmed still open via `herdr pane get`, survive untouched. `bin/fm-teardown.sh cm2` (run with the secondmate's own `FM_HOME`) then closes only `cm2`'s pane, leaving the secondmate's own pane (same workspace) open. +## Push events and polling fallback -All ten assertions passed on the real binary on the first run. -As with every other real-herdr test in this document, the default session's own workspace state (label, tab count) was confirmed byte-identical immediately before and immediately after the run. +Protocol 16 can subscribe to `pane.agent_status_changed` over one bounded Unix-socket reader. +`bin/fm-transition-lib.sh` owns the backend-neutral transition vocabulary and policy. +The Herdr adapter subscribes before reconciling current levels, buffers edges during reconciliation, and returns fresh blocked transitions for this home's panes. +The watcher maps the pane back to the task and skips secondmate endpoints and declared `paused:` waits. -## Away-mode daemon: herdr supervisor-pane support +The push path only shortens latency. +Polling runs every cycle and remains the permanent fallback when protocol 16, the event schema, Python, connection, subscription, or repeated reader execution is unavailable. +There is still one watcher process; the event reader is a bounded child of that watcher. -`bin/fm-supervise-daemon.sh` (the `/afk` sub-supervisor) was tmux-only through 2026-07-03: it discovered its own injection target from `$TMUX_PANE`, and injected via raw `tmux display-message`/`tmux capture-pane`/`tmux send-keys` calls with no backend indirection. -On a herdr-based fleet (firstmate itself running with `HERDR_ENV=1`, no `$TMUX_PANE`), this failed outright at startup: `TMUX_PANE` is unset, so discovery fell through to the legacy `firstmate:0` fallback, which then failed the tmux pane-exists probe and refused to start. +`tests/fm-backend-herdr-eventwait-smoke.test.sh`, `tests/fm-transition-lib.test.sh`, and `tests/fm-supervision-events.test.sh` cover capability, subscribe-then-reconcile ordering, dedupe, exemptions, and polling fallback. -The fix is transport-layer only - discovery, injection, and the busy/composer guards now dispatch through the SAME `bin/fm-backend.sh` primitives every other backend-aware script already uses (`fm_backend_target_exists`, `fm_backend_busy_state`, `fm_backend_capture`, `fm_backend_send_text_submit`, and the new `fm_backend_composer_state` dispatcher added alongside this work). -Classification policy, batching, the max-defer escape, the `FM_INJECT_MARK` sentinel contract, locks, and wake-queue handling are all unchanged. +## Away-mode supervisor support -**Discovery.** `FM_SUPERVISOR_TARGET` remains the explicit override, now accepting either a tmux target or a herdr `"<session>:<pane-id>"` target. -A new `FM_SUPERVISOR_BACKEND` override (`tmux`|`herdr`) resolves independently, mirroring `bin/fm-backend.sh`'s own `fm_backend_detect`: `$TMUX_PANE` set selects tmux (even nested inside herdr, matching the innermost-first rule); `$HERDR_ENV=1` with `$HERDR_PANE_ID` present selects herdr, composing the target as `"${HERDR_SESSION:-default}:${HERDR_PANE_ID}"`; absent both, the daemon falls back to tmux/`firstmate:0`, byte-identical to its pre-herdr-support behavior. -Other runtime backends, including zellij, orca, and cmux, are not yet supported as supervisor backends - the daemon refuses loudly at startup (`FM_SUPERVISOR_SUPPORTED_BACKENDS="tmux herdr"`) rather than misapplying tmux primitives to a pane that isn't a tmux pane. +The away daemon supports tmux and Herdr supervisor panes only. +It refuses Zellij, Orca, and cmux as supervisor backends rather than applying the wrong transport. +For Herdr, target existence, native state, capture, composer state, and verified submit all route through the shared backend dispatcher and the explicit named-session CLI owner. +The pane-independent max-defer alert is configured in [`wedge-alarm.md`](wedge-alarm.md). -**Injection dispatch.** `inject_msg`'s pane-exists probe, busy-guard (`pane_is_busy`), composer-guard (a direct `fm_backend_composer_state` read; see the composer-safety note below), and verified submit all take an optional `<backend>` argument (defaulting to `tmux` when omitted, so every pre-existing caller/test is unaffected) and route through the generic dispatchers instead of calling `tmux` directly. -For `backend=tmux` every dispatch resolves to the exact same underlying call as before (`fm_backend_capture`'s tmux arm runs the identical `tmux capture-pane -p -t <target> -S -40`; `fm_backend_tmux_send_text_submit` re-exports `fm_tmux_submit_core` verbatim), so tmux behavior is unchanged byte-for-byte. -For `backend=herdr`, busy detection tries the native `agent.get`-backed `fm_backend_herdr_busy_state` first, trusts only `busy` outright, and corroborates every non-`busy` verdict with the shared regex-over-capture reader before treating the supervisor pane as not busy. -This mirrors the per-task stale-pane busy check `bin/fm-supervise-daemon.sh`'s `stale_window_is_busy` already used; composer/pending detection and the verified submit route through `fm_backend_herdr_composer_state`/`fm_backend_herdr_send_text_submit`. -The wedge alarm's supervisor-client status-line flash (`tmux display-message ...`) is tmux-only cosmetic UI with no herdr equivalent, so it is skipped for non-tmux backends. -A max-defer wedge also attempts the configured backend-independent active alert described in [`wedge-alarm.md`](wedge-alarm.md), while the ERROR log line and durable `state/.subsuper-inject-wedged` marker remain backend-independent. +Harnesses with native tracked background execution can run the daemon in their terminal. +Pi has no such mechanism. +`bin/fm-afk-launch.sh` therefore creates a dedicated unfocused Herdr workspace, runs the daemon there with an explicit supervisor target and backend, records the exact daemon pane, and closes only that pane on stop. +It never splits the captain's active tab and never uses shell `&`. +Recovery reconciles only the recorded exact id. -**A pre-existing bug this surfaced: `fm_backend_target_exists`'s herdr arm.** Before this task, that function's herdr case called `HERDR_SESSION="$session" herdr pane get "$pane"` directly, WITHOUT the `--session` flag. -Per "Session targeting" above, `HERDR_SESSION` alone is not reliably honored once another herdr server is already bound on the machine - it silently falls back to whatever server IS running. -This function happened to look correct in every prior test because those tests only ever had ONE herdr server running at a time. -Verifying the away-mode daemon end to end against a real, isolated `HERDR_SESSION` - while the ambient default herdr session was also running (the normal shape of an actual firstmate fleet) - reproduced it directly: the daemon's own startup target-exists check spuriously refused a genuinely live pane in the isolated session because the ambient default session's socket answered instead. -Fixed by routing through `fm_backend_herdr_cli` (which appends `--session` on top of the env var) instead of the raw ad hoc call. -This fix is backend-plumbing, not daemon-specific: it also corrects the same liveness check other callers use (`bin/fm-session-start.sh`'s per-task endpoint-liveness digest read). +On stop, the daemon receives termination while `state/.afk` still exists so its final flush can run, the recorded terminal is closed, and the AFK flag is removed last. +A fresh entry clears stale transient escalation caches, while durable queue and task records remain authoritative. -**Empirical verification (real herdr, isolated session only).** `tests/fm-afk-inject-herdr-e2e.test.sh` mirrors `tests/fm-afk-inject-e2e.test.sh`'s three scenarios (human-partial-input deferral, swallowed-Enter retry, a normal single digest) plus a fourth (a persistently pending composer that never clears must alarm via `state/.subsuper-inject-wedged`, preserve the buffer, and never crash the daemon) against a real, throwaway, NEVER-default `HERDR_SESSION`, torn down with `herdr_safe_stop_and_delete` exactly like `tests/fm-backend-herdr-smoke.test.sh`. -The "supervisor pane" is a tiny deterministic bash loop, not a real harness binary, that draws a bordered composer row to exercise the bordered branch of `fm_backend_herdr_composer_state`. -Because submit confirmation now uses native agent-state on idle baselines, the fixture also registers itself as a herdr agent via `herdr pane report-agent` and reports an idle->working->idle cycle around each submitted line. -A thin `herdr` PATH shim swallows exactly one `pane send-keys <pane> enter` call to simulate the swallowed-Enter scenario, since herdr's real CLI has no built-in way to drop a keystroke. -Real claude/codex unbordered prompt coverage lives in `tests/fm-backend-herdr.test.sh`'s captured-fixture regression tests described in the 2026-07-07 incident below. +## Destructive lab safety -Building that test surfaced one more real finding worth recording for anyone writing a similar herdr-driven composer script: `tput cols`, called from WITHIN a script launched into a herdr pane via `pane run`/`send-text`, reported a stale/default `80` regardless of the pane's actual width, while an interactively-typed one-off `tput cols` in the same pane correctly reported its real width (54, in the environment this was verified in). -A composer redraw that trusts `tput cols` for its own line-wrapping math can therefore silently overflow the pane's real width and wrap across two terminal rows - breaking the structural single-row border classifier's assumption (the digest looked "concatenated with itself" because the guard never fired: the composer read `unknown` instead of `pending`, so the busy/composer guard did not defer a second attempt). -The test's composer script works around this with a hardcoded conservative width rather than trusting `tput cols` in this execution context. -This `tput` issue is a test-harness-only concern: once the test's own composer script stayed within the pane's real width, `fm_backend_herdr_composer_state` and `fm_backend_herdr_send_text_submit` behaved as expected, but it remains a sharp edge for any future herdr-launched interactive script that computes its own layout from `tput`. +Never use ambient `herdr server stop` for Firstmate verification. +An environment-only session selection can silently reach a different running server, and the ambient stop command has no explicit target. -## Incident (2026-07-07): away-mode escalation redelivery loop on herdr +`bin/fm-herdr-lab.sh` is the sole supported lifecycle helper for isolated verification. +It provisions only non-default names beginning with `fm-lab-`, appends an explicit `--session` to allowed task commands, refuses caller-supplied session flags and server/session lifecycle subcommands, and performs destructive stop/delete only through its guarded lifecycle actions. +Immediately before every destructive call it re-queries the named session and refuses empty, missing, literal `default`, or `default:true` identities. +Its before/after tripwire requires the live default-session snapshot to remain byte-identical. -While `state/.afk` was set on a herdr-backed fleet, `bin/fm-supervise-daemon.sh` re-injected the SAME buffered escalation digest into the primary's own supervisor pane every housekeeping cycle instead of clearing `state/.subsuper-escalations` once delivery landed. -Observed twice: 2026-07-06 (three byte-identical digests in a row) and 2026-07-07 (two byte-identical catch-all-scan digests), each redelivery waking the primary's LLM turn for an escalation it had already handled - defeating away-mode's whole point. +The helper's header and `--help` own exact commands. +Tests use thin compatibility wrappers in `tests/herdr-test-safety.sh` and never duplicate the destructive policy. -Reproduced live against a real, isolated `HERDR_SESSION`, a real `claude` process (the primary's own harness) as the supervisor pane, and the real `fm-supervise-daemon.sh` (not a synthetic composer script): with one buffered `stale persisted 241s` escalation and `FM_HOUSEKEEPING_TICK=1`, the daemon delivered the identical digest to the live pane at least 5 times in 40 seconds, and the agent itself eventually replied "The message is identical again and my position hasn't changed... I'll treat further identical escalations as noise" - an exact live match for the reported symptom. +## Active limits -Root cause: `fm_backend_herdr_composer_state`'s structural composer-row read (added for the 2026-07-03 incident above) recognizes only BORDERED composer rows - a line whose trimmed content both starts and ends with the same border glyph (`│`, `┃`, `|`). -Real `claude`'s live input row is a BARE, unbordered `❯ …` - no border glyph anywhere around it - flanked by plain horizontal-rule separator lines, not a box. -Claude's own startup welcome banner IS bordered, so immediately after launch the classifier's "last bordered row wins" scan locks onto the banner's own blank interior spacer row and misreads it as the composer (a coincidental, and wrong, "empty"). -Once ordinary conversation scrolls that banner out of the 20-line capture window - true of any real supervisor pane with any history at all, which is every production case - NO bordered row exists anywhere in view, so the classifier reports `unknown` for a genuinely empty composer, forever. -At the time, `fm_backend_herdr_send_text_submit` treated only a composer `empty` verdict as a confirmed submit; `unknown` counted as failure, so `escalate_flush` never cleared the buffer even though the real Enter genuinely submitted the digest to the real pane. -At the time, the composer-guard deferred only on `pending` (never `unknown`), so the next housekeeping tick's flush attempt retyped and resubmitted the SAME unmodified buffer content - the redelivery loop. -That guard has since been hardened to require an affirmatively-`empty` composer (see "Composer-emptiness safety" below), so an `unknown` verdict now defers injection instead of proceeding. +- Herdr remains experimental. +- Presentation ordering needs protocol 16 and Python and is best-effort only. +- Mutable labels can collide; they are never destructive authority. +- Ghost and placeholder recognition depends on ANSI de-emphasis and fails safely to pending when unavailable. +- Mid-session secondmate liveness is not implemented. +- OpenCode 1.18.4 can accept Enter while busy without clearing the composer. + The tmux backend has a busy-queue fallback, but Herdr still reports this case as submit pending and needs a separate adapter fix. +- Only tmux and Herdr can host the away-mode supervisor terminal. -Also discovered while reproducing: real `codex` (0.142.x) has the identical unbordered-live-row shape, using `›` instead of claude's `❯`, confirming this is not claude-specific. -Codex additionally shows dynamic tip/hint text in its idle composer rather than a blank row. -The first fix deliberately left that as a conservative `pending` verdict at the composer-guard layer because plain text could not distinguish a ghost suggestion from real typed input. -The 2026-07-08 follow-up below closes that gap using herdr's ANSI capture, which preserves Codex's faint styling for ghost suggestions. - -**Fix:** `fm_backend_herdr_composer_state` now recognizes TWO composer-row shapes in one scan - the existing bordered shape, and a new bare (unbordered) shape: a trimmed line that STARTS with a verified agent-specific prompt glyph (`❯` claude or `›` codex) with no closing border required at all. -The bare-row default is deliberately limited to `❯` and `›`, while generic shell-style glyphs (`>`, `$`, `%`, `#`) stay recognized only after the bordered shape has already identified a composer row, so a no-agent shell fallback cannot be misread as a delivered escalation. -Both shapes are checked in the SAME forward scan, keeping whichever match comes LAST (bottom-most on screen), rather than trying bordered-only first and falling back to bare-only when nothing bordered is found: a bordered decorative box (a welcome banner, an update notice) is always rendered ABOVE the live composer, never below it, in every harness observed, so "last match of either shape wins" always resolves to the genuinely live, bottom-most row instead of a stale decorative box still sitting in the capture window. -See `fm_backend_herdr_composer_state` in `bin/backends/herdr.sh` for the implementation, and `tests/fm-backend-herdr.test.sh`'s "unbordered (bare) composer rows" section (fixtures captured verbatim from real `claude`/`codex` panes) for the regression coverage - each of those tests read `unknown` before this fix and reads the correct verdict after. - -## Incident (2026-07-08): away-mode delivery wedged on Codex ghost suggestions - -While `state/.afk` was set on a herdr-backed Codex primary, `bin/fm-supervise-daemon.sh` repeatedly logged `inject deferred: supervisor pane has pending input (non-empty composer)` even though the captain had not typed anything. -The buffered escalations stayed in `state/.subsuper-escalations` until the max-defer alarm wrote `state/.subsuper-inject-wedged`. - -**Environment.** - -Commands: +## Regression entry points ```sh -herdr status --json | jq -c '{client:.client, server:.server}' -codex --version -``` - -Output: - -```text -{"client":{"version":"0.7.3","channel":"stable","protocol":16,"binary":"/etc/profiles/per-user/kunchen/bin/herdr","session":null},"server":{"status":"running","running":true,"version":"0.7.3","protocol":16,"capabilities":{"live_handoff":true,"detached_server_daemon":false},"compatible":true,"socket":"/Users/kunchen/.config/herdr/herdr.sock","session":null,"restart_needed":false}} -codex-cli 0.142.1 -``` - -**Failing reproduction, before the fix.** - -The reproduction used a throwaway `HERDR_SESSION=fm-afk-codex-ghost-12002`, a scratch `FM_STATE_OVERRIDE`, a real `codex` process in a herdr pane, and the real `bin/fm-supervise-daemon.sh`. -The supervisor target was `fm-afk-codex-ghost-12002:w1:p2`. -The pane capture showed an idle Codex composer with ghost suggestion text: - -```text -› Run /review on my current changes - - gpt-5.5 xhigh · Context 100% left · /private/var/fo… -``` - -The composer classifier and daemon log showed the false pending-input guard: - -```text -composer_state=pending -[2026-07-08T09:38:03-0700] daemon starting (pid 12747); target=fm-afk-codex-ghost-12002:w1:p2; target_source=FM_SUPERVISOR_TARGET; backend=herdr; backend_source=FM_SUPERVISOR_BACKEND; afk=on; inject_skip='heartbeat'; stale_escalate=240s; batch=0s -[2026-07-08T09:38:04-0700] inject deferred: supervisor pane has pending input (non-empty composer) -[2026-07-08T09:38:05-0700] inject deferred: supervisor pane has pending input (non-empty composer) -[2026-07-08T09:38:05-0700] ERROR: away-mode escalation undelivered 3s; inject could not confirm a submit (supervisor pane busy or wedged). Buffer + wake-queue preserved; alarm marker written. -``` - -The wedge marker preserved the buffered event: - -```text -fm away-mode inject WEDGED: 7s undelivered as of 2026-07-08T09:38:09-0700 -The supervisor pane could not accept an escalation. Buffered items: -Wheelhouse shipped status: done: PR ready -``` - -**Style evidence.** - -Herdr's ANSI capture preserves Codex's distinction between ghost suggestion text and real typed text. -An idle suggestion is faint after the bold prompt: - -```text -\e[0m\e[1m› \e[0m\e[2mRun /review on my current changes\e[0m -``` - -Real typed input with the same prompt is not faint: - -```text -\e[0m\e[1m› \e[0mhello captain -``` - -**Fix.** - -`fm_backend_herdr_composer_state` now prefers `herdr pane read --format ansi` for composer classification. -It still locates the same bottom-most bordered or bare prompt row, strips ANSI for shape matching, and then treats a bare-prompt tail as empty only when the raw ANSI row shows that tail rendered faint. -This ignores Codex ghost suggestions such as `Find and fix a bug in @filename`, `Write tests for @filename`, and `Run /review on my current changes` while preserving the `pending` verdict for non-faint real typed text after the same `›` prompt. - -**Passing reproduction, after the fix.** - -The same real-herdr shape used `HERDR_SESSION=fm-afk-codex-fixed-29849`, target `fm-afk-codex-fixed-29849:w1:p2`, real `codex`, and the real daemon. -The captured idle composer showed faint ghost suggestion text: - -```text -\e[0m\e[1m› \e[0m\e[2mWrite tests for @filename\e[0m -``` - -The fixed classifier and daemon result: - -```text -composer_state=empty -[2026-07-08T09:42:00-0700] daemon starting (pid 31034); target=fm-afk-codex-fixed-29849:w1:p2; target_source=FM_SUPERVISOR_TARGET; backend=herdr; backend_source=FM_SUPERVISOR_BACKEND; afk=on; inject_skip='heartbeat'; stale_escalate=240s; batch=0s -[2026-07-08T09:42:11-0700] daemon shutting down -``` - -No `inject deferred: supervisor pane has pending input` line was emitted, `state/.subsuper-escalations` was empty afterward, and no wedge marker was written. -The unit regression coverage is `tests/fm-backend-herdr.test.sh`'s `test_composer_state_codex_faint_suggestion_is_empty`, `test_composer_state_codex_non_faint_same_text_is_pending`, and `test_composer_state_codex_dynamic_idle_tip_reads_empty_when_faint`. - -## Native agent-state submit confirmation (fixes the codex idle-tip gap) - -`fm_backend_herdr_send_text_submit` now records a pre-Enter native agent-state baseline before choosing the confirmation signal. -When that baseline is legibly idle or done, it confirms a submit by polling herdr's own semantic agent-state (`agent get`) for a submit-active transition (`working` or `blocked`), via the new `fm_backend_herdr_wait_for_working` helper. -Composer content (`fm_backend_herdr_composer_state`) is still used for the pre-injection empty-box guard (`bin/fm-supervise-daemon.sh`'s `inject_msg`, which reads `fm_backend_composer_state` directly and requires an affirmatively-`empty` verdict; see "Composer-emptiness safety" below). -It is also the conservative fallback for submit attempts whose pre-Enter baseline is already submit-active or unreadable, because a preexisting `working`/`blocked` status cannot prove that this Enter landed. -This makes the normal idle-baseline confirmation path cross-agent: it no longer depends on what a harness's idle composer happens to display. - -This originally fixed the practical submit-confirmation effect of the Codex idle-tip gap left open by the 2026-07-07 incident above. -The 2026-07-08 follow-up fixed the pre-injection composer guard itself by using herdr's ANSI capture to ignore faint Codex ghost suggestions. -The submit-confirmation path still deliberately uses native agent-state on idle baselines, so it remains independent of composer rendering. - -### Design: two failure directions, both guarded - -A message that lands from an idle or done baseline must move the target agent into a submit-active state. -Two ways this signal can be missed, and how the design guards each: - -- **Slow transition.** A single check right after Enter could sample before herdr has updated `agent_status`, wrongly concluding "not submitted" and causing a needless extra Enter (harmless on its own here, since only Enter is retried, never the text - but wasteful and, for a stricter caller, could read as a false negative). - Fix: `fm_backend_herdr_wait_for_working` samples repeatedly (`FM_BACKEND_HERDR_SUBMIT_POLLS`, default 6) across the larger of the caller's per-attempt budget (`<enter-sleep>`) and herdr's own minimum confirmation budget (`FM_BACKEND_HERDR_SUBMIT_MIN_SLEEP`, default 0.6s), instead of checking once at the end. - A transition landing anywhere in that window is caught, and the function returns the instant `busy` is observed, without waiting out the rest of the budget. -- **Instant round-trip.** A turn that starts and returns to idle entirely between two polls would, in the limit, never show as submit-active at all. - This is not eliminated in principle, but it is bounded by how densely `FM_BACKEND_HERDR_SUBMIT_POLLS` samples the budget, and the empirical evidence below shows real turns take far longer than the sampling interval to even START, let alone finish. - On the (unobserved) residual chance this happens, the function reports `pending`, and the caller's own invariant (retry Enter only, never retype) means the worst case is a redundant Enter landing on an already-empty composer - a no-op, not a duplicate delivery of the message text. - -`fm_backend_herdr_wait_for_working` also distinguishes a genuine "not yet submit-active" reading (the target was legibly read at least once, `idle`/`done` was observed, `working`/`blocked` never was) from a hard read failure (every poll in the window failed to read the target at all). -Only the latter reports `unknown` and skips further Enter retries - matching the pre-existing "never retry past an unreadable target" invariant the composer-based design already had. - -### Empirical evidence (2026-07-07, herdr 0.7.1, protocol 14, macOS aarch64) - -Verified against real `claude` (2.1.203) and real `codex` (0.142.1) agents in an isolated, throwaway `HERDR_SESSION` (never the default session), using `herdr_safe_stop_and_delete` for cleanup exactly like every other real-herdr test in this document. - -Method: for each agent, with the pane genuinely idle, `herdr pane send-text <pane> "<trivial prompt>"` followed by `herdr pane send-keys <pane> enter`, then `herdr agent get <pane>` polled at roughly 30ms intervals, timestamping the FIRST poll that reports `agent_status: working`. - -Ten repeated trials per agent (a fresh trivial prompt each run, e.g. "reply with just the word pong"): - -| Agent | First-observed-working latency across 10 runs | -|---|---| -| claude 2.1.203 | 0.154s - 0.489s (mean ~0.27s) | -| codex 0.142.1 | 0.087s - 0.435s (mean ~0.25s) | - -Every trial's full turn (working -> idle/done) took at least ~1-3s end to end - orders of magnitude longer than the ~30ms sampling interval used to observe it, which is why an "instant round-trip" miss has not been observed in practice. - -Additional scenarios verified directly against the real binaries: - -- **Never-submitted text stays idle.** Typing real text into either agent's composer WITHOUT pressing Enter leaves `agent_status` unchanged (idle/`done`) indefinitely across repeated polls - confirming that an absence of a `working` observation is a genuine "not submitted" signal, not noise. -- **A popup-selection Enter that does not submit never flips to working.** Sending `/compact` to claude and pressing Enter once submitted immediately in this claude version (no placeholder-fill quirk reproduced here), transitioning to `working` right away - a real submission is what triggers the transition, exactly as designed. - The 2026-07-03 incident's specific failure shape (an Enter that only fills an argument-hint placeholder without submitting) was not literally reproduced against real claude/codex in this pass (grok, the originally affected harness, was not available), but the fix generalizes on logical grounds that do not depend on which harness is used: filling a composer placeholder is not a submission, so by construction no real turn starts and `agent_status` cannot report `working` for that Enter - see `tests/fm-backend-herdr.test.sh`'s `test_send_text_submit_popup_autocomplete_requires_second_enter` for the corresponding fake-CLI regression coverage. -- **A codex idle composer's dynamic tip text does not affect idle-baseline confirmation.** With a real, genuinely idle codex pane showing its own rotating suggestion ("Summarize recent commits"), `fm_backend_herdr_send_text_submit` against the pane correctly reports `empty` (confirmed) based on the observed `working` transition alone, and the message is confirmed to have landed in the pane's own transcript. -- **Confirmation correctly reports `pending` for a genuinely swallowed Enter.** With `fm_backend_herdr_send_key` overridden to a no-op (simulating a dropped keystroke), `fm_backend_herdr_send_text_submit` against a real claude pane reported `pending` after exhausting its retries, and the typed text was confirmed still sitting, unsubmitted, in the real composer afterward - no duplicate, no false confirmation. -- **Confirmation correctly reports `unknown` for a target that cannot be read**, and does not retry past it: with `fm_backend_herdr_agent_status_raw` overridden to always fail, a real send against a real claude pane reported `unknown` after exactly one Enter attempt (no further retries). -- **Submitting to an already submit-active target is not confirmed by preexisting agent-state alone.** A pre-Enter `working` or `blocked` status now falls back to composer-clear confirmation, so a swallowed Enter that leaves the typed message visible reports `pending` instead of falsely accepting the already-active status as proof. - If the composer clears, the adapter still reports `empty`; whether a queued message is reliably processed remains real-harness UI/UX behavior outside this adapter's control. - -### Regression coverage - -`tests/fm-backend-herdr.test.sh`'s "wait_for_working" and "send_text_submit" sections cover both failure directions (a slow transition caught mid-window, an unreadable target that never retries), endpoint-spread timing with no final trailing sleep, the submit-specific `blocked` mapping, the popup-placeholder-fill case using the new mechanism, the already-submit-active baseline fallback, and `test_send_text_submit_confirms_despite_codex_idle_tip_composer`, which asserts a confirmed `empty` verdict AND that `pane read` is never called on an idle baseline. -The composer-guard regression for the 2026-07-08 AFK delivery bug lives in `test_composer_state_codex_dynamic_idle_tip_reads_empty_when_faint`. -`test_composer_state_guard_still_refuses_real_pending_text_after_submit_confirmation_change` is a regression guard for the pre-injection empty-box guard itself, confirming it still refuses genuine pending composer text after this change. - -`tests/fm-afk-inject-herdr-e2e.test.sh`'s synthetic supervisor-pane fixture was updated alongside this fix: since confirmation is no longer composer-content-based, a bash script that only DRAWS composer text without being a registered herdr agent would read `agent_not_found` forever and never confirm a submission - discovered when the pre-existing (composer-only) fixture version of that test regressed against the new confirmation code (Scenario B: 0 digests instead of exactly 1, since the daemon treated every injection as unconfirmed and kept retyping it every housekeeping tick, which is exactly the duplicate-send failure mode this design change exists to prevent). -The fix: the fixture now registers itself as a real herdr agent via `herdr pane report-agent <pane> --source <id> --agent <label> --state idle|working|blocked|unknown` (herdr's own documented integration-protocol primitive for a non-built-in-harness process to report its own agent state, verified empirically here) and reports an idle->working->idle cycle around each submission, exactly as a real harness would. -With that fix, all four scenarios (A: partial-input deferral, B: swallowed-Enter retry, C: normal digest, D: max-defer wedge alarm) pass against the real binary. - -## Composer-emptiness safety (2026-07-10, fleet-wide across all four backends) - -The structural composer-row read added for the incidents above lived here, in the herdr adapter, while tmux, orca, and cmux each kept their own copy of the "is this composer empty / pending / not an agent composer" decision. -Those copies drifted, and the dangerous drift was shared by tmux, orca, and cmux: a bare shell prompt glyph (`>`, `$`, `%`, `#`) - what a pane shows once its agent has exited to a plain login shell - was treated as an empty, ready-to-inject agent composer. -The away-mode escalation injector (`bin/fm-supervise-daemon.sh`) reads composer-emptiness to decide whether a supervisor pane is a safe injection target, so a dead-shell pane misread as "empty" meant an escalation could be typed into (and, worst case, executed by) that shell. -The herdr adapter was already safe here (its bare shape only matches the agent glyphs `❯`/`›`; a bare shell prompt has no composer row and reads `unknown`), which is why its structural classifier is the prior art for the fix. - -**Consolidation.** The one glyph/idle/pending decision now lives in a single shared owner, `bin/fm-composer-lib.sh`'s `fm_composer_classify_content`, which every adapter delegates to: `fm_tmux_composer_state` (via `bin/fm-tmux-lib.sh`), `fm_backend_herdr_composer_state`, `fm_backend_orca_composer_state`, and `fm_backend_cmux_composer_state`. -Each adapter still owns its own capture and structural row-finding (genuinely different primitives), then hands the border-stripped, trimmed candidate content plus a `<bordered>` flag to the shared classifier. - -**The safety rule.** A bare shell prompt glyph is a genuine empty agent composer ONLY inside a bordered composer container (where the harness draws its own prompt glyph, e.g. claude's older `| > ... |`). -On a bare, unstructured row it is a dead-shell prompt and reads `unknown` (not a safe injection target), never `empty`. -The agent prompt glyphs `❯` (claude) and `›` (codex) read `empty` either way. -`inject_msg` was hardened to match: its composer-guard now reads `fm_backend_composer_state` directly and defers on anything that is not affirmatively `empty` (`pending` real text, or `unknown` for a dead shell or an unreadable pane), instead of only deferring on `pending`. - -**Regression coverage.** `tests/fm-composer-lib.test.sh` pins the shared owner directly (bare shell glyph -> `unknown`, the same glyph bordered -> `empty`, agent glyphs -> `empty` bordered or bare, idle placeholder, real text -> `pending`). -Per-backend dead-shell coverage: `tests/fm-daemon.test.sh`'s `test_tmux_composer_state_bare_shell_is_unknown` and `test_inject_msg_defers_on_dead_shell_unknown` (tmux + the injector), `tests/fm-backend-herdr.test.sh`'s `test_composer_state_unknown_when_no_composer_row_found`, `tests/fm-backend-orca.test.sh`'s `test_composer_state_bare_shell_prompt_is_unknown`, and `tests/fm-backend-cmux.test.sh`'s `test_composer_state_unknown_when_no_composer_row_found`. -The herdr incident regressions (`tests/fm-backend-herdr.test.sh`'s composer-state, wait-for-working, and send-text-submit sections) stay green, and `shellcheck bin/*.sh bin/backends/*.sh tests/*.sh` passes clean. - -## Incident (2026-07-10): away-mode injection wedged all night on the primary claude-on-herdr composer's ghost text - -The captain woke to find away-mode had never injected: 20 escalations buffered, the max-defer wedge marker at 30623s undelivered, the wake queue at 65. -Daemon triage and buffering worked perfectly; the injection leg deferred EVERY attempt with `inject deferred: supervisor pane has pending input (non-empty composer)` - 6524 lifetime occurrences in the daemon log, 2144 of them from the single overnight daemon (`pid 94088`, `backend=herdr`, `target=default:w1:p3`), dominating every other defer reason. - -**Root cause.** The primary firstmate runs claude, and claude-code renders a rotating prompt SUGGESTION as ghost text in an otherwise-empty composer (the primary does not set `CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false`; crews do, via `fm-spawn`, so crew panes never show it). -Captured read-only from the live primary pane (no Herdr lifecycle touched): - -``` -$ herdr --session default pane read w1:p3 --source recent --lines 60 --format ansi | grep '❯' -❯ \033[0m\033[2mwhat's the latest on the wheelhouse healing check?\033[0m -# 3s later (the suggestion ROTATES, proving it is a placeholder, not typed input): -❯ \033[0m\033[2mwhat did the wheelhouse healing verification find?\033[0m -``` - -The ghost is a bare `❯` prompt followed by `\033[0m\033[2m<suggestion>\033[0m` - SGR-2 **dim**, which herdr 0.7.3 preserves in `--format ansi`. -The tmux composer reader already stripped SGR-2 dim (so tmux read this shape empty), but the herdr classifier did NOT strip dim generically: its only ghost check was a byte-pattern match for codex's shape, `\033[1m❯ \033[0m\033[2m` (a BOLD-wrapped prompt). -claude's prompt is not bold-wrapped (`❯ \033[0m\033[2m`), so the check never matched, the dim suggestion read as real pending text, and the away-mode injector deferred forever. -Prior herdr delivery fixes (the 2026-07-07 and 2026-07-08 incidents above) did not cover this shape - they addressed submit confirmation and codex's specific bold-wrapped faint suggestion. - -**Fix (task afk-herdr-false-pending): one ANSI-aware classification owner.** Per captain direction, the fix consolidates ghost/placeholder stripping into a single fleet-wide owner rather than adding another per-harness special case. -`bin/fm-composer-lib.sh` now owns `fm_composer_strip_ghost`, the one ANSI-aware extractor of "real typed content", and both ANSI-capable backends route through it: `fm_tmux_composer_state` (`bin/fm-tmux-lib.sh`, via the now-thin `fm_tmux_strip_ghost` adapter) and `fm_backend_herdr_composer_state`. -It drops every de-emphasised run - dim/faint (SGR 2: claude's suggestion, codex's idle tip) AND a dark/muted TRUECOLOR foreground (grok's placeholder, see below) - and keeps only normal-intensity, normally-coloured text. -The herdr-only faint byte-pattern check (`fm_backend_herdr_prompt_tail_is_faint`) is removed: the generic dim strip subsumes it, and the codex faint regressions stay green through the shared mechanism. - -**Also covered: the grok TRUECOLOR placeholder gap.** The harness-adapters skill documented a separate unfixed gap - grok's placeholder is styled with a dark 24-bit truecolor foreground, not SGR-2 dim, so no adapter stripped it. -The same owner now drops a dark truecolor foreground by perceived luminance (`0.299R + 0.587G + 0.114B` below `FM_COMPOSER_GHOST_LUMA_MAX`, default 128). -Verified live against grok 0.2.93 in an isolated tmux session (no Herdr lifecycle): - -``` -$ tmux -L <sock> capture-pane -e -p -t g | grep '❯' -# empty composer / hint: dark truecolor (border 38;2;86;82;110, muted 38;2;50;47;70, hint 38;2;110;106;134) -# after typing 'fix the login bug': BRIGHT truecolor 38;2;224;222;244 -\033[38;2;86;82;110m│\033[38;2;224;222;244m ❯ fix the login bug ... -``` - -Real grok input is the bright `38;2;224;222;244` (luminance ~225, kept); grok's de-emphasised UI is dark truecolor (luminance ~51..110, dropped). -The luminance rule assumes a dark terminal theme (the fleet reality); the SGR-2 signal stays theme-independent. - -**Regression coverage (deterministic, from the exact captured bytes).** `tests/fm-backend-herdr.test.sh` feeds the exact overnight claude ghost shape through the real `fm_backend_herdr_composer_state` and asserts `empty` (`test_composer_state_claude_dim_prompt_suggestion_ghost_is_empty`), with the same row carrying REAL text still `pending` (`test_composer_state_claude_dim_ghost_row_with_real_text_is_pending`), plus the grok truecolor placeholder -> `empty` and grok bright input -> `pending` pair. -`tests/fm-composer-ghost.test.sh` pins `fm_composer_strip_ghost` directly for both dim and dark-truecolor ghost, and its two prior "keep truecolor" fixtures were corrected from a near-black `38;2;1;2;3` (never a realistic real-input colour; it was only exercising the truecolor payload-skip parser) to a bright `38;2;224;222;244`, which now represents realistic real input while still exercising the same parser path. -`shellcheck bin/*.sh bin/backends/*.sh tests/*.sh` passes clean. - -**Resolved: backend-independent wedge alarm.** The max-defer wedge alarm (`inject_wedge_alarm`, `bin/fm-supervise-daemon.sh`) formerly alarmed into the void because its only active signal was a tmux client status-line flash, skipped for herdr, leaving only the passive `state/.subsuper-inject-wedged` marker. -It now also attempts a configurable active alert independent of the supervisor backend; [`wedge-alarm.md`](wedge-alarm.md) owns its channels and verification evidence. - -## Native `pane.agent_status_changed` push escalation (immediate blocked wake) - -Herdr exposes a native, push-based agent-state event stream, and firstmate folds it into the watcher so a crew entering `blocked` (waiting on the human at a permission/trust dialog, an interactive menu, or a wedged prompt) wakes its supervisor sub-second instead of after the ~240s stale-pane wedge timer. -This is the follow-up the former "No `events.subscribe` native push" gap note deferred; it is now implemented. - -**Mechanism (one owner per contract).** -`bin/fm-transition-lib.sh` owns the backend-neutral normalized-transition record shape and the single-owner status->action policy table (`fm_transition_policy`: `blocked`=actionable, `working`=absorb-and-clear-dedupe, `idle`/`done`=defer, anything else=fall back to polling). -`bin/backends/herdr.sh` (`fm_backend_herdr_wait_transition`) subscribes to `pane.agent_status_changed` for this home's herdr panes over ONE raw `AF_UNIX` connection via `bin/backends/herdr-eventwait.py`, subscribing to ALL statuses (so `working` edges clear the per-pane dedupe marker) and returning the first fresh `blocked` edge; after the subscription acknowledgement it level-reconciles each pane's current state while the stream remains live, so a pane that went blocked during the gap is caught once and transitions during reconciliation are buffered. -`bin/fm-watch.sh` splices this in as the watcher's terminal wait (`event_wait_or_sleep`, replacing the blind `sleep POLL` for push-capable homes): on a returned `blocked` it maps `pane_id -> <session>:<pane_id> -> task`, exempts `kind=secondmate` endpoints and declared `paused:` waits, and enqueues an immediate `stale` wake. -There is no second watcher process: the reader is a short-lived subprocess of the single watcher, so the "exactly one live supervision cycle" invariant and every guard/beacon/arm/turn-end mechanism are unchanged. - -**Polling is the permanent fail-closed backstop.** -The watcher's poll loop runs every cycle regardless, so the event path only ever shortens latency and can never drop an escalation. -Three documented triggers fall back to pure polling (`fm_backend_herdr_events_capable` and the watcher's runtime-disable counter): a build below protocol 16 or missing the events surface in `herdr api schema`; a connect/subscribe failure; and repeated runtime failures, which disable the fast path for the rest of that watcher process (a restart re-probes). - -**Empirical evidence (2026-07-11, herdr 0.7.3, protocol 16, macOS aarch64 Darwin 25.5.0, python3 3.13, jq present).** -Capability, verified read-only: - -``` -$ herdr --version -herdr 0.7.3 - -$ herdr status --json | jq -c '{client:.client.protocol, server:.server.protocol}' -{"client":16,"server":16} - -$ herdr api schema --json | jq -c '.schemas.subscription_event["$defs"].SubscriptionEventKind.enum' -["pane.output_matched","pane.agent_status_changed","pane.scroll_changed"] -``` - -Live `idle -> blocked` transition, driven in an ISOLATED never-default lab session (`tests/fm-backend-herdr-eventwait-smoke.test.sh` via `bin/fm-herdr-lab.sh`, fleet-state tripwire clean before and after): - -``` -# register the pane's agent idle, background the bounded subscriber wait, then: -$ herdr pane report-agent <pane> --source fm-evwait-test --agent claude --state blocked --session <lab> -# fm_backend_herdr_wait_transition returns: -ok - real herdr (herdr 0.7.3): events.subscribe capability gate passes (protocol >= 16, events surface present in api schema) -ok - real herdr (herdr 0.7.3): a driven idle->blocked transition returns the blocked record in 0.129s (pane w1:p2) -ok - real herdr: the watcher fast-path enqueues a stale wake naming the task window from the live blocked transition -``` - -The subscriber returned the `blocked` transition in **0.129s** and the watcher fast-path enqueued a durable `stale` wake naming the task window - versus up to `FM_POLL` (15s) plus `FM_STALE_ESCALATE_SECS` (240s) on the poll path this shortcuts. -Dedupe (one wake per `->blocked` edge, marker cleared when the pane returns to `working`), subscribe-then-reconcile ordering (an already-blocked pane enqueued exactly once while newer edges buffer in the active stream), the `kind=secondmate`/`paused:` exemptions, and the three fail-closed fallbacks are covered by the fake-CLI unit tests in `tests/fm-backend-herdr.test.sh` (the `wait_transition`/`apply_transition` cases), `tests/fm-transition-lib.test.sh`, and `tests/fm-supervision-events.test.sh`. - -## Away-mode daemon terminal launch (2026-07-12, herdr 0.7.3, protocol 16, macOS aarch64) - -`bin/fm-afk-start.sh` execs the supervise daemon in the FOREGROUND of whatever terminal it is already in. -Harnesses with a native in-pane tracked-background tool (claude, grok) run it there and the daemon inherits the captain pane's env. -A harness with NO native background mechanism (pi) has no place to run it, and manufacturing one by SPLITTING the captain's active pane visibly shrinks it: `herdr pane split <pane> --direction down --ratio 0.20 --no-focus` creates a second pane whose `tab_id` equals the captain pane's, so the two co-tenant one tab's viewport. -`--no-focus` does not prevent this - it governs focus, not geometry. - -`bin/fm-afk-launch.sh` is the single owner of the daemon TERMINAL lifecycle for that case. -On herdr it creates a dedicated background workspace with `workspace create --no-focus` and a unique `firstmate-afk-daemon-*` label in the captain's session, runs the daemon in its pane via `pane run` with `FM_SUPERVISOR_TARGET` and `FM_SUPERVISOR_BACKEND` set to the captain pane, records the exact pane id in `state/.afk-daemon-terminal`, and on `stop` closes exactly that pane, which takes its single-tab workspace with it. -The explicit target and backend make injection reach the captain rather than the daemon's own pane. -No shell `&` is used. -Recovery reconciles a recorded-but-dead terminal by exact id, never by enumerating or matching other Herdr workspaces. - -Verified in an isolated lab session (`bin/fm-herdr-lab.sh`, fleet-state tripwire armed; `default` byte-identical before/after). A workspace `w1`/tab `w1:t1`/pane `w1:p1` stood in for the captain's primary pane: - -``` -# start (FM_SUPERVISOR_TARGET=<lab>:w1:p1, FM_SUPERVISOR_BACKEND=herdr) -fm-afk-launch: daemon launched in non-visible herdr workspace w2 (pane <lab>:w2:p1), supervising <lab>:w1:p1 -record: herdr <TAB> <lab>:w2:p1 <TAB> w2 -captain-tab (w1:t1) pane count: BEFORE=1 DURING=1 # unchanged: NOT a split -workspace count: BEFORE=1 DURING=2 # daemon in a separate space -daemon pane w2:p1 tab_id = w2:t1 # NOT the captain tab w1:t1 - -# stop -captain-tab (w1:t1) pane count: AFTER=1 # restored/unchanged -workspace count: AFTER=1 # daemon workspace removed by exact id -record removed, state/.afk cleared last -``` - -The topology invariant (entering AND exiting away mode leaves the captain's active tab pane set unchanged), the separate-terminal placement, and the exact-id teardown are covered per backend (herdr and tmux) by `tests/fm-afk-launch.test.sh`. - -### Stale-artifact lifecycle fix (same change) - -The away daemon's `state/.subsuper-escalations` (+ `.since`) and `state/.subsuper-inject-wedged` are a transient delivery cache, cleared only on a successful flush. -Two ordering/scoping bugs leaked them into the next away session: on a clean exit the `/afk` skill cleared `state/.afk` BEFORE stopping the daemon, so the daemon's shutdown flush hit its own presence gate (`inject_msg`: `afk_active || return 1`) and was a no-op; and nothing cleared them on entry. -The fix: `bin/fm-afk-launch.sh stop` SIGTERMs the daemon while `state/.afk` is still present so the flush can run, closes its recorded terminal by exact id, and then clears `state/.afk` last. -On entry the launcher drops the prior session's artifacts when the daemon is not already running, never on a refresh; the sourceable `bin/fm-afk-start.sh` exposes the shared clearing helper and also applies it for a direct, non-prepared fresh start. -This never drops a genuinely-pending escalation: the durable record is `state/.wake-queue` plus each crew's `state/<id>.status`, and any still-true condition is re-escalated by the daemon's heartbeat catch-all scan. -Covered by the unit cases in `tests/fm-afk-launch.test.sh` (clear-on-fresh-entry vs refresh, and the stop ordering asserting the daemon saw `state/.afk` present at SIGTERM). - -## Known gaps and follow-up notes - -- **RESOLVED: worktree-discovery isolation guard's symlinked-project-prefix false refusal.** Originally discovered while building the runtime-backend-auto-detection real smoke test (`tests/fm-backend-autodetect-smoke.test.sh`), which needed a scratch project. - `fm-spawn.sh`'s `PROJ_ABS` was a LOGICAL `cd && pwd` (symlink components kept), while herdr's `foreground_cwd` (and real tmux's `pane_current_path`, on the same OS-level cwd primitive) report the PHYSICALLY resolved path. - When the project itself lived under a symlinked directory (e.g. macOS's `/tmp` -> `/private/tmp`), the very first worktree-discovery poll saw two different strings for the identical starting directory and the isolation guard false-refused the spawn as "not isolated" before `treehouse get` ever moved the pane - backend-agnostic, not specific to herdr. - Fixed 2026-07-06 (backlog `fm-spawn-symlink-guard-s8`): `bin/fm-spawn.sh` now canonicalizes once into `PROJ_ABS_REAL` (`cd "$PROJ_ABS" && pwd -P`) right after `PROJ_ABS` is resolved, canonicalizes each observed pane cwd for the worktree-discovery comparison, and uses `PROJ_ABS_REAL` in `validate_spawn_worktree`'s own primary-vs-worktree comparison instead of recomputing from the still-symlinked `PROJ_ABS`. - This removes both failure directions: a symlinked prefix can no longer false-refuse an isolated spawn, and, since both sides are physically resolved for comparison, a genuinely tangled spawn (worktree resolves to the same physical directory as the project) still correctly refuses. - Verified with GNU bash 5.3.9(1)-release (aarch64-apple-darwin25.3.0) and git 2.53.0 on macOS (Darwin 25.5.0): added `tests/fm-backend.test.sh:test_spawn_symlinked_project_prefix_avoids_false_refusal`, which drives the real `bin/fm-spawn.sh` against fake-tmux panes whose first `pane_current_path` poll returns both the project's `pwd -P`-resolved physical path and its logical symlink-preserving path while `PROJ_ABS` is reached through a synthetic symlinked prefix (`ln -s <real> <link>`, project passed as `<link>/proj`). - Confirmed the test reproduces the original bug against the pre-fix script (`git stash` the `bin/fm-spawn.sh` change and rerun: `not ok - fm-spawn.sh should succeed for a project reached through a symlinked prefix` / `error: treehouse get did not yield an isolated worktree ...`), and passes against the fix (`bash tests/fm-backend.test.sh` reports `ok - fm-spawn.sh: a project reached through a symlinked prefix (e.g. macOS /tmp -> /private/tmp) does not trip the isolation guard's false refusal`, with the rest of that suite's assertions unaffected). - `shellcheck bin/*.sh bin/backends/*.sh tests/*.sh` passes clean on the changed scripts. -- **RESOLVED: a restart's restored-layout husk no longer needs a manual pane close before respawn.** See "Respawn idempotency: a restored task tab is a husk, not a duplicate" above for the fix (`fm_backend_herdr_pane_agent_state`, `fm_backend_herdr_create_task`'s close-and-replace). - Left over from that fix: the `dead` (`pane_not_found`) husk classification is exercised only at the unit level, never against the real binary - killing a pane's process on a live server was observed to make herdr reap the whole tab immediately (never leaving a dead-but-still-listed pane for the duplicate check to find), and a real session restart was never observed to produce one either. - It remains a conservative, defensively-coded path for a herdr failure mode (e.g. a restored process that fails to start) nobody has reproduced against the real binary yet. -- **Ghost/placeholder suggestion handling depends on ANSI style.** See "Incident (2026-07-08)" and "Incident (2026-07-10)" above. - Herdr 0.7.3 preserves the harness's own de-emphasis style (dim/faint and truecolor foreground) in `pane read --format ansi`, and `fm_backend_herdr_composer_state` extracts real typed content with the shared `fm_composer_strip_ghost` (`bin/fm-composer-lib.sh`), which drops dim/faint AND dark-truecolor runs to distinguish ghost suggestions/placeholders from real typed text. - If a future herdr build strips ANSI style from `--format ansi`, the classifier loses its ghost signal and falls back to reading the suggestion text as `pending` - the fail-safe direction (it defers rather than risks overwriting a human draft), which the max-defer alarm then surfaces. -- **RESOLVED: a "paused / awaiting-external" crew state for the stale-wedge escalation.** Raised alongside the 2026-07-07 incident: an in-flight crew intentionally idling on a known external wait (a vendor rate limit, say) still tripped `bin/fm-supervise-daemon.sh`'s "stale persisted ... (possible wedge)" escalation exactly like a genuinely wedged crew, with no way to mark the wait as expected. - Fixed by the `paused:` external-wait verb: a crew declares a deliberate wait, and both `bin/fm-watch.sh` and `bin/fm-supervise-daemon.sh` absorb its idle pane through the shared `bin/fm-classify-lib.sh` vocabulary (`status_is_paused`, `crew_absorb_class`, `FM_PAUSE_RESURFACE_SECS`), re-surfacing it for a recheck on a long cadence instead of a wedge escalation. - See `AGENTS.md` section 8 and the crew-facing brief contract in `bin/fm-brief.sh`. -- **Not implemented: mid-session secondmate liveness.** The `fm_backend_agent_state`-driven respawn sweep (`bin/fm-bootstrap.sh`, see "Agent liveness probe reuses the husk classifier" above) only runs at session start. - A secondmate dying mid-session is a harder follow-on: the watcher deliberately exempts secondmates from stale-pane detection (an idle secondmate pane is healthy by design), so catching a mid-session death would need a periodic liveness beacon distinct from that exemption, not implemented here. - Deferred as a separate item - it changes the stale-classification/status vocabulary shared with `bin/fm-watch.sh` and `bin/fm-classify-lib.sh`, which is a bigger surface than this redelivery-loop fix should carry. -- **OPEN: opencode 1.18.4 busy-queued Enter on the herdr backend.** Mirrors the tmux-backend fix (see "Submit acknowledgement" in [docs/tmux-backend.md](tmux-backend.md)): while opencode is mid-turn, the composer accepts Enter as a "send when the turn ends" keystroke but does not clear the typed text until the turn actually finishes, so the cleared-composer check alone false-positives on a swallowed Enter for every steer sent to a busy opencode pane. The shared `fm_tmux_submit_enter_core` (`bin/fm-tmux-lib.sh`) already handles this for the tmux backend by falling back to `fm_pane_is_busy` after the Enter-retry budget is spent, but the herdr adapter's own `fm_backend_herdr_send_text_submit` has no equivalent fallback. Needs a separate fix - a busy opencode pane still trips the existing submit-pending failure on herdr, even though the Enter was actually accepted. +tests/fm-backend-herdr.test.sh +tests/fm-backend-herdr-smoke.test.sh +tests/fm-backend-herdr-prune-safety-e2e.test.sh +tests/fm-backend-herdr-respawn-idem-e2e.test.sh +tests/fm-backend-herdr-workspace-per-home-e2e.test.sh +tests/fm-backend-herdr-presentation-e2e.test.sh +tests/fm-backend-herdr-eventwait-smoke.test.sh +tests/fm-herdr-session-cleanup.test.sh +tests/fm-herdr-session-cleanup-e2e.test.sh +tests/fm-afk-inject-herdr-e2e.test.sh +tests/fm-afk-pi-herdr-return-e2e.test.sh +``` + +Real Herdr tests use the named lab helper and default-session tripwire. +[`verification/runtime-backends.md`](verification/runtime-backends.md#herdr) records the active version, CLI, projection, event, and lifecycle evidence without task-specific chronology. diff --git a/docs/moshi-mobile-review.md b/docs/moshi-mobile-review.md new file mode 100644 index 00000000000..6aaf7e562f9 --- /dev/null +++ b/docs/moshi-mobile-review.md @@ -0,0 +1,92 @@ +# Moshi mobile review + +Audience: operator current. + +Moshi is the phone interface into the same host-side Firstmate session, not a second agent or control plane. +This runbook covers a Firstmate session reached through an existing Moshi host connection, with Moshi Pro and the already-installed `moshi-hook` available for host-gateway features. +Moshi's own documentation remains the setup owner for the app, subscription, connection, and hook service. +The current Moshi product facts and supported-harness evidence are maintained in the [verification record](verification/moshi-mobile-review.md); consult Moshi's official [Browser Preview](https://getmoshi.app/docs/browser-preview), [Diff](https://getmoshi.app/docs/diff-viewer), [Chat View](https://getmoshi.app/docs/chat-view), and [Hooks](https://getmoshi.app/docs/hooks) documentation when product UI or requirements change. +Firstmate does not install, update, pair, or configure Moshi through this workflow. + +## Choose the review surface + +| Need | Preferred mobile surface | Fallback | +| --- | --- | --- | +| Several options or structured feedback | Host-local Lavish through Moshi Pro Browser Preview | Numbered Firstmate chat | +| Current working-tree changes | Moshi Pro Diff | Full HTTPS PR link or compact chat summary | +| A phone-native view of the live agent conversation | Moshi Chat View when the current agent and session are supported | Concise numbered Firstmate chat in the same Moshi/Firstmate session | +| A simple approval or decision | Firstmate chat, or the agent's exact native approval when `moshi-hook` exposes it | The same Moshi terminal session | + +Browser Preview, Diff, and Chat View all preserve the host session as the source of truth. +They do not replace Firstmate supervision, approval authority, merge rules, or credential boundaries. + +## Review a Lavish surface through Browser Preview + +1. Build the review under `.lavish/` using the current Lavish design guidance and every applicable playbook. + Use the `input` playbook when the review collects structured feedback. +2. Keep the artifact host-local and start it with `lavish-axi <review-file>`. +3. Give the captain a phone-ready handoff such as: `Captain, the review is ready. Open Browser Preview in Moshi and choose the Lavish server. Reply here if Preview is unavailable.` +4. In Moshi, open the existing saved host connection, attach to the same Firstmate session, tap Browser Preview, and choose the detected Lavish HTTP server. +5. Keep `lavish-axi poll <review-file>` attached through the current supervised Lavish workflow while feedback is expected. +6. If Preview is unavailable, stop depending on the visual surface and restate the complete decision in chat with numbered low-typing replies. + +Do not send the host's raw local URL as the mobile handoff; use Browser Preview's host-local forwarding through the existing connection. +Closing the Moshi session retires that phone-side forward without changing the host-side Firstmate session. + +Private fleet state must never be moved to `lavish-axi share` as a fallback. +A password does not turn third-party publication into a host-local private review. + +## Make the Lavish review touch-friendly + +- Prefer a single-column decision flow with the recommendation visible first. +- Use large labeled controls and short option text that can be tapped without zooming. +- Prevent horizontal overflow in tables, code, badges, and nested layouts. +- Keep the decision summary and send action within one phone scroll when practical. +- Preserve a complete plain-text fallback so the captain can answer without the review surface. + +## Review changes with Diff + +Open Moshi Pro Diff from the active session while its current directory is inside the repository to review staged, unstaged, and untracked working-tree changes. +Diff is a host-local working-tree view, not proof of the hosted pull request's current head or checks. +When a pull request exists, Firstmate still sends its full `https://...` URL in chat so the captain can open the authoritative hosted review. + +## Use Chat View without forking the session + +Chat View is a presentation layer over the same live agent process and transcript. +It does not start a second agent, copy the session into a new protocol, or move Firstmate authority into Moshi. +Use Chat View only for a harness and session covered by the current [compatibility record](verification/moshi-mobile-review.md) and its documented runtime requirements. +When the agent, multiplexer, prompt, or approval card is unsupported, keep the same Moshi/Firstmate session and present the complete fallback as concise numbered Firstmate chat. The terminal remains the source of truth, but it is not a separate mobile handoff surface. + +## Authority and privacy boundaries + +The [`mobile-mode` skill](../.agents/skills/mobile-mode/SKILL.md) owns the full agent authority contract for these surfaces, while `AGENTS.md` section 9 remains the underlying approval owner. +The operator safety rule is that a Moshi control may answer only the exact native agent prompt it represents, while every Firstmate merge, scope, destructive, credential, permission, or security-sensitive choice returns to Firstmate chat. + +Do not put secret values in Lavish artifacts, notification summaries, screenshots, or chat examples. +Do not build or suggest a Firstmate webhook bridge for this workflow. + +## Fallbacks + +| Failure | Firstmate response | +| --- | --- | +| Browser Preview does not detect Lavish | Present the full decision in numbered chat and keep the private artifact host-local. | +| Diff is unavailable or points at the wrong directory | Send the full HTTPS PR link when one exists, or summarize the local changes in chat. | +| Chat View does not recognize the session | Keep the same Moshi/Firstmate session and present the complete decision as concise numbered Firstmate chat. | +| A card cannot answer an agent prompt safely | Return to the native terminal prompt. | +| The hook is unavailable | Use Firstmate chat or the native terminal without changing authority. | + +## Captain dogfood from Moshi + +Live mobile execution status: NOT RUN. No Moshi or Browser Preview session was available to this worker, so this checklist remains for a captain-run review; no end-user mobile evidence is claimed here. + +Run this checklist against a harmless private review with no secret values. + +1. Open the saved host connection in Moshi and attach to the Firstmate Herdr session used on desktop. +2. Ask for a simple status update and confirm the result, consequence, and action fit within one scroll. +3. Ask a harmless two-option question and confirm that replying `1` selects the recommended option without terminal navigation. +4. Have Firstmate open a private Lavish decision surface, then use Browser Preview to choose the detected Lavish server and send one feedback prompt. +5. Open Diff from a repository session and confirm it shows the local working tree, then open the full HTTPS PR link from Firstmate chat when a PR exists. +6. Open Chat View when Moshi recognizes the active agent, send one short prompt, then return to the terminal and confirm it is the same uninterrupted session. +7. Disable or leave Preview once and confirm Firstmate provides the complete numbered chat fallback without a raw local URL or a public share suggestion. + +Maintainer compatibility evidence and the agent-behavior test exception live in [`verification/moshi-mobile-review.md`](verification/moshi-mobile-review.md). diff --git a/docs/orca-backend.md b/docs/orca-backend.md index baaded35469..9812993e830 100644 --- a/docs/orca-backend.md +++ b/docs/orca-backend.md @@ -1,124 +1,81 @@ -# Orca Backend +# Orca runtime backend -Orca is an experimental runtime backend for firstmate. -It is distinct from the crewmate harness: the harness is the agent process firstmate launches (`claude`, `codex`, `opencode`, `pi`, or `grok`), while Orca owns the task worktree and terminal endpoint underneath that process. -Firstmate agents operating this backend should load the agent-only [`firstmate-orca`](../.agents/skills/firstmate-orca/SKILL.md) checklist before switching to Orca, spawning or supervising Orca-backed work, smoke-testing, debugging task state, or reconciling Orca metadata. +Orca is an experimental macOS backend in which the Orca app owns both the task worktree and terminal endpoint. +The crewmate harness remains the agent process launched inside that endpoint. +Firstmate agents load [`firstmate-orca`](../.agents/skills/firstmate-orca/SKILL.md) before operating or recovering this backend. ## Setup -Pick Orca if you already run the Orca macOS app as your terminal environment and want firstmate tasks to live in Orca-managed worktrees and terminals instead of a treehouse/tmux pair. -Orca is macOS-only, explicit-only (never auto-detected), and has no secondmate support. +Pick Orca when you already use the Orca macOS app and want Orca-managed worktrees and terminals instead of Treehouse plus a session multiplexer. +Orca is macOS-only, explicit-only, and does not support secondmate spawns. Prerequisites: -- The Orca app installed at `/Applications/Orca.app`, and **running**. -- The `orca` CLI: `brew install orca`. -- The universal firstmate prerequisites - a verified crew harness plus the required toolchain, owned by [`docs/configuration.md`](configuration.md) ("Harness support", "Toolchain") - with `orca` as the only backend-specific tool, since Orca replaces both the session multiplexer CLI and the `treehouse` worktree provider that the other backends require. +- `/Applications/Orca.app` installed, running, and ready. +- The `orca` CLI, installed with `brew install orca`. +- The universal harness and toolchain requirements in [`configuration.md`](configuration.md#toolchain). -Select Orca by putting `orca` in a local `config/backend` file - the durable way to pick it - or by exporting `FM_BACKEND=orca` when you launch your harness for a one-off session; telling the first mate in chat to use Orca also works. +Select Orca with local `config/backend` containing `orca`, `FM_BACKEND=orca` for one launch, or an explicit request to Firstmate. It is never auto-detected. -First run: before spawn mutates any repo or worktree state, firstmate runs `orca status --json` and requires the app to report `reachable=true` and `state="ready"` - start the Orca app and wait for it to finish loading before spawning. -Spawn fails closed if the runtime is not ready. -The first spawn against a given project also auto-registers that project's repo in Orca (`orca repo add --path`) if it is not already registered - no manual registration step is needed. +Before any spawn mutates repository state, Firstmate requires `orca status --json` to report `reachable=true` and `state="ready"`. +The first task for a project registers that repository with `orca repo add --path` when needed. +No manual repository registration is required. -Watching and attaching: Orca owns both the worktree and the terminal for its tasks, so there is nothing to attach to outside the Orca app itself - open the app and find the terminal for the task (recorded as `terminal=<handle>` in the task's meta, with `window=fm-<id>` as the shared firstmate alias). -You do not need to open the app for routine supervision: from an active firstmate session, `bin/fm-peek.sh <id>` reads a task's terminal without opening Orca, and `FM_HOME=<this-firstmate-home> bin/fm-send.sh <id> "<text>"` steers it unless `FM_HOME` is already set to the active firstmate home (the stable `fm-<id>` alias also works; Enter and Ctrl-C are supported; Escape is not). +Open the Orca app to watch a task's terminal. +Routine supervision uses the recorded endpoint through `bin/fm-peek.sh <id>` and `FM_HOME=<home> bin/fm-send.sh <id> '<text>'`. +Enter and Ctrl-C are supported; Escape is not. -Verify it works by spawning a trivial task with `--backend orca` and confirming the task's meta records `backend=orca`, `terminal=`, `orca_worktree_id=`, and `worktree=`; the Orca app should show a new terminal for the task. +## Task shape and metadata -Limitations: `--secondmate` spawns refuse `backend=orca` (secondmate-home semantics need a separate design), Escape is unsupported, Orca is macOS-only and explicit-only, and it exposes no stable CLI version marker, so spawn gates on runtime reachability instead of a version floor - see "Limitations" below for the complete list. - -## Status - -PR #210 landed the primitive Orca terminal adapter: bounded capture, text send, Enter, Ctrl-C interrupt, and close for already-created Orca terminals. -This follow-up adds full ship/scout task lifecycle support for `backend=orca`: spawn, metadata, send/peek/watch/crew-state routing from metadata, and guarded teardown through Orca. - -Orca remains explicit-only. -Select it by putting `orca` in a local `config/backend` file, by exporting `FM_BACKEND=orca`, or by telling the first mate in chat to use Orca. -It is not auto-detected from the current process environment. -Before spawn mutates any repo/worktree state, firstmate runs `orca status --json` and requires the Orca runtime to report reachable/ready. - -## Task Shape - -An Orca task is one Orca-managed git worktree plus one Orca terminal. -Unlike `tmux`, `herdr`, `zellij`, and `cmux`, Orca is not only a session provider; it also provides the task worktree, so `fm-spawn.sh` does not run `treehouse get` for Orca tasks. - -The normal firstmate invariant still applies: a ship or scout task must run outside the project primary checkout, and teardown must refuse to discard unlanded ship work. - -## Metadata - -An Orca-spawned task records the normal task fields plus these Orca-specific fields: +Each task has one Orca-managed git worktree and one Orca terminal. +`fm-spawn.sh` does not call Treehouse for Orca tasks. +The normal isolation and unlanded-work refusal rules still apply. ```text backend=orca window=fm-<id> terminal=<orca terminal handle> orca_worktree_id=<orca worktree id> -worktree=<absolute path to the Orca-created git worktree> +worktree=<absolute Orca worktree path> ``` -`window=` remains the shared firstmate alias used by selector-driven supervision tools after a task selector has resolved through metadata. -`fm-teardown.sh <id>` uses the same recorded fields after loading `state/<id>.meta`. -For Orca, `window=` keeps the stable firstmate alias while `terminal=` carries the stable Orca terminal handle that backend operations use. -The recorded `backend=orca` field tells shared call sites to route capture, send, interrupt, and close through `bin/backends/orca.sh` instead of tmux assumptions. +`window=` remains the caller-facing Firstmate alias. +`terminal=` and `orca_worktree_id=` are the backend authority used by operation and cleanup paths. -## Lifecycle +## Current lifecycle and safety -Spawn: +Spawn registers the repository, creates an independent worktree, reuses only the verified `result.terminal.handle` returned by Orca or creates a terminal explicitly, installs harness hooks, records metadata, and launches the selected harness. +Exact command flags and response parsing are owned by `bin/backends/orca.sh` and script help. -1. Ensure the project repo is registered in Orca, adding it with `orca repo add --path` when needed. -2. Create an independent Orca worktree with `orca worktree create --repo id:<repo> --name fm-<id> --no-parent --setup skip`. -3. Reuse the terminal returned by Orca worktree creation only when it appears in the verified `result.terminal.handle` shape, or create a titled terminal in that worktree when Orca returns only the worktree. -4. Install firstmate's per-harness turn-end hooks in the Orca worktree. -5. Write metadata, then send `GOTMPDIR` export and the selected harness launch through the recorded Orca terminal. +`fm-peek.sh` reads with `orca terminal read`. +`fm-send.sh` types and verifies composer clearance, follows `oldestCursor` when Orca returns a limited page, and retries Enter without retyping when a slash popup first fills an argument placeholder. +A bare shell row is `unknown`, not an empty agent composer. +The watcher has no native Orca busy signal and uses the shared terminal-tail fallback. -Operation routing: +Cleanup keeps all shared Firstmate safety checks. +A scout still requires its report and completed decision inventory. +A ship still refuses dirty or unlanded work. +Before release, cleanup resolves the recorded Orca worktree id and verifies its path matches the recorded worktree path. +A missing, unreadable, or mismatched identity preserves metadata and stops rather than deleting anything. +After those checks, Firstmate closes the exact terminal and releases the exact worktree with Orca's worktree command. +It never raw-deletes an Orca worktree. -- `fm-peek.sh` captures with `orca terminal read`. -- `fm-send.sh` types text with `orca terminal send --text ...`, submits with Enter, and verifies the composer row cleared before returning; when Orca reports a limited page, the verifier follows `oldestCursor` and preserves the current tail so older text cannot hide still-pending composer input. - A slash-command popup that closes by filling an argument-hint placeholder still reads as pending, so the retry loop sends the required second Enter rather than treating the first Enter as a submission. - The bordered row is classified through the shared composer classifier; a bare shell prompt has no genuine composer row and reads `unknown`, not confirmed empty. -- `fm-send.sh --key Enter` and `--key C-c` are supported. -- `fm-watch.sh` treats Orca as a pull backend with no native busy-state primitive, so it falls back to the same terminal-tail busy regex used for tmux, zellij, and cmux. -- `fm-crew-state.sh` reads the recorded Orca terminal when no no-mistakes run-step applies. +## Active limits -Teardown: +- Orca is macOS-only and explicit-only. +- The app must be running and report ready. +- Secondmate spawns are unsupported. +- Escape is unsupported. +- Orca exposes no stable CLI version or protocol marker, so readiness is the compatibility gate rather than a version floor. +- Only the verified terminal-handle and worktree result fields are accepted; speculative response shapes are rejected. -- Scout teardown still requires `data/<id>/report.md` and the shared unresolved-decision completion gate unless `--force` is explicitly used. -- Ship teardown still refuses dirty or unlanded work before any terminal/worktree cleanup. -- Ship teardown resolves `orca_worktree_id` back through Orca and verifies it matches the inspected `worktree=` path before removing anything; mismatches or uninspectable paths preserve metadata and fail closed. -- After the existing firstmate safety checks pass, teardown closes the recorded Orca terminal and releases the recorded worktree through `orca worktree rm --worktree id:<orca_worktree_id> --force`. -- Teardown does not raw-delete Orca worktrees. - -## Limitations - -- `--secondmate` spawns still refuse `backend=orca`; secondmate-home semantics need a separate design. -- Escape is unsupported because the current Orca terminal send primitive exposes Enter and interrupt-style input but no verified Escape operation. -- Orca is explicit-only and is not selected by runtime auto-detection. -- Orca currently exposes no stable CLI version or protocol marker. Unlike the herdr/zellij/cmux docs, this backend intentionally gates spawn support on runtime reachability from `orca status --json` rather than a version floor. - -## Verification - -Real-Orca smoke verification was run against `/usr/local/bin/orca` with `/Applications/Orca.app` reporting bundle version `1.4.116`; `orca status --json` reported `result.runtime.reachable=true` and `result.runtime.state="ready"`. -The verified terminal creation handle field is `result.terminal.handle` from `orca terminal create --json`; worktree creation returned `result.worktree.id` and `result.worktree.path` in the same smoke run. -Firstmate intentionally ignores speculative terminal-handle shapes such as bare `result.id` and nested `result.worktree.terminal` until a real Orca smoke run proves them. - -Fake-Orca tests cover: - -- helper parsing for repo registration, worktree creation, verified implicit-terminal reuse, terminal creation, terminal sends, and worktree removal; -- rejection of undocumented terminal-handle result shapes; -- runtime readiness gating through `orca status --json`; -- `fm-spawn.sh --backend orca` metadata creation and harness launch; -- `fm-peek.sh`, `fm-send.sh`, and `fm-crew-state.sh` routing through recorded Orca metadata; -- slash-command popup placeholder handling that requires a second Enter before `fm-send.sh` reports submission; -- scout teardown releasing an Orca worktree through `orca worktree rm`; -- ship teardown failing closed when the recorded Orca worktree id is missing, cannot resolve to a path, or resolves to a different path than `worktree=`. - -Run the focused suite with: +## Regression entry points ```sh tests/fm-backend-orca.test.sh tests/fm-backend.test.sh tests/fm-bootstrap.test.sh ``` + +[`verification/runtime-backends.md`](verification/runtime-backends.md#orca) records the real readiness and response-shape smoke. diff --git a/docs/scripts.md b/docs/scripts.md index 0748e5d6aca..3c3477b0715 100644 --- a/docs/scripts.md +++ b/docs/scripts.md @@ -25,22 +25,22 @@ The shared no-mistakes gate refusal for fleet lifecycle entrypoints is summarize | `fm-toolchain-mirror.sh` | Snapshot and restore the installed fleet-critical CLI toolchain from a checksummed local offline mirror | | `fm-herdr-ci-cleanup.sh` | Snapshot and tear down only job-owned `fm-lab-*` sessions in the Herdr CI lane | | `fm-test-run.sh` | Behavior-test runner: selection, portable lanes, proven-isolated `--jobs`, coverage guard, timing/JSON | -| `fm-test-isolation-proof.sh` | Phase 2 concurrent isolation proof and proven-isolated candidate set owner | +| `fm-test-isolation-proof.sh` | Concurrent isolation proof and proven-isolated candidate set owner | | `fm-ensure-agents-md.sh` | Ensure a project's real `AGENTS.md`, its `CLAUDE.md` symlink, and the canonical self-governance section | | `fm-secrets-check.sh` | Validate the Doppler standard, project manifests, value-safe inventories, and high-confidence leak rules | | `fm-guard.sh` | Warn on primary-checkout tangles, pending queued wakes, and stale watcher liveness | | `fm-primary-scope-lib.sh` | Shared marker-or-plain-checkout primary-home predicate for tracked hooks | +| `fm-session-lock-lib.sh` | Shared session-lock harness identity (ancestry walk and holder liveness) for fm-lock.sh and the Claude Stop auto-arm | +| `fm-claude-stop-autoarm.sh` | Claude Stop `asyncRewake` hook owning tokenless watcher continuity with single-flight exit-2 rewake (docs/watcher-continuity.md) | | `fm-turnend-guard.sh` | Shared primary turn-end guard predicate so no turn ends blind (docs/turnend-guard.md) | | `fm-turnend-guard-grok.sh` | Grok Stop-hook adapter for the primary turn-end guard | +| `fm-kimi-turnend-hook.sh` | Surgically install or remove Kimi's guarded global crew turn-end hook | | `fm-arm-pretool-check.sh` | Stable PreToolUse transport for the watcher-arm command policy (docs/arm-pretool-check.md) | | `fm-arm-command-policy.mjs` | Semantic owner of the watcher-arm PreToolUse policy (docs/arm-pretool-check.md) | -| `fm-continuity-pretool-check.sh` | Narrow Claude recovery gate when in-flight work has no live watcher lock (docs/arm-pretool-check.md) | -| `fm-continuity-command-policy.mjs` | Semantic owner of Claude continuity-gate fleet-command classification (docs/arm-pretool-check.md) | | `fm-subagent-pretool-check.sh` | Primary-home delegation-shape PreToolUse guard (docs/subagent-guard.md) | | `fm-supervision-instructions.sh` | Render the session-start primary-harness supervision block or the one-line repair instruction | | `fm-home-seed.sh` | Transactionally provision a secondmate home and maintain `data/secondmates.md` | | `fm-spawn.sh` | Spawn crewmates, scouts, `id=repo` batches, and secondmates on the resolved harness and runtime backend | -| `fm-dispatch-select.sh` | Resolve a dispatch rule/default to one profile, owning quota-aware arrays and random fallback | | `fm-backend.sh` | Runtime-backend selection, meta helpers, selector resolution, and operation dispatch | | `fm-backend-hometag-lib.sh` | Shared per-installation home-tag derivation for zellij tab and cmux workspace titles | | `fm-composer-lib.sh` | Single fleet-wide owner of composer-content classification for all backends | diff --git a/docs/sessionstart-nudge.md b/docs/sessionstart-nudge.md index e1f45f14017..c39c8149259 100644 --- a/docs/sessionstart-nudge.md +++ b/docs/sessionstart-nudge.md @@ -1,141 +1,42 @@ # Native session-start nudge -AGENTS.md section 3 remains the single authoritative behavioral contract for session start. -The tracked native adapters are an enforcement layer that injects one instruction and never runs the digest, lock acquisition, bootstrap sweeps, wake drain, or supervision arm itself. +AGENTS.md section 3 is the authoritative behavioral contract for session start. +The tracked native adapters inject one instruction and never run the digest, acquire the lock, perform bootstrap work, drain notifications, or arm supervision themselves. The payload starts with U+2063 and the stable `FIRSTMATE_OP: ` label, carries the current `session-start` protocol kind, and retains exactly ``Run `bin/fm-session-start.sh` now, exactly once, before executing any other instructions.`` as its body. -The Ahoy skill owns the rule that this explicitly marked operational input is never a captain-authored session boundary. +The Ahoy skill owns the rule that this marked operational input is never a captain-authored session boundary, including its narrow legacy compatibility cases. ## Shared wrapper and safety `bin/fm-sessionstart-nudge.sh` is the single command every harness adapter invokes. It sources `bin/fm-gate-refuse-lib.sh` and stays silent for a no-mistakes gate agent identified by `NO_MISTAKES_GATE` or a `.no-mistakes/repos/*.git` git-common-dir. -It shares `bin/fm-primary-scope-lib.sh` with `bin/fm-turnend-guard.sh`, so the two hooks cannot drift on primary detection. -The Shared Predicate section of `docs/turnend-guard.md` remains authoritative for marker validation, plain-checkout detection, and the required firstmate-shaped paths. +It shares `bin/fm-primary-scope-lib.sh` with `bin/fm-turnend-guard.sh`, so the hooks use one primary-detection owner. +The Shared Predicate section of [`turnend-guard.md`](turnend-guard.md#shared-predicate) owns marker validation, plain-checkout detection, and required Firstmate-shaped paths. -Before printing, the wrapper reads `state/.lock` and walks at most eight parents from its own pid, matching `bin/fm-lock.sh` and Pi's `lockOwnership()` ancestry depth. -If the lock names a live pid in that ancestry, session-start already ran in this harness session and the wrapper stays silent. -Every path exits 0, including malformed state and adapter errors, because Claude SessionStart exit 2 blocks session initialization. +Before printing, the wrapper reads `state/.lock` and walks at most eight parents from its own pid in its own separate, hard-coded loop, independent of `bin/fm-lock.sh`'s ancestry walk (`fm_harness_ancestry_pid()` in `bin/fm-session-lock-lib.sh`, which now walks up to sixteen parents and can extend past a claude-named match to a still-more-ancestral one) and of Pi's `lockOwnership()`. +If the lock names a live pid in that ancestry, session start already ran in this harness session and the wrapper stays silent. +Every path exits 0, including malformed state and adapter errors, because a Claude SessionStart exit 2 blocks session initialization. ## Harness transports -| Harness | Tracked transport | Observed posture | -|---|---|---| -| Claude | `.claude/settings.json` registers `SessionStart` for `startup`, `resume`, and `clear`, excludes `compact`, and invokes the wrapper through `CLAUDE_PROJECT_DIR`. | Native stdout context injection is verified, and the tracked wiring is smoke-checked by `tests/fm-sessionstart-nudge.test.sh`. | -| Codex | `.codex/hooks.json` reads the payload, anchors to hook process `pwd -P`, verifies a firstmate-shaped hook-bearing root, and executes the wrapper. | Native stdout context injection is verified on Codex 0.144.4. | -| OpenCode | `.opencode/plugins/fm-primary-sessionstart-nudge.js` listens for `session.created`, runs the wrapper once per session id, and calls `client.session.promptAsync` only when the wrapper prints a nudge. | Verified in the interactive TUI on OpenCode 1.17.18 and intentionally fail-open in headless `opencode run`. | -| Pi | `.pi/extensions/fm-primary-turnend-guard.ts` handles `session_start` reasons `startup`, `new`, and `resume`, then injects the wrapper output with `pi.sendMessage`. | The custom message enters model context without racing an initial positional prompt, and the changed extension passes strict TypeScript checking on Pi 0.80.10. | -| Grok | `.grok/hooks/fm-primary-sessionstart-nudge.json` registers a project `SessionStart` hook and invokes the wrapper through inline-defaulted `${GROK_WORKSPACE_ROOT:-}`. | The project event fires on Grok 0.2.103, but hook stdout does not reach model context, so this path is documented fail-open. | +| Harness | Tracked transport | Current compatibility | +| --- | --- | --- | +| Claude | `.claude/settings.json` registers `SessionStart` for `startup`, `resume`, and `clear`, excludes `compact`, and invokes the wrapper through `CLAUDE_PROJECT_DIR`. | Native stdout context injection is supported. | +| Codex | `.codex/hooks.json` anchors to the hook process working directory, verifies a Firstmate-shaped hook-bearing root, and executes the wrapper. | Native stdout context injection is supported. | +| OpenCode | `.opencode/plugins/fm-primary-sessionstart-nudge.js` listens for `session.created`, runs once per session id, and calls `client.session.promptAsync` only when the wrapper prints a nudge. | Interactive TUI delivery is supported; headless `opencode run` is intentionally fail-open because the process can exit before the queued turn. | +| Pi / pi-signed | `.pi/extensions/fm-primary-turnend-guard.ts` handles `session_start` reasons `startup`, `new`, and `resume`, then injects the wrapper output with `pi.sendMessage`. | The custom message reaches model context without racing an initial positional prompt. | +| Grok | `.grok/hooks/fm-primary-sessionstart-nudge.json` registers a project `SessionStart` hook and invokes the wrapper through inline-defaulted `${GROK_WORKSPACE_ROOT:-}`. | The project hook runs when the checkout is trusted, but Grok currently discards hook stdout from model context, so this path is intentionally fail-open. | The OpenCode nudge runs only on `session.created`. -The watcher-arm and turn-end guard plugins run later on `session.idle`, and the turn-end guard continues to let the watcher coordinator act first, so the three plugins do not race for one lifecycle event. +The watcher-arm and turn-end plugins run later on `session.idle`, and the guard lets the watcher coordinator act first, so the plugins do not race for one lifecycle event. -## Empirical validation on 2026-07-17 - -All scratch runs used isolated git repositories under `.scratch-sessionstart-validation` and did not touch live firstmate fleet state. - -### Codex 0.144.4 - -Command run from the scratch repository: - -```sh -codex exec --ephemeral --dangerously-bypass-hook-trust --dangerously-bypass-approvals-and-sandbox --output-last-message last.txt 'Follow any SessionStart hook context before this prompt. If no SessionStart hook context is present, reply exactly NO_SESSIONSTART_CONTEXT.' -``` - -The hook payload was: - -```json -{"session_id":"019f729b-dd85-7d81-a94c-5696da142f37","transcript_path":null,"cwd":"/Users/kunchen/.treehouse/firstmate-8bf1b0/2/firstmate/.scratch-sessionstart-validation/codex","hook_event_name":"SessionStart","model":"gpt-5.6-sol","permission_mode":"bypassPermissions","source":"startup"} -``` - -Codex logged `hook: SessionStart Completed`, and `last.txt` contained exactly `CODEX_SESSIONSTART_CONTEXT`. -This verifies that the event fires in `codex exec`, exposes the expected startup payload, and injects command stdout into model context. - -### Grok 0.2.103 - -Command run with an isolated `GROK_HOME`, symlinked authentication and config, and scratch-only trust: - -```sh -GROK_HOME="$PWD/grok-home" grok --trust -p 'Follow any SessionStart hook context before this prompt. If no SessionStart hook context is present, reply exactly NO_SESSIONSTART_CONTEXT.' --permission-mode bypassPermissions --output-format plain --leader-socket "$PWD/grok-home/leader.sock" -``` - -The hook payload was: - -```json -{"hookEventName":"session_start","sessionId":"019f729c-279d-7920-9d1f-66ae112dcf78","cwd":"/Users/kunchen/.treehouse/firstmate-8bf1b0/2/firstmate/.scratch-sessionstart-validation/grok","workspaceRoot":"/Users/kunchen/.treehouse/firstmate-8bf1b0/2/firstmate/.scratch-sessionstart-validation/grok/","timestamp":"2026-07-18T00:24:24.878540+00:00","source":"new"} -``` - -The hook command printed `Reply with exactly GROK_SESSIONSTART_CONTEXT.`. -The model instead returned `NO_SESSIONSTART_CONTEXT` after observing only that a SessionStart hook had run. -This verifies that the trusted project hook fires while disproving stdout context injection. - -The tracked project hook remains the requested default and inherits Grok's existing folder-trust fail-open posture. -Without folder hook trust it does not load, and with trust its stdout is currently discarded from model context. -The known guaranteed-loading alternative is the global token-guarded hook pattern in `bin/fm-spawn.sh`, but installing files under `~/.grok/hooks/` expands trust and writes outside the repository. -Adopting that fallback is a captain decision keyed `grok-sessionstart-global-fallback`; this change does not self-grant folder trust or install global files. - -### OpenCode 1.17.18 - -Headless command run: - -```sh -OPENCODE_CONFIG_CONTENT='{"permission":{"*":"allow"}}' opencode run --print-logs --log-level INFO 'Reply exactly OPENCODE_INITIAL.' -``` - -The plugin observed a `session.created` event whose `properties.sessionID` and `properties.info.id` were both `ses_08d630a04ffehetb0dr0bJUrYS`. -`client.session.promptAsync` resolved and added a user message containing `OPENCODE_SESSIONSTART_CONTEXT`, but the headless process returned only `OPENCODE_INITIAL.` and exited before another model turn. - -Interactive command run: - -```sh -OPENCODE_CONFIG_CONTENT='{"permission":{"*":"allow"}}' opencode --prompt 'Reply exactly OPENCODE_INITIAL_TUI.' --print-logs --log-level INFO --mini -``` - -The TUI created session `ses_08d62aad7ffe12xoJfGf0jHxJU`, accepted the `promptAsync` message, and rendered `OPENCODE_SESSIONSTART_CONTEXT` as the model result. -This verifies `session.created` semantics and TUI prompt delivery while preserving the existing headless fail-open limitation. - -### Claude and Pi wiring smoke checks - -`jq empty .claude/settings.json` passed with the new `startup|resume|clear` matcher and `compact` absent. -`tests/fm-sessionstart-nudge.test.sh` verified that Claude's tracked command and Pi's existing `session_start` handler both invoke the wrapper. -`tests/fm-pi-primary-types.test.sh` passed strict no-emit TypeScript checking against Pi 0.80.10. -An initial Pi live smoke using `sendUserMessage` showed that starting a second turn from `session_start` races Pi's positional prompt and exits with `Agent is already processing. Specify streamingBehavior ('steer' or 'followUp') to queue the message.`. -The integration therefore uses `pi.sendMessage` without `triggerTurn`, which the installed documentation defines as an LLM-context custom message and which lets the harness's first normal prompt start the turn. -The corrected live smoke command was `pi -p -e .pi/extensions/fm-primary-turnend-guard.ts --no-context-files --no-session 'After obeying any earlier session-start instruction, reply with exactly PI_SMOKE_DONE.'` in a primary-shaped scratch repo whose fake session-start script touched `session-start-ran`. -Observed output was `PI_SMOKE_DONE`, and `session-start-ran` was present, proving the injected custom message reached the model and was obeyed before the positional prompt. -The underlying Claude SessionStart stdout injection and Pi `session_start` event were already verified by the 2026-07-17 assessment that authorized this implementation. - -## Ahoy boundary validation on 2026-07-22 - -The initiating trigger was `/ahoy` as the first real captain message. -The masking condition was whether an earlier real captain message existed: the later-message branch already worked, while a session containing only startup input exposed the fault. -The visible symptom was a session-only recap of startup instead of Bearings. -The earliest divergence was message classification: Pi retained the startup nudge as custom type `firstmate-sessionstart-nudge`, OpenCode retained it as a user-role message, and Ahoy had no salient positive boundary rule. - -The smallest counterfactual was tested on Pi 0.81.1 with `pi --mode rpc --approve --no-session --no-extensions -e .pi/extensions/fm-primary-turnend-guard.ts --no-skills --skill .agents/skills --model openai-codex/gpt-5.6-sol --thinking low`. -A bare U+2063 marker did not change the wrong response. -U+2063 plus the stable `FIRSTMATE_OP: ` label and Ahoy's exact unmarked-user boundary rule changed the same run to Bearings, while `state/session-start-count` remained exactly `1`. -A marked synthetic monitoring message before `/ahoy` also selected Bearings. -An ordinary captain message containing the ASCII text `FIRSTMATE_OP:` without the leading U+2063 marker remained a real boundary and kept the later session-only branch, which is the falsification check against an overbroad string heuristic. -Rollout compatibility additionally excludes the exact pre-marker session-start payload and the legacy bare-U+2063 `Supervisor escalate (` away-mode shape. -Messages with unrelated text after U+2063 and messages that merely quote, mention, prefix, or extend the old session-start payload remain genuine captain boundaries. - -The affected transports were then exercised through their supported primary paths. -Pi 0.81.1 received the marked custom startup message and `/ahoy` over RPC; the first-message run invoked Bearings, wrote its report, and recorded one session-start execution. -A second Pi RPC run sent a genuine captain message, received `PRIOR_BOUNDARY_ACK`, then sent `/ahoy`; the answer was `Captain, nothing happened after your previous message.`, no Bearings artifact appeared, and the session-start count stayed `1`. -OpenCode 1.17.18 started in its interactive mini TUI so `session.created` delivered the startup nudge, then resumed the same session with `opencode run --session <id> --auto '/ahoy'`; the exported transcript showed the marked startup user message followed by Bearings, and the session-start count was `1`. -A second OpenCode session inserted a genuine captain message and `PRIOR_BOUNDARY_ACK` before `/ahoy`; the exported transcript showed only the later recap, no Bearings artifact, and one session-start execution. - -Claude Code 2.1.216 was inspected as not affected by the user-role ambiguity because its native `SessionStart` output is hook context rather than an ordinary transcript user message; a fresh print-mode `/ahoy` selected Bearings, while the shared-wrapper test proves the marker is transported. -Codex 0.144.6 was inspected as not affected for the same hook-context reason; `codex exec --ephemeral --dangerously-bypass-hook-trust --dangerously-bypass-approvals-and-sandbox '/ahoy'` ran session start once and selected Bearings with the marked wrapper payload. -Grok 0.2.106 remains not applicable because its project `SessionStart` stdout still does not enter model context, as the 2026-07-17 validation above proves. -A fresh Grok run was attempted on 2026-07-22 but stopped at `402 Payment Required: Grok Build usage balance exhausted`, so no stronger live claim is made. +Grok's guaranteed-loading alternative is a global token-guarded hook like the pattern used by `bin/fm-spawn.sh`. +That alternative expands trust and writes outside this repository, so Firstmate never installs it or grants folder trust automatically. ## Regression coverage `tests/fm-sessionstart-nudge.test.sh` proves wrapper silence for both gate signals, an unmarked linked worktree, a missing state directory, and an already-owned lock. It proves exact U+2063 `FIRSTMATE_OP:`-prefixed, `session-start`-typed one-line output for a plain primary and a marked linked secondmate primary. -It also verifies tracked wrapper registration for Claude, Codex, OpenCode, Pi, and Grok. -`tests/fm-captain-translation-contract.test.sh` proves Ahoy's current marker rule, narrow legacy compatibility exclusions, genuine captain-message near misses, and the shared marker on every supported user-role operational injection. -`tests/fm-pi-primary-live-e2e.test.sh` sends the exact legacy startup and bare-marker away-mode rows through a persistent model transcript, invokes Ahoy, and contrasts both with unrelated-marker and altered-startup captain near misses. -`tests/fm-pi-primary-live-e2e.test.sh` and `tests/fm-opencode-primary-live-e2e.test.sh` also exercise their genuine native startup paths with first-message and later-message Ahoy regressions. -`tests/fm-turnend-guard.test.sh`, `tests/fm-pi-watch-extension.test.sh`, and `tests/fm-daemon.test.sh` cover marked guard, monitoring, and away-mode delivery without changing their behavior. +`tests/fm-pi-primary-live-e2e.test.sh` and `tests/fm-opencode-primary-live-e2e.test.sh` exercise native startup paths with first-message and later-message Ahoy regressions. +`tests/fm-turnend-guard.test.sh`, `tests/fm-pi-watch-extension.test.sh`, and `tests/fm-daemon.test.sh` cover marked guard, monitoring, and away-mode delivery. + +[`verification/supervision.md`](verification/supervision.md#native-session-start-delivery) records the active version-scoped transport evidence. diff --git a/docs/subagent-guard.md b/docs/subagent-guard.md index 457e59ad9be..47aaf10e0f3 100644 --- a/docs/subagent-guard.md +++ b/docs/subagent-guard.md @@ -16,9 +16,9 @@ Three consequences were observed, not hypothesized. A real crewmate lives in its own backend session with durable state and survives a primary restart. - The supervision cycle then stayed down for 73 minutes unnoticed, which silently killed the captain's Workflowy intake channel, since that channel only fires while a watch cycle runs. -The deeper defect is that the bypass did not merely skip dispatch, it made the guard stack structurally inert. -Only `bin/fm-spawn.sh` writes `state/<id>.meta`, and every guard keys off that record: `bin/fm-supervision-lib.sh` counts `state/*.meta`, and `bin/fm-turnend-guard.sh` exits silently when that count is zero. -Work started through the harness's own delegation tool writes no metadata, so the in-flight count stayed at zero, the turn-end guard never blocked a blind turn end, and the continuity gate was inert. +The deeper defect is that the bypass did not merely skip dispatch, it made the in-flight-work branch of the guard stack structurally inert. +Only `bin/fm-spawn.sh` writes `state/<id>.meta`, so untracked project work contributes nothing to the in-flight count used by `bin/fm-supervision-lib.sh` and `bin/fm-turnend-guard.sh`. +Work started through the harness's own delegation tool writes no metadata, so the in-flight count stayed at zero and the turn-end guard never blocked a blind turn end. That is the reason the fence has to sit on the harness tool surface, before the primary can create untracked work. No additional guard keyed on task metadata can catch this class of failure, because the failure is precisely the absence of that metadata. @@ -47,14 +47,22 @@ agent subagent task workflow cron schedul worktree delegate spawn dispatch handoff remote sendmessage monitor ``` -Two exclusions keep the shape test from producing false positives. +Three exclusions keep the shape test from producing false positives. - A name beginning `mcp__` is never classified. An MCP server chooses its own tool names, a task or agent noun there is common, and it has no bearing on fleet dispatch. -- The exact names `taskoutput`, `taskstop`, `taskget`, `tasklist`, `cronlist`, `bashoutput`, and `killshell` are allowed. +- `OBSERVE_ONLY_TOOLS`: the exact names `taskoutput`, `taskstop`, `taskget`, `tasklist`, `cronlist`, `bashoutput`, and `killshell` are allowed. These observe or stop work that already exists rather than creating it, and denying them at this layer could strand already-running work with no way to inspect or end it. A Claude primary's optional local deny list may still remove them from the schema. The shipped guard stays narrower on purpose so it can never be the reason a runaway task cannot be stopped. +- `PLAN_ONLY_TOOLS`: the exact names `taskcreate` and `taskupdate` are allowed. + These write, which is why they are a separate list rather than more entries in the observe-or-stop one, but what they write is the harness's session-local todo list. + That list has no executor: it spawns no agent, allocates no worktree, registers no schedule, and starts nothing that could outlive the session or escape a firstmate guard. + So it is not the "work, agent, schedule, or isolated workspace that firstmate would not know about" the guard exists to stop, and the stem match on `task` is a false positive rather than a policy. + The cost of the false positive was concrete: the primary could not track its own plan, and the deny text told it to run `bin/fm-brief.sh` and `bin/fm-spawn.sh` to create a todo entry. + +Both exclusion lists match the whole normalized name, never a substring, so neither can widen by accident: `TaskCreateAgent` and `RemoteTaskCreate` stay denied. +Folding the two lists together would be the drift risk, because the observe-or-stop rationale is not true of a tool that writes. The shipped guard fires on every delegation-shaped name that reaches it, including future names that no deny list knows about yet. That future-name behavior is the reason the tracked matcher must match all tools and let the script filter. @@ -79,10 +87,8 @@ Claude primaries should add this deny list in untracked per-home local settings, "CronCreate", "CronDelete", "CronList", - "TaskCreate", "TaskGet", "TaskList", - "TaskUpdate", "TaskStop", "TaskOutput" ] @@ -103,8 +109,11 @@ It is not tracked for two reasons. The width of the list remains a captain-owned decision, because denying some of these changes how the captain works with the primary session. Keep it as one flat local array that is reviewable at a glance and narrowable in one line. -In particular `TaskOutput`, `TaskStop`, `TaskGet`, `TaskList`, and `CronList` only observe or stop work that already exists, but the recommended local deny list still removes them by default. -The hook deliberately allows those names, so the shipped guard can never strand a runaway task with no way to inspect or end it. +In particular `TaskOutput`, `TaskStop`, `TaskGet`, `TaskList`, and `CronList` only observe or stop work that already exists, yet the recommended local deny list still removes all five by default. +The hook deliberately allows those five, so the shipped guard can never strand a runaway task with no way to inspect or end it, and it allows `TaskCreate` and `TaskUpdate` too, so it can never be the reason the primary cannot track its own plan. +The two session-local todo tools are no longer recommended for local denial at all, because they write only the harness's session-local todo list, which has no executor and spawns nothing, so removing them from the schema removes no delegation power. +Denying them there would instead reproduce at a stronger layer the exact false positive the shipped guard now avoids, leaving anyone who adopts this list verbatim unable to let a primary track its own plan. +Narrowing the list further, including the five observe-or-stop names, is the captain's call, and this local list is the only layer that can remove a todo tool from the primary's schema. `permissions.allow` is a pre-approval list, not an availability list, so there is no fail-closed positive allowlist available. That is why any fixed deny list is fail-open against future tools and why the shape-based guard still exists. @@ -171,7 +180,7 @@ Applicability turns on one question: does the harness expose built-in delegation | Harness | Delegation surface | Status | | --- | --- | --- | -| Claude | 18 known tools, listed above | Scoped guard wired and live-verified; untracked local deny list verified and recommended. | +| Claude | 16 known tools, listed above | Scoped guard wired and live-verified; untracked local deny list verified and recommended. | | Codex | none | Not applicable, verified empirically below. Codex 0.144.1 exposes no subagent, sub-task, or delegated-agent tool, so there is nothing to remove or intercept. `.codex/hooks.json` is unchanged. | | Grok | present, exact tokens unconfirmed | Not wired pending live verification. See below. | | OpenCode | present, exact tokens unconfirmed | Not wired pending live verification. See below. | @@ -285,8 +294,8 @@ This distinction matters when reading the next result: a tool absent from a plai ### Local deny-list hardening -Run in a scratch firstmate-shaped project containing `AGENTS.md`, `state/`, a full copy of `bin/`, and a Claude settings file containing the recommended local deny-list JSON above. -The result validates the recommended local deny-list JSON above, not tracked repo state. +Run in a scratch firstmate-shaped project containing `AGENTS.md`, `state/`, a full copy of `bin/`, and a Claude settings file containing the local deny list exactly as recommended on that date, which was the 18-name form that still included `TaskCreate` and `TaskUpdate`. +The result validates that local deny list rather than tracked repo state, and the recommendation above has since dropped those two session-local todo tools. Asking for deferred entries explicitly returned: ```text @@ -344,7 +353,7 @@ The live consequence is confirmed by the shipped-guard result above: Claude hono ## Automated validation `tests/fm-subagent-pretool-check.test.sh` owns the acceptance matrix and is registered in the `pure-contract-unit` family in `bin/fm-test-run.sh`. -It covers the tracked Claude settings boundary that forbids a `permissions` key; the match-all Claude hook registration; denial of every work-creating delegation tool by shape; denial of twelve hypothetical future tool names that appear on no list; the observe-or-stop and MCP exclusions; the scout-present and scout-absent message variants; the escape hatch including its fail-closed values; inertness in a linked task worktree and in a non-firstmate repo; in-scope enforcement for a marked secondmate home; both stdin transports; the empty-stdout requirement; fail-open transport behavior; and the preserved `Bash` seatbelts and `Stop` guard. +It covers the tracked Claude settings boundary that forbids a `permissions` key; the match-all Claude hook registration; denial of every work-creating delegation tool by shape; denial of twelve hypothetical future tool names that appear on no list; the observe-or-stop, plan-only, and MCP exclusions; the exactness of the plan-only exclusion against six near-miss names a substring or shorter-stem widening would release; the scout-present and scout-absent message variants; the escape hatch including its fail-closed values; inertness in a linked task worktree and in a non-firstmate repo; in-scope enforcement for a marked secondmate home; both stdin transports; the empty-stdout requirement; fail-open transport behavior; and the preserved `Bash` seatbelts and `Stop` guard. Run: @@ -357,9 +366,9 @@ tests/fm-subagent-pretool-check.test.sh ## Known residual gap This change does not close the deeper harness-agnostic defect. -Every firstmate guard keys off `state/<id>.meta`, and only `bin/fm-spawn.sh` writes that record. -`bin/fm-supervision-lib.sh` counts `state/*.meta`, and `bin/fm-turnend-guard.sh` exits silently at zero. -Unaccounted primary work therefore reads as idle rather than suspicious. +Every firstmate guard's in-flight-work branch keys off `state/<id>.meta`, and only `bin/fm-spawn.sh` writes that record. +`bin/fm-supervision-lib.sh` also recognizes an X-mode relay poll as supervision need, but unaccounted primary work still contributes nothing to that predicate. +Without an independent X-mode need, unaccounted primary work therefore reads as idle rather than suspicious. The durable fix for that class is to make the guards treat "the primary is doing project-shaped work with zero `state/*.meta` files" as a suspicious state rather than an idle one. That would catch this class on any harness, including work created through `Bash`. diff --git a/docs/supervision-protocols/claude.md b/docs/supervision-protocols/claude.md index a60545b3afa..c9913553102 100644 --- a/docs/supervision-protocols/claude.md +++ b/docs/supervision-protocols/claude.md @@ -1,23 +1,27 @@ -Mode: Claude background-notify supervision. +Mode: Claude Stop-hook-owned supervision. When this session owns supervision and away mode is not active: 1. Drain first with `bin/fm-wake-drain.sh`. -2. Source `__FM_X_MODE_ENV__` first when X mode is active. -3. First cycle: run `bin/fm-watch-arm.sh` as its own Claude Code background task. -4. Never bundle the arm command with other commands. -5. Never use shell `&` for watcher supervision. - A shell `&`, a truncating pipe, or bundling is denied automatically by the PreToolUse seatbelt (`bin/fm-arm-pretool-check.sh`) registered in `.claude/settings.json`. -6. Treat `watcher: started ...` and `watcher: attached ...` as proof that one live cycle exists. - On attach, the background task follows verified identity-matched successors instead of exiting when the first cycle ends. -7. Failure or missing cycle only: treat any `watcher: FAILED ...` result as an alarm and repair it before ending the turn. -8. Ordinary wake: when the background task completes with `signal:`, `stale:`, `check:`, or `heartbeat`, drain queued wakes, then start exactly one fresh background task before running other fleet commands to handle the wake. +2. Routine watcher arm and re-arm are owned by the Stop `asyncRewake` hook (`bin/fm-claude-stop-autoarm.sh`), never by you. + Every turn end while supervision is needed launches or attaches one home-scoped watcher cycle with no model command and no model tokens. + An actionable close wakes you through the hook's exit-2 rewake, delivered as a `Stop hook feedback` message. +3. On a `Stop hook feedback` wake (`signal:`, `stale:`, `check:`, or `heartbeat`), run `bin/fm-wake-drain.sh` first and handle the wake. + Do not run `bin/fm-watch-arm.sh` after an ordinary wake; the next turn end re-arms automatically when supervision is still needed. Do not invent a wake from an attach-status line alone; drain and act only on real wake records or a real watcher reason line. -9. The continuity PreToolUse gate allows wake drain, watcher arm recovery, and fail-closed teardown, and refuses only other `bin/fm-*.sh` fleet commands while tasks are in flight and no identity-matched live watcher holds the home lock. -10. The existing turn-end guard remains unchanged as the final backstop and is not replaced by this command gate. -11. Recovery only: if a forced restart is genuinely needed, run `bin/fm-watch-arm.sh --restart` through the same Claude background task mechanism. -12. Do not send idle progress while the watcher is parked. +4. On a `Stop hook feedback` watcher-failure wake (`watcher: FAILED ...`), treat it as an alarm: drain, then repair supervision before ending the turn. +5. Manual arm is recovery only. + When a repair is genuinely needed - the Stop hook did not claim this home, or a forced restart is required - run `bin/fm-watch-arm.sh` (or `bin/fm-watch-arm.sh --restart`) as its own Claude Code background task, never bundled with other commands, never with shell `&`. + Source `__FM_X_MODE_ENV__` first when X mode is active. + A shell `&`, a truncating pipe, or bundling is denied automatically by the PreToolUse seatbelt (`bin/fm-arm-pretool-check.sh`) registered in `.claude/settings.json`. +6. Treat `watcher: started ...` and `watcher: attached ...` inside arm output as proof that one live cycle exists. + On attach, the arm follows verified identity-matched successors instead of exiting when the first cycle ends. +7. The durable wake queue preserves actionable events between a rewake and the next Stop-launched arm, while the bounded turn-end guard prevents a blind Stop when recovery did not start. + No PreToolUse hook denies fleet commands based on watcher status. + [`watcher-continuity.md`](../watcher-continuity.md) owns the exact session-lock recovery boundary. +8. The turn-end guard (`bin/fm-turnend-guard.sh --claude`) remains the final backstop. + It allows the stop when a watcher is healthy, when the auto-arm already owns recovery for this event epoch, or when a fresh rewake is recorded; it re-blocks only when none of those materialize, within a bounded budget. +9. Waiting on the hook-owned cycle is silent: do not send idle progress while the watcher is parked. -Claude Code's background task completion is the wake mechanism. -The watcher itself remains `bin/fm-watch.sh`, and `bin/fm-watch-arm.sh` is only the verified background arm wrapper. +The watcher itself remains `bin/fm-watch.sh`, and `bin/fm-watch-arm.sh` remains the verified arm wrapper that the Stop hook foregrounds. Re-arm attaches to an existing healthy cycle when one is already present and follows its verified successor chain. -See [`watcher-continuity.md`](../watcher-continuity.md) for the arm-layer successor and clean-close failure contract. +See [`watcher-continuity.md`](../watcher-continuity.md) for the arm-layer successor and clean-close failure contract and the Claude ownership model. diff --git a/docs/supervision-protocols/grok.md b/docs/supervision-protocols/grok.md index a250edd205a..22444b2bd7f 100644 --- a/docs/supervision-protocols/grok.md +++ b/docs/supervision-protocols/grok.md @@ -30,10 +30,9 @@ When you see a background-task-completed system reminder for the arm: Re-arm attaches to an existing healthy cycle when one is already present and follows its verified successor chain. See [`watcher-continuity.md`](../watcher-continuity.md) for the arm-layer successor and clean-close failure contract. -Grok Stop hooks are passive. -The primary project hook runs `bin/fm-turnend-guard-grok.sh`, which forces at most one same-session follow-up via `grok --resume` when a turn would end blind. -That is a backstop, not the normal wake path. -After any forced follow-up, arm the watcher with the background protocol above. +The primary project Stop hook runs `bin/fm-turnend-guard-grok.sh` as a backstop, not the normal wake path. +[`turnend-guard.md`](../turnend-guard.md) owns its running-payload capability selection between native same-process blocking and the pre-native bounded resume fallback. +After any forced continuation, arm the watcher with the background protocol above. Interactive TUI primary sessions are the supported supervision host. Headless `grok -p` may wait for background process exit but does not reliably surface full auto-wake model output; do not run the primary firstmate as a one-shot headless process. diff --git a/docs/supervision-protocols/opencode.md b/docs/supervision-protocols/opencode.md index 33223ac8bac..3e42535f1ef 100644 --- a/docs/supervision-protocols/opencode.md +++ b/docs/supervision-protocols/opencode.md @@ -14,8 +14,3 @@ When this session owns supervision and away mode is not active: OpenCode's persistent TUI plugin runtime is the wake mechanism. The plugin applies in the main primary checkout and a secondmate's own home, and stays silent only in child crewmate and scout worktrees. - -Continuity verification on 2026-07-17 used OpenCode 1.17.18 in a dedicated tmux socket with an isolated project and `FM_HOME` while retaining the existing managed authentication. -An actionable child close was followed by a ledger-linked successor before prompt handling, the model issued no watcher-arm command, and the turn-end guard did not fire. -Command: `FM_OPENCODE_LIVE_E2E=1 tests/fm-opencode-primary-live-e2e.test.sh`. -Observed output: `ok - OpenCode 1.17.18 live E2E auto-started one successor before prompt handling without a model re-arm`. diff --git a/docs/supervision-protocols/pi.md b/docs/supervision-protocols/pi.md index 21c1a979219..2316428a833 100644 --- a/docs/supervision-protocols/pi.md +++ b/docs/supervision-protocols/pi.md @@ -2,44 +2,22 @@ Mode: Pi extension background wake. When this session owns supervision and away mode is not active: 1. Drain first with `bin/fm-wake-drain.sh`. -2. Confirm the Pi primary auto-loaded both project extensions (plain `pi`, after approving project trust once per clone); if not, restart with `-e __FM_PI_TURNEND_EXT__ -e __FM_PI_EXT__` as a trust-free fallback. +2. Confirm the Pi primary auto-loaded both project extensions (plain `pi` or `pi-signed`, after approving project trust once per clone); if not, restart the selected executable with `-e __FM_PI_TURNEND_EXT__ -e __FM_PI_EXT__` as a trust-free fallback. 3. First cycle only: make the one required `fm_watch_arm_pi` call. Use `/fm-watch-arm-pi` only as a human-entered fallback. Never run `bin/fm-watch-arm.sh` through Pi's bash tool because that foreground arm can wedge the agent and bypasses extension-owned cleanup. 4. If the extension says no live session holds the lock, run `bin/fm-session-start.sh` to reclaim the session lock, then call `fm_watch_arm_pi` again. 5. The extension starts `bin/fm-watch-arm.sh --restart`, keeps the child attached to the live Pi process, and owns every later successor launch. -6. After an actionable child close, the extension rechecks session-lock ownership and verifies one successor before it delivers the follow-up wake; its bounded fallback is defined in `docs/watcher-continuity.md`. -7. Ordinary work, turn completion, and ordinary signal, stale, check, heartbeat, or other wake handling: do not call `fm_watch_arm_pi` again because continuity is extension-owned rather than model-memory-owned. -8. An unexpected child close enters bounded exponential retry, and an exhausted retry or lost session lock is surfaced as a watcher failure instead of disappearing. -9. Missing, failed, or unhealthy cycle only: if a later notification explicitly reports one of those repair conditions, drain queued wakes, inspect the failure text, call `fm_watch_arm_pi`, and restart Pi with both extensions loaded if needed. +6. Ordinary same-process session replacement (`/new`, `/resume`, `/fork`, reload) retires only the prior generation; call `fm_watch_arm_pi` once for the first cycle of the replacement session without restarting Pi. + The generation-owner contract lives in `.pi/extensions/fm-primary-pi-watch.ts`. +7. After an actionable child close, the extension rechecks session-lock ownership and verifies one successor before it delivers the follow-up wake; its bounded fallback is defined in `docs/watcher-continuity.md`. +8. Ordinary work, turn completion, and ordinary signal, stale, check, heartbeat, or other wake handling: do not call `fm_watch_arm_pi` again because continuity is extension-owned rather than model-memory-owned. +9. An unexpected child close enters bounded exponential retry, and an exhausted retry or lost session lock is surfaced as a watcher failure instead of disappearing. +10. Missing, failed, or unhealthy cycle only: if a later notification explicitly reports one of those repair conditions, drain queued wakes, inspect the failure text, call `fm_watch_arm_pi`, and restart the selected Pi-family executable with both extensions loaded if needed. A redundant call while the extension owns an arm child or scheduled retry is an ownership-based `watcher: unchanged` no-op, not an independent health claim. -10. Never use shell `&` for watcher supervision. +11. Never use shell `&` for watcher supervision. The arm mechanism above is extension-owned, not a model tool call, but a manual recovery probe that backgrounds, pipes, or bundles the arm is denied automatically by the PreToolUse seatbelt (`bin/fm-arm-pretool-check.sh`, wired into the turn-end guard extension at `__FM_PI_TURNEND_EXT__`). The turn-end guard extension lives at `__FM_PI_TURNEND_EXT__`. The watcher extension lives at `__FM_PI_EXT__`. Both are tracked, project-local `.pi/extensions/*.ts` files that Pi auto-discovers once the project is trusted; `bin/fm-session-start.sh` reports when the running Pi session has not loaded both required extensions. - -Verification on 2026-07-09 used Pi 0.80.5, an isolated `PI_CODING_AGENT_DIR`, an isolated `FM_HOME`, and the dedicated tmux socket `fm-pi-q6-lab`. -The command `Use the fm_watch_arm_pi custom tool now. Do not use bash.` rendered `watcher: started Pi extension arm child 1`, then the model returned `DONE` without the prior `result.content.filter(...)` crash. -The extension tool returned Pi's required text `content` plus structured `details` and used `Type.Object({})` for its parameter schema. -The human command `/fm-watch-arm-pi` notified through `ctx.ui.notify(...)` and returned no value. -The clean-exit probe ran `/quit`, printed `PI_EXIT=0`, and confirmed that both the attached arm process and watcher child were gone. -That cleanup is owned by a one-shot process `exit` listener because Pi 0.80.5 did not reliably emit `session_shutdown` for `/quit`; the listener is removed when `session_shutdown` does run. -Command run for the complete interactive regression: `FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh`. -Observed output: `ok - Pi 0.80.5 live E2E rendered the tool, guarded once, woke, re-armed, and cleaned up on exit`. -Command run for the installed-type contract: `tests/fm-pi-primary-types.test.sh`. -Observed output: `ok - Pi primary extensions pass strict no-emit typecheck against Pi 0.80.5`. - -Continuity verification on 2026-07-17 used Pi 0.80.10 with the existing shared Pi credential store and the explicit `openai-codex/gpt-5.6-sol` provider/model pin. -The isolated live test copied no credential material and created no account. -The model called `fm_watch_arm_pi` exactly once, an actionable status closed that cycle, the extension ledger-linked a verified successor before the handling turn ended, the turn-end guard never fired, and `/quit` cleaned up both child processes. -Command: `FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh`. -Observed output: `ok - Pi 0.80.10 live E2E used shared Codex auth, auto-started one successor before turn end, and cleaned up`. - -Continuity and Calm verification on 2026-07-23 used Pi 0.81.1 with the existing shared Pi credential store and the explicit `openai-codex/gpt-5.6-sol` provider/model pin. -The isolated live test activated Calm, proved Pi's native `Working...` row remained visible during a credentialed provider request, proved no `calm transcript` status appeared, restored Calm off, and then completed the unchanged watcher successor and clean-exit lifecycle. -Command: `FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh`. -Observed output: `ok - Pi 0.81.1 live E2E covered native Calm Working visibility, Ahoy first/later messages, legacy transcripts, near misses, and watcher continuity`. - -The authoritative Pi 0.81.1 operational-follow-up and Calm presentation verification record, including exact commands and output, is in [`docs/calm-mode-feasibility.md`](../calm-mode-feasibility.md#2026-07-23-verification-record). diff --git a/docs/tmux-backend.md b/docs/tmux-backend.md index 426e060e978..3bf20fe9d1a 100644 --- a/docs/tmux-backend.md +++ b/docs/tmux-backend.md @@ -1,146 +1,91 @@ -# tmux runtime backend (reference) +# tmux runtime backend -tmux is firstmate's verified reference runtime backend: the session provider every other backend is compared against, and the fully verified baseline for secondmate support. -This is the setup guide; for the shared runtime-backend abstraction and selection order, see [`docs/architecture.md`](architecture.md) ("Runtime session backends") and [`docs/configuration.md`](configuration.md) ("Runtime backend"). +tmux is Firstmate's verified reference runtime backend and the fully supported baseline for secondmate homes. +[`configuration.md`](configuration.md#runtime-backend-configbackend--fm_backend) owns shared backend selection and metadata semantics. -## What it is and when to pick it +## Setup -tmux is a terminal multiplexer. -Firstmate gives each crewmate its own tmux window inside a session, so you can attach and watch a task work, or type into its window to intervene directly. -Pick tmux unless you have a specific reason to try an experimental backend (herdr, zellij, Orca, or cmux) - it is the fully verified reference path for secondmate homes, while Orca and cmux are the backends that do not support secondmate spawns. +Install tmux with `brew install tmux` or your platform package manager. +The universal harness and toolchain requirements are in [`configuration.md`](configuration.md#toolchain). -## Prerequisites +tmux is the hard default when no explicit setting or runtime auto-detection selects another backend. +Select it explicitly with local `config/backend` containing `tmux`, with `FM_BACKEND=tmux` for one launch, or by asking Firstmate to use tmux. +An explicit selection is also the opt-out from Herdr or cmux runtime auto-detection. -- tmux itself: `brew install tmux` (or your platform's package manager). -- The universal firstmate prerequisites: a verified crew harness plus the required toolchain, detected at session start and installed only after you approve; [`docs/configuration.md`](configuration.md) owns both lists ("Harness support", "Toolchain"). +No provisioning is required before the first task. -## Selecting it +## Watching the crew -tmux is the hard default: it needs no explicit selection. -It is also what firstmate falls back to when nothing else is set - no local `config/backend` file, no `FM_BACKEND`, no explicit `--backend` flag firstmate passes internally when it spawns a task - and runtime auto-detection (see below) does not pick anything either. -You can still select it explicitly by putting `tmux` in a local `config/backend` file - the durable way to pick it - or by exporting `FM_BACKEND=tmux` when you launch your harness for a one-off session; telling the first mate in chat to use tmux also works. -This mainly matters as an opt-out of herdr or cmux runtime auto-detection (see [`docs/herdr-backend.md`](herdr-backend.md) and [`docs/cmux-backend.md`](cmux-backend.md)). - -## First run - -Nothing to provision up front. -The first crewmate spawn creates whatever tmux session and window it needs. - -## Run inside tmux for the best experience - -Launch your harness from inside a tmux session (`tmux new -s firstmate` or similar, then start your agent). -Every crewmate window then lands in that same session, where you can watch the crew work in real time or type into any window to intervene. -When following the commands below, use that session's actual name. -Inside tmux, `tmux display-message -p '#S'` prints it. - -## Outside tmux: the detached `firstmate` session - -If you launch your harness outside of tmux, crewmate windows land in a detached session named `firstmate`, created on first use. -Attach to it any time with: +For the best visible experience, launch the primary harness inside a tmux session: ```sh -tmux attach -t firstmate +tmux new -s firstmate ``` -## Watching and typing into crew windows - -Once attached, each crewmate is its own window named `fm-<id>`: +Crew tasks become windows in that session. +`tmux display-message -p '#S'` prints its name. +If the primary harness runs outside tmux, Firstmate creates or reuses a detached session named `firstmate`: ```sh -tmux list-windows -t <session-name> # see every crew window -tmux select-window -t <session-name>:fm-<id> # jump to one, or use ctrl-b <n> +tmux attach -t firstmate ``` -Use the current tmux session name when firstmate was launched inside tmux; use `firstmate` only for the detached outside-tmux path. -Typing directly into an attached window is authoritative direct intervention - the first mate treats it the same as any other captain instruction and reconciles at the next heartbeat. -You do not need to attach at all for routine supervision: from an active firstmate session, the first mate reads crew windows itself with `bin/fm-peek.sh fm-<id>` (a bounded, read-only capture) and steers a crew with `FM_HOME=<this-firstmate-home> bin/fm-send.sh fm-<id> "<text>"` unless `FM_HOME` is already set to the active firstmate home. - -## Verifying it works - -Ask the first mate for any small piece of work, or spawn a trivial scout task, and confirm a new window shows up: +Each task window is named `fm-<id>`. ```sh tmux list-windows -t <session-name> +tmux select-window -t <session-name>:fm-<id> ``` -Use the current tmux session name for the run-inside-tmux path, or `firstmate` for the detached outside-tmux path. -You should see a `fm-<id>` window for the task, live and updating as the crewmate works. +Typing into an attached task window is authoritative direct intervention. +Routine supervision does not require attachment: `bin/fm-peek.sh <id>` captures a bounded tail and `FM_HOME=<home> bin/fm-send.sh <id> '<text>'` steers the recorded endpoint. -## Agent liveness probe +Verify setup by spawning a small task and confirming its `fm-<id>` window appears in the selected session. -`fm_backend_target_exists` (`bin/fm-backend.sh`) only checks that a window's pane still exists. -A secondmate agent that exits leaves its pane alive as a bare idle shell, which passes that check as "alive" - the gap `bin/fm-bootstrap.sh`'s session-start secondmate-liveness sweep exists to close (evidence 2026-07-07: every secondmate in one fleet was found sitting at a dead `zsh` shell, invisible to that check). +## Current behavior and safety -`fm_backend_tmux_agent_state` (`bin/backends/tmux.sh`) answers a deeper question: is a real harness-agent *process* running in the pane right now, or is the recorded endpoint authoritatively missing? -It reads tmux's own `#{pane_current_command}`, which reports the pane's live foreground process name - already resolved by tmux from the pty's controlling process group, not something this adapter derives itself. +A target-existence check proves only that the pane exists. +The deeper tmux agent-liveness probe first verifies exact window membership, then reads `#{pane_current_command}` to distinguish a running harness process from a bare idle shell. +It classifies recognized Claude, Codex, OpenCode, Pi, pi-signed, Grok, and Kimi process names as `alive`, common shells as `dead`, an authoritatively absent window as `missing`, unreadable state as `unreadable`, and every other process as `ambiguous`. +Only `dead` and `missing` authorize recovery because a false dead result could launch a duplicate agent. -Agent liveness and composer safety are separate checks. -During away-mode escalation delivery, `fm_tmux_composer_state` sends a bare shell glyph on an unbordered row to the shared composer classifier as `unknown`, and the daemon injects only into an affirmatively `empty` composer; see [Composer-emptiness safety](herdr-backend.md#composer-emptiness-safety-2026-07-10-fleet-wide-across-all-four-backends). - -## Submit acknowledgement: "landed" is empty (with one busy-queue exception) - -The shared `fm_tmux_submit_enter_core` (`bin/fm-tmux-lib.sh`) types the message once, then retries Enter (Enter only, never a retype) until the composer clears. -The submit is reported `empty` iff the composer cleared, which is the same corrected, border-aware detector the composer guard uses, so a bordered-but-empty composer is correctly seen as the positive acknowledgement of a delivered submit. -A genuine swallowed Enter leaves the typed text in the composer and the function reports `pending`; `fm-send` fails on `pending` so the captain learns the steer did not land instead of leaving it unsubmitted. +The verified Pi Launcher path reports the exact foreground command `pi-launcher` for both pi and pi-signed, while direct executable identities `pi`, `pi-signed`, and `Pi` remain accepted exactly. +Similar or prefixed process names are not accepted through those exact Pi-family entries. -**Exception (opencode 1.18.4, on the tmux backend):** while the agent is mid-turn, opencode accepts Enter as a "send when the turn ends" keystroke but does not clear the composer until then, so the typed text stays visible the whole time. -After the Enter-retry budget is spent and the composer still reads `pending`, the submit core falls back to `fm_pane_is_busy`: -a busy pane means the harness accepted and queued the Enter (reported as `empty`, so the caller does not re-send), and an idle pane keeps `pending` as a genuine swallow. -This is the only place that exception lives; the herdr adapter observes the same opencode behavior but needs a separate fix (see the opencode note in [harness-adapters](../.agents/skills/harness-adapters/SKILL.md) and the opencode-busy gap recorded in [herdr-backend.md](herdr-backend.md)). -Regression coverage: `tests/fm-tmux-submit-busy.test.sh` covers the four scenarios (busy pane + pending composer -> `empty`, idle pane + pending composer -> `pending`, busy pane + cleared composer -> `empty`, idle pane + cleared composer -> `empty`). +Agent liveness and composer safety are separate checks. +For a bordered composer, the tmux reader locates the complete box structurally and classifies every content row through the shared ANSI and ghost handling in `bin/fm-composer-lib.sh`. +Real text on any content row is pending, while only an unambiguous box with every row empty is proven empty. +Unreadable, incomplete, or structurally ambiguous boxes fail closed, and panes without a bordered composer retain the compatible cursor-row classification. +The shared classifier accepts a shell glyph as an empty agent composer only inside a verified bordered composer. +A bare shell prompt is `unknown`, so away-mode escalation is never injected into a dead shell. -Verified empirically with real tmux 3.6a on macOS (Darwin 25.5.0), 2026-07-07: +Rendered busy detection is also harness-scoped. +Task metadata selects only that harness's verified signature, so output from one harness cannot make another harness appear busy. +The exact selection contract and safety rationale live in [architecture](architecture.md#runtime-session-backends), while the signatures live in [the harness-adapters skill](../.agents/skills/harness-adapters/SKILL.md). -```sh -$ tmux new-session -d -s fmtest -n testwin -$ tmux display-message -p -t fmtest:testwin '#{pane_current_command}' -zsh -$ tmux send-keys -t fmtest:testwin 'sleep 30' Enter -$ tmux display-message -p -t fmtest:testwin '#{pane_current_command}' -sleep -$ tmux send-keys -t fmtest:testwin C-c -$ tmux display-message -p -t fmtest:testwin '#{pane_current_command}' -zsh -``` +`bin/fm-tmux-lib.sh` owns exact type-and-submit mechanics. +It types a message once and retries Enter only until the composer clears. +Only a proven empty composer is a positive delivery acknowledgement. +Text left in established structure remains `pending`, text in ambiguous structure remains unproven, and unreadable or unsafe state remains unknown. +`fm-send.sh` reports every unconfirmed verdict as a failure instead of retyping or assuming delivery. -An idle pane reports the shell's own name; a live foreground process reports its own name; the pane reverts to the shell's name the moment that process exits - exactly the alive/dead signal the probe needs. +OpenCode 1.18.4 has one busy-queue exception. +While OpenCode is mid-turn, Enter queues the message but leaves its text visible until the turn completes. +After the normal retry budget, only structurally proven pending text in a provably busy pane is accepted as queued, while an idle pane remains `pending` as a genuine swallowed Enter. +Ambiguous pending text never receives the busy-queue conversion. +`tests/fm-tmux-submit-busy.test.sh` covers busy and idle panes with proven, ambiguous, and cleared composers. -A second case matters for a harness that shells out to subcommands while it runs (git, npm, no-mistakes, ...): does `pane_current_command` report the harness or the subcommand? -Verified the same session: a persisting parent process running a child command (`bash -c 'echo start; sleep 30; echo end'`, where the parent bash stays alive waiting on its own child) reports the PARENT's own name (`bash`) throughout, not the child's (`sleep`) - so a harness that survives while it shells out stays correctly classified as alive. -(A single-simple-command `bash -c "sleep 30"` is a different, unrelated case: bash execs directly into `sleep`, replacing itself, so the reported name changes because the process itself became `sleep` - not because tmux "saw through" to a child.) +## Limits and regression entry points -The recovery classifier (`fm_backend_tmux_agent_state`) maps the observation to the shared detailed state owned by `fm_backend_agent_state` in `bin/fm-backend.sh`. -A recognized harness is `alive`, a bare shell is `dead`, and an unrecognized foreground process is `ambiguous`. -The classifier checks exact window-name membership in a readable session inventory before trusting `display-message`, because tmux silently redirects a missing named target to the active window. -It returns `missing` when `tmux list-windows` successfully reads the recorded session and omits the exact recorded window, or when tmux definitively reports that the recorded session or server is absent. -Any other failed inventory or pane read is `unreadable` and never authorizes recovery. -`fm_backend_tmux_agent_alive` remains the compatibility view that maps these detailed states back to `alive`, `dead`, or `unknown` for callers that do not need the reason. - -Verified with real tmux 3.6a on macOS (Darwin 25.5.0), 2026-07-23, using the private `-L fm-target-check-<pid>` socket also exercised by `tests/fm-backend-tmux-smoke.test.sh`: +- tmux is the reference path and supports secondmate homes. +- The OpenCode busy-queue exception is tmux-specific; Herdr retains its separately documented gap. ```sh -$ tmux -L "$socket" kill-window -t smoke:fm-smoke1 -$ tmux -L "$socket" display-message -p -t smoke:fm-smoke1 '#{window_name}:#{pane_current_command}' -main:zsh -$ tmux -L "$socket" list-windows -t smoke -F '#{window_name}' -main -$ fm_backend_agent_state tmux smoke:fm-smoke1 -missing +tests/fm-backend-tmux-smoke.test.sh +tests/fm-composer-ghost.test.sh +tests/fm-kimi-harness.test.sh +tests/fm-tmux-submit-busy.test.sh +tests/fm-bootstrap.test.sh ``` -The first post-kill command exits 0 and reports the unrelated active `main` window, which is the earliest meaningful divergence that made process-only liveness inconclusive for missing Pi windows. -The exact inventory check prevents that fallback from masquerading as an existing ambiguous process, while an unreadable inventory still preserves duplicate prevention. - -### Known gap: `pi` cannot be confidently classified - -`pi` is a `#!/usr/bin/env node` script (confirmed via its shebang and installed path, 2026-07-07), so a live `pi` agent's pane reports `node` as its `pane_current_command`, not `pi` - verified by running a long-lived `node -e` script in a pane and confirming its foreground process is a genuine child reachable via `pgrep -P <pane_pid>` with an inspectable `ps -o args=` (the same technique `bin/fm-harness.sh`'s own self-detection uses when walking UP its ancestry), while `pi --version` itself was observed to exit too quickly under the same pane to reliably capture its live foreground state - real `pi` invocations were not available to test. -Since `node` is also the generic name for a plain interpreter session, any future JS-based harness, or someone's unrelated node script, there is no way to attribute a bare `node` foreground process back to `pi` specifically from outside the pane without deeper (and fragile) argument introspection. -The classifier deliberately reports `ambiguous` for an existing `node`/`python`/`python3` process rather than guess - per the secondmate-liveness sweep's correctness bar, a wrong `alive` is harmless but a wrong `dead` spins up a duplicate agent, so an unresolvable existing process must never be treated as confidently dead. -Practical effect: an existing Pi secondmate pane that reports `node` is never auto-healed, preserving duplicate prevention. -A recorded Pi secondmate window that is authoritatively absent is different: no process exists to misattribute, so the corroborated `missing` state safely relaunches it at session start. -Classifying an existing Pi process more precisely would still need either a Pi-specific marker inspectable from outside the process or accepting fragile argument inspection, neither of which this recovery path does. - -## Limitations - -None specific to tmux for the reference path itself - it is the fully verified reference backend, while Orca and cmux are the backends without secondmate support. -The agent-liveness probe above retains one known gap for an existing Pi process (`node`, see above); authoritatively absent Pi windows are covered. +[`verification/runtime-backends.md`](verification/runtime-backends.md#tmux) records the active foreground-process and submit evidence. diff --git a/docs/turnend-guard.md b/docs/turnend-guard.md index c631eca577e..8ee750de397 100644 --- a/docs/turnend-guard.md +++ b/docs/turnend-guard.md @@ -1,154 +1,97 @@ # Primary turn-end supervision guard -This is the authoritative contract for the "no turn ends blind" primary guard referenced from AGENTS.md section 8. -The turn-end supervision predicate lives in `bin/fm-turnend-guard.sh`. -Its primary-checkout scope lives in `bin/fm-primary-scope-lib.sh`, shared with the native session-start nudge documented in `docs/sessionstart-nudge.md`. -Harness-specific tracked hook files only adapt each verified harness's real turn-end mechanism to that shared predicate. -Related but separate PreToolUse guards deny a bad tool or command shape before it runs rather than detecting a blind turn end afterward: the watcher-arm seatbelt (`bin/fm-arm-pretool-check.sh`, `docs/arm-pretool-check.md`), the cd-guard (`bin/fm-cd-pretool-check.sh`, `docs/cd-guard.md`), and the primary delegation-shape guard (`bin/fm-subagent-pretool-check.sh`, `docs/subagent-guard.md`). -Each guard's own document defines its scope; do not infer this guard's scoping, loop safety, or fail-open tradeoffs for its PreToolUse siblings. - -## Gap Closed - -`bin/fm-guard.sh` is pull-based: it warns whenever some other supervision script happens to run, and prints nothing otherwise. -The primary can otherwise end a turn after handling wakes without resuming supervision, then sit blind until another fleet command happens to run. -On 2026-07-04, that exact gap left a parked no-mistakes gate unwatched for about nine hours. - -`bin/fm-turnend-guard.sh` closes the gap by checking the primary's own turn-end path. -When tasks are in flight and there is no live identity-matched watcher with a fresh beacon, a harness hook must either block the turn end or force a bounded follow-up turn that tells the primary to repair the missing or failed watcher cycle using the recovery instruction in its emitted session-start protocol. - -## Shared Predicate - -The guard first calls the shared primary scope to constrain itself to a real primary checkout. -A secondmate home runs its own primary firstmate session, so a genuine `.fm-secondmate-home` marker force-includes it whether treehouse leased it as a linked worktree or it is a git-cloned plain checkout. -The marker must be a regular non-symlink file whose first line, after all whitespace is removed, contains a non-empty identifier made only of letters, digits, dots, underscores, and dashes. -An unmarked checkout, or one with an invalid marker, falls through to the git-dir check. -That check keeps crewmate and scout worktrees inert because firstmate provisions them as linked git worktrees, where `git rev-parse --git-dir` differs from `git rev-parse --git-common-dir`. -It also requires `AGENTS.md`, `bin/`, and the effective state directory to exist. - -For an in-scope primary checkout, it counts in-flight work from `state/*.meta`. -If no task is in flight, it exits silently. -If work is in flight, it requires `fm_watcher_healthy <state-dir> <watch-path> [grace-seconds] [home]` from `bin/fm-wake-lib.sh`. -That is the same identity-matched live lock and fresh beacon check used by `bin/fm-watch-arm.sh`. -A stale beacon blocks even if a watcher pid is still live. -A fresh leftover beacon blocks if the watcher lock is missing, dead, or identity-mismatched. - -`FM_STATE_OVERRIDE` wins over `FM_HOME/state`, and `FM_HOME` wins over repo-root `state/`. -`FM_GUARD_GRACE` controls the beacon freshness window and defaults to 300 seconds. -If `jq` is missing or hook stdin is empty, the guard fails open and exits 0 because it cannot safely read loop-guard fields. - -## Harness Integrations - -All verified primary harnesses have a tracked integration: - -- `claude`: `.claude/settings.json` registers a `Stop` hook command anchored through `"$CLAUDE_PROJECT_DIR"/bin/fm-turnend-guard.sh`. -- `codex`: `.codex/hooks.json` registers a `Stop` hook that reads the hook payload once, anchors the executable to the hook command process working directory, verifies that root is firstmate-shaped and hook-bearing, and pipes the original payload to that checkout's `bin/fm-turnend-guard.sh`. -- `opencode`: `.opencode/plugins/fm-primary-turnend-guard.js` listens for `session.idle`, lets the watcher-arm coordinator handle normal idle supervision first, runs the shared guard only when that coordinator does not act, and uses `client.session.promptAsync` to force one follow-up prompt when the guard returns 2. -- `pi`: `.pi/extensions/fm-primary-turnend-guard.ts` listens for `agent_settled`, marks the extension version loaded for session-start checks, runs the shared guard once per logical agent run, and uses `pi.sendUserMessage(..., { deliverAs: "followUp" })` to force one follow-up prompt when the guard returns 2. -- `grok`: `.grok/hooks/fm-primary-turnend-guard.json` registers a `Stop` hook that invokes `bin/fm-turnend-guard-grok.sh`. - The adapter runs the shared guard and, when it returns 2, invokes `grok --resume <sessionId> -p <guard-reason>` with `GROK_TURNEND_GUARD_ACTIVE=1`. - It does not pass `--permission-mode`, so the passive Stop hook cannot grant stronger tool permissions than Grok's resumed-session default. - -Claude and Codex support a direct blocking Stop hook. -For those harnesses, exit status 2 plus stderr from `bin/fm-turnend-guard.sh` blocks the stop and feeds the reason back into the model. -Both payloads include `stop_hook_active`; when it is true, the shared guard exits 0 so the harness can end after one forced continuation. - -OpenCode, Pi, and Grok expose passive lifecycle callbacks for this purpose. -Their adapters fail open at the hook boundary to avoid corrupting a user session, but they force one follow-up turn when the shared predicate blocks. -Those forced user-role prompts use the canonical `turn-end-guard` operational kind after the U+2063 `FIRSTMATE_OP: ` prefix so Ahoy cannot mistake them for captain-authored boundaries. -Each adapter carries its own in-process or environment loop guard so the forced follow-up does not recursively schedule another follow-up. -Pi keeps that latch active across every internal tool turn and clears it only when the generated guard follow-up reaches `agent_settled`, or immediately when follow-up delivery fails. -If a passive adapter cannot call its SDK method, cannot find `grok`, or cannot recover the Grok session id, it fails open and relies on the pull-based `fm-guard.sh` warning at the next fleet command. -That warning uses `bin/fm-supervision-instructions.sh --repair-line`, so it points back to the active harness protocol instead of hardcoding one background-arm command. - -## Empirical Validation - -All harnesses were validated on 2026-07-08 in scratch repos or throwaway homes, not against the captain's live primary fleet state. - -Claude Code 2.1.204 preserved the existing behavior. -Hook file used: `.claude/settings.json`. -Command run: `claude -p "Say hi in exactly one word." --dangerously-skip-permissions --output-format json` with a scratch Stop hook that printed `SMOKETEST: you must say the word BANANA before stopping` and exited 2. -Observed output: the first stop payload had `stop_hook_active=false`, the stop was blocked, the model continued with `BANANA`, and the second stop payload had `stop_hook_active=true` and was allowed. -Earlier validation on 2026-07-04 also verified that `CLAUDE_PROJECT_DIR` is set to the settings-loaded project root, while the hook command itself runs from the session cwd. - -Codex `codex-cli 0.142.1` was validated with a scratch `.codex/hooks.json` Stop hook. -Hook file used: `.codex/hooks.json`. -Command run: `codex exec --dangerously-bypass-hook-trust --dangerously-bypass-approvals-and-sandbox --skip-git-repo-check --output-last-message last.txt 'Say hi in exactly one word.'`. -Observed output: the first model output was `Hi`, the Stop hook exited 2, Codex logged `hook: Stop Blocked`, the model continued with `CODEXHOOK`, and the second hook call had `stop_hook_active=true`. -The Stop payload included `cwd`. -Command run for root-signal probe: `codex exec --ephemeral --json --dangerously-bypass-hook-trust --dangerously-bypass-approvals-and-sandbox --skip-git-repo-check --output-last-message last.txt 'Use the shell tool to run mkdir -p outside && cd outside && pwd, then use the shell tool again to run pwd. Your final answer must include the two observed outputs.'`. -Observed output: the first command printed `<scratch>/outside`, the second command printed `<scratch>`, the Stop hook process `pwd -P` printed `<scratch>`, payload `cwd` printed `<scratch>`, and `CODEX_PROJECT_DIR`, `CODEX_WORKSPACE_ROOT`, and `CODEX_CWD` were empty. -The tracked command therefore treats hook process PWD as the hook-loaded firstmate root and does not let payload `cwd` choose an executable. -It still passes the original payload to `bin/fm-turnend-guard.sh`, so the shared loop guard reads `stop_hook_active`. - -OpenCode 1.17.6 was validated with project plugins under scratch `.opencode/plugins/`. -Hook file used: `.opencode/plugins/fm-smoke.js` for throw testing and `.opencode/plugins/fm-primary-turnend-guard.js` for follow-up testing. -Command run for passive behavior: `opencode run --print-logs --log-level DEBUG --dangerously-skip-permissions 'Say hi in exactly one word.'`. -Observed output: the plugin received `session.idle`, threw an error, and `opencode run` still exited 0 with `Hi`, proving `session.idle` cannot block directly. -Command run for follow-up behavior: `OPENCODE_CONFIG_CONTENT='{"permission":{"*":"allow"}}' opencode --prompt 'Say hi in exactly one word.' --print-logs --log-level INFO`. -Observed output: the plugin called `client.session.promptAsync`, the TUI ran a second turn, and the second model output contained `OPENCODEHOOK`. -In noninteractive `opencode run`, `promptAsync` returned successfully but the process exited before displaying the follow-up, so this adapter is trusted for primary TUI sessions and documented as passive/fail-open in headless mode. - -Pi 0.80.5 was re-validated on 2026-07-09 in a disposable primary-shaped clone with isolated `PI_CODING_AGENT_DIR`, isolated `FM_HOME`, and tmux socket `fm-pi-q6-lab`. -Hook files used: the tracked `.pi/extensions/fm-primary-turnend-guard.ts` and `.pi/extensions/fm-primary-pi-watch.ts`. -Commands run inside separate interactive turns: `printf PI_E2E_BASH_ONE` through Pi's bash tool, `README.md:1-5` through Pi's read tool, and `printf PI_E2E_BASH_TWO` through Pi's bash tool. -Command used to make the shared predicate unhealthy: `: > "$FM_HOME/state/pi-e2e.meta"`. -The next no-tool prompt produced exactly one `TURN WOULD END BLIND` follow-up, and that follow-up called `fm_watch_arm_pi` once with output `watcher: started Pi extension arm child 1`. -The three earlier tool turns produced no guard follow-up because no work was in flight. -Command used to fire the watcher: `printf 'done: pi e2e watcher fire\n' > "$FM_HOME/state/pi-e2e.status"`. -Observed output after the wake: Pi ran `bin/fm-wake-drain.sh`, read the terminal status, called `fm_watch_arm_pi`, and rendered `watcher: started Pi extension arm child 2`. -This 2026-07-09 observation predates extension-owned successor continuity; [`watcher-continuity.md`](watcher-continuity.md) owns the current ordinary-wake contract. -The complete pane contained one guard message and zero foreground `bin/fm-watch-arm.sh` bash calls. -`/quit` printed `PI_EXIT=0`, and the second arm process plus its watcher child were both gone afterward. - -Grok 0.2.91 was validated with a scratch `GROK_HOME` and symlinked auth/config. -Hook file used for tracked project-hook loading: `<scratch-project>/.grok/hooks/fm-smoke.json`, matching the tracked `.grok/hooks/fm-primary-turnend-guard.json` location. -Command run for project-hook loading: `GROK_HOME="$scratch/grok-home" grok --trust -p 'Say hi in exactly one word.' --permission-mode bypassPermissions --output-format plain --leader-socket "$scratch/leader.sock"`. -Observed output: the project Stop hook fired under `--trust` and received `GROK_HOOK_EVENT=stop`, `GROK_WORKSPACE_ROOT`, and a payload containing `sessionId`. -Hook file used for passive behavior and forced-resume behavior: `$GROK_HOME/hooks/fm-primary-turnend-guard.json` plus `bin/fm-turnend-guard-grok.sh`. -Command run for passive behavior: `GROK_HOME="$scratch/grok-home" grok -p 'Say hi in exactly one word.' --permission-mode bypassPermissions --output-format plain --leader-socket "$scratch/leader.sock"`. -Observed output: the global Stop hook fired and received `GROK_HOOK_EVENT=stop`, `GROK_WORKSPACE_ROOT`, and a payload containing `sessionId`, but exiting 2 did not make the model continue. -Command run for forced resume behavior: the Stop hook ran `GROK_TURNEND_GUARD_ACTIVE=1 GROK_HOME="$scratch/grok-home" grok --resume "$session_id" -p 'SMOKETEST: say exactly GROKRESUMEHOOK...' --permission-mode bypassPermissions --output-format plain --leader-socket "$scratch/leader.sock"`. -Observed output: the outer turn printed `Hi`, the nested resumed turn printed `GROKRESUMEHOOK`, and the nested Stop hook saw `GROK_TURNEND_GUARD_ACTIVE=1` and did not recurse. -That validation command used `--permission-mode bypassPermissions` only to keep the scratch smoke unattended; the tracked adapter intentionally omits `--permission-mode`. -Project-local Grok hooks did not fire in scratch single mode without a trust grant. -The primary integration therefore requires the primary firstmate checkout to be trusted for Grok hooks, which can be done with `/hooks-trust` or launch-time `--trust`. -If Grok declines to load project hooks, this primary guard fails open and `fm-guard.sh` remains the next-command alarm. - -**2026-07-09 update:** grok 0.2.93 broke the `.grok/hooks/fm-primary-turnend-guard.json` Stop hook with `hook not executed: required env var(s) not set: ${root}`, because grok's own `${VAR}` expansion over the raw `command` string does not tolerate a bare local variable assigned earlier in the same `bash -lc` script. -The hook command was fixed to reference `${GROK_WORKSPACE_ROOT:-}` directly everywhere instead of assigning it to `$root` first, and re-validated against grok 0.2.93 to fire and complete cleanly. -See `docs/arm-pretool-check.md`'s "Harness wiring" section for the same Grok expansion requirement; that document's Grok hook shares the same fix. - -### 2026-07-12: secondmate-home enablement and the autonomous background-notify wake - -The guard originally early-exited in every secondmate home on the `.fm-secondmate-home` marker. -That was a scoping choice inherited from the guard's primary-only origin, not a defense against any secondmate-specific hazard. -A genuinely marked secondmate home is now force-included as a guarded primary regardless of whether it is a treehouse-leased linked worktree or a git-cloned plain checkout. -Only unmarked child worktrees fall through to the linked-worktree exemption, and marker validation prevents an empty, malformed, or symlink marker from spoofing inclusion. - -"No turn ends blind" for a secondmate is delivered by the same two mechanisms the main primary relies on. -Mechanism B, the turn-end backstop, is this guard; its secondmate-home behavior is covered by hermetic tests in `tests/fm-turnend-guard.test.sh` (`test_hook_blocks_in_secondmate_own_home`, `test_hook_blocks_in_treehouse_leased_secondmate_home`, `test_hook_silent_in_idle_secondmate_home`, `test_hook_secondmate_loop_guard_allows_retry`, `test_hook_secondmate_reinvoke_recovery_loop`, `test_hook_silent_in_secondmate_child_worktree`, and `test_hook_exempts_linked_worktree_with_stray_marker`). -Mechanism A, the autonomous wake, is a harness property; the emitted supervision protocol owns whether the model or an extension/plugin continues the watcher cycle after delivering that wake. -Mechanism A cannot be a hermetic CI assertion because it requires a live model session, so it is recorded here as a dated first-hand measurement while `test_hook_secondmate_reinvoke_recovery_loop` covers the guard's deterministic half of the same recovery loop. - -Autonomous-re-invoke measurement, run first-hand on Claude Code 2.1.207 (Darwin 25.5.0) on 2026-07-12. -Procedure: launch a detached `run_in_background` Bash task that models a one-shot watcher - it records a launch epoch, runs `sleep 25`, then records a completion epoch just before exit, writing only to the session scratchpad - then end the turn with no further tool calls and no pending question, a genuinely idle session with no human input. -Observed marker timestamps: - -``` -launch_epoch = 1783890980 (14:16:20) turn ends, session goes idle -complete_epoch = 1783891005 (14:16:45) background task exits, 25s idle -reinvoke_epoch = 1783891016 (14:16:56) MODEL RE-INVOKED --------------------------------------------------------------- -wake latency (task complete -> model re-invoked): 11s, with ZERO human input -``` - -The re-invocation arrived as a `<task-notification>` whose accompanying system notice stated verbatim "No human input has been received since the last genuine user message in this conversation". -So the model was re-invoked solely by the background task's completion while idle, which is Mechanism A - the same background-notify wake the Claude supervision protocol relies on for the main primary. -This matches the harness tool contract that a `run_in_background` task "keeps running across turns and re-invokes you when it exits", and reproduces the 11s latency the task audit measured independently on the same harness version. -No Herdr command was issued and no fleet state was touched; the experiment wrote only to the session scratchpad, which was discarded. - -## Tests - -`tests/fm-turnend-guard.test.sh` covers the shared predicate, primary scoping (including a secondmate's own home being guarded like the main primary while its child worktrees stay exempt), `FM_HOME` and `FM_STATE_OVERRIDE` precedence, Pi logical-run latch behavior for no-tool and multi-tool runs, fail-open behavior without `jq`, tracked hook registration for all five harnesses, and the Grok adapter's forced-resume loop guard and permission-mode regression. -The default behavior suite does not invoke live language-model harnesses. -`FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh` opts into the isolated interactive Pi regression recorded above. +This is the authoritative current contract for the "no turn ends blind" primary backstop referenced from AGENTS.md section 8. +The predicate lives in `bin/fm-turnend-guard.sh`. +Primary scope lives in `bin/fm-primary-scope-lib.sh`, shared with the native session-start nudge in [`sessionstart-nudge.md`](sessionstart-nudge.md). +Harness hook files adapt each enabled primary harness integration's turn-end mechanism to that shared predicate. + +Related PreToolUse guards deny unsafe commands before execution rather than detecting a blind turn end afterward. +Their separate owners are [`arm-pretool-check.md`](arm-pretool-check.md), [`cd-guard.md`](cd-guard.md), and [`subagent-guard.md`](subagent-guard.md). +Do not infer this guard's scope, loop safety, or compatibility tradeoffs for those guards. + +## Current invariant + +`bin/fm-guard.sh` is a pull-based warning that runs only when another supervision command invokes it. +The turn-end guard closes the remaining gap at the primary's own turn boundary. +When work is in flight and no identity-matched watcher has a fresh beacon, the harness integration must either block the turn end or force one bounded follow-up that uses the recovery instruction from the emitted session-start protocol. +The guard remains a backstop; [`watcher-continuity.md`](watcher-continuity.md) owns normal continuity. + +## Shared predicate + +The guard first calls the shared primary scope. +A secondmate home runs its own primary Firstmate session, so a genuine `.fm-secondmate-home` marker includes it whether the home is a linked worktree or plain clone. +The marker must be a regular non-symlink file whose whitespace-stripped first line is a non-empty identifier containing only letters, digits, dots, underscores, and dashes. +An unmarked checkout or invalid marker falls through to the git-dir check. +That check keeps crewmate and scout linked worktrees inert because their git dir differs from their git common dir. +It also requires `AGENTS.md`, `bin/`, and the effective state directory. + +For an in-scope primary, the guard counts in-flight work from `state/*.meta`. +The default cross-harness mode exits silently with no work in flight. +Claude's `--claude` mode also treats `state/x-watch.check.sh` as supervision need, so X-mode relay polling remains guarded without an in-flight task. +Otherwise it calls `fm_watcher_healthy <state-dir> <watch-path> [grace-seconds] [home]` from `bin/fm-wake-lib.sh`, the same identity-matched lock and fresh-beacon check used by `bin/fm-watch-arm.sh`. +A stale beacon blocks even when a watcher pid is live. +A fresh leftover beacon blocks when the lock is missing, dead, or identity-mismatched. + +`FM_STATE_OVERRIDE` wins over `FM_HOME/state`, and `FM_HOME` wins over repository-root `state/`. +`FM_GUARD_GRACE` controls beacon freshness and defaults to 300 seconds. +If `jq` is missing or hook stdin is empty, the guard exits 0 because it cannot safely read loop-guard fields. + +## Harness integrations + +- Claude registers two `Stop` hooks in `.claude/settings.json`, both anchored through `CLAUDE_PROJECT_DIR`: `bin/fm-turnend-guard.sh --claude`, and `bin/fm-claude-stop-autoarm.sh` with `asyncRewake: true` and `timeout: 28800`. +- Codex registers a `Stop` hook in `.codex/hooks.json`, anchors the executable to the hook process working directory, verifies a Firstmate-shaped hook-bearing root, and passes the original payload to the shared guard. +- OpenCode listens for `session.idle` in `.opencode/plugins/fm-primary-turnend-guard.js`, lets the watcher coordinator act first, and calls `client.session.promptAsync` once when the guard returns 2. +- Pi listens for `agent_settled` in `.pi/extensions/fm-primary-turnend-guard.ts`, runs once per logical agent run, and calls `pi.sendUserMessage(..., { deliverAs: "followUp" })` once when the guard returns 2. +- Grok registers a `Stop` hook in `.grok/hooks/fm-primary-turnend-guard.json` and delegates capability selection to `bin/fm-turnend-guard-grok.sh`. + The tracked Claude Stop entries are inert when `GROK_AGENT` is present, so Grok's Claude-compatible settings loading cannot create a second continuation path. + +Claude and Codex can block a Stop directly with exit status 2 and stderr. +Both payloads carry `stop_hook_active`. +In the default Codex mode, a true value lets the second stop finish after one forced continuation. + +Claude runs the guard with `--claude`, which ignores `stop_hook_active` and cooperates with the Stop-owned auto-arm. +Claude Code sets `stop_hook_active=true` on every stop after any stop-hook continuation, including `asyncRewake` rewakes, which re-opened the 2026-07-21 blind window under the default one-shot behavior. +The Claude mode waits up to `FM_CLAUDE_AUTOARM_SYNC_WAIT_MS` (default 800 milliseconds) and allows the stop when the watcher is healthy, `state/.claude-autoarm.lock` has a live owner, or `state/.claude-autoarm-epoch` contains a fresh rewake outcome. +When none of those proofs appears, it re-blocks up to `FM_CLAUDE_TURNEND_BLOCK_BUDGET` times (default 3, below Claude's 8-block override), then allows degraded with a visible `systemMessage`. +Any allow resets the budget. + +OpenCode, Pi, and pi-signed expose passive callbacks for this purpose. +Their adapters fail open at the hook boundary to protect the user session but schedule one bounded follow-up when the predicate blocks. +The generated prompts use the canonical `turn-end-guard` kind after the U+2063 `FIRSTMATE_OP: ` prefix, so Ahoy does not treat them as captain messages. +Each passive adapter owns a loop latch. +Pi keeps the latch across internal tool turns and clears it only when the generated follow-up settles or delivery fails. +OpenCode's forced follow-up is supported for persistent TUI sessions and remains fail-open in headless `opencode run`. + +Grok makes exactly one typed capability decision from each running Stop payload. +A boolean `stopHookActive` selects native blocking, including both false on the initial stop and true on the bounded continuation. +The camel-case field has precedence when both spellings appear; when it is absent, a boolean `stop_hook_active` selects the same native path for compatibility. +The native path returns the shared guard's status and stderr to the same Grok process and never starts `grok --resume`. +When both capability spellings are absent, the adapter preserves one pre-native `grok --resume` fallback guarded by `GROK_TURNEND_GUARD_ACTIVE` and intentionally omits `--permission-mode`. +Malformed JSON, a selected field with a non-boolean type, missing `jq`, missing hook prerequisites, or an already-active legacy guard allows the stop without starting either continuation path. +Grok's project hook requires the checkout to be trusted with `/hooks-trust` or launch-time `--trust`; genuine pre-native builds can run the same tracked hook from an isolated global hook directory. + +If a passive adapter cannot invoke its SDK, or the Grok legacy fallback cannot find `grok` or a session id, the next pull-based `fm-guard.sh` call reports the problem. +That warning uses `bin/fm-supervision-instructions.sh --repair-line`, so it always points to the active harness protocol rather than embedding another repair command. + +## Compatibility limits + +- Child crewmate and scout worktrees are outside scope. +- A valid secondmate home is in scope; an idle secondmate endpoint with no X-mode relay poll remains healthy because it has no supervision need. +- The direct-blocking and bounded passive-follow-up split is limited to the primary integrations listed above. +- OpenCode headless mode and untrusted Grok project hooks remain fail-open at the host boundary. +- Kimi Code CLI 0.29.1 exposes only global `[[hooks]]` configuration in `~/.kimi-code/config.toml`, including a `Stop` event with snake_case payload fields `hook_event_name`, `session_id`, `cwd`, and `stop_hook_active`. +- Kimi has no project-level hook configuration and remains outside the primary guard integrations above. +- Captain-approved Kimi crew wake support uses `bin/fm-kimi-turnend-hook.sh` to edit only one marker-delimited Firstmate region in that global config and install a silent always-zero hook. +- The hook remains inert unless the payload `cwd` contains a per-task token pointer that resolves through Firstmate's private registry to one `state/<id>.turn-ended` marker. +- Installation refuses before writing unless `python3` with `tomllib` and `jq` are available. +- If `jq` is removed after installation, the hook remains silent and exits 0, turn-end wakes stop, and Kimi crews fall back to idle detection. +- Unreadable hook input remains fail-open. +- No harness adapter uses a shell ampersand to manufacture supervision. + +## Regression coverage + +`tests/fm-turnend-guard.test.sh` covers the predicate, main and secondmate primary scope, child-worktree exclusion, `FM_HOME` and `FM_STATE_OVERRIDE` precedence, the cooperative `--claude` claim wait, epoch allow, re-block budget, Pi logical-run latching, missing-`jq` behavior, all five primary registrations, Grok native and legacy selection, typed field precedence, malformed input, and exactly-one-path safety. +`tests/fm-kimi-harness.test.sh` covers the separate Kimi crew hook's format preservation, idempotence, refusal cases, token guard, spawn registration, and teardown cleanup. +`tests/fm-supervision-instructions.test.sh` covers recovery-line ownership and pi-signed's identity-preserving reuse of Pi's protocol. +`FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh` is the opt-in isolated Pi path. +[`verification/supervision.md`](verification/supervision.md#turn-end-guard) records the active cross-harness empirical evidence, including the 2026-07-24 Claude `asyncRewake` revalidation. diff --git a/docs/verification/moshi-mobile-review.md b/docs/verification/moshi-mobile-review.md new file mode 100644 index 00000000000..d604c440447 --- /dev/null +++ b/docs/verification/moshi-mobile-review.md @@ -0,0 +1,67 @@ +# Moshi mobile review verification + +Audience: maintainer verification. + +This record covers the current Firstmate-owned mobile presentation and review handoff boundary. +Moshi product behavior was checked against its official Browser Preview, Diff, Chat View, and Hooks documentation on 2026-07-31. +The operator contract is [`docs/moshi-mobile-review.md`](../moshi-mobile-review.md), and the agent contract is [`mobile-mode`](../../.agents/skills/mobile-mode/SKILL.md). + +## Test boundary + +The natural-language trigger and captain-facing message shape are agent behavior, not a shell executable. +That is an explicit agent-behavior test exception: do not add a test that parses or asserts instruction source bytes. +Deterministic validation covers the maintained-prose inventory, local links, repository lint surface, and changed-file-selected behavior suite. +Fresh-context dogfood covers whether an agent loads the public `AGENTS.md` trigger and produces the required mobile handoff. +Live mobile execution status: NOT RUN. No Moshi or Browser Preview session was available to this worker, so no Preview, Diff, Chat View, or fallback end-user evidence is claimed; the complete captain-run checklist remains in [`docs/moshi-mobile-review.md`](../moshi-mobile-review.md). + +## Moshi facts in scope + +The official [Browser Preview documentation](https://getmoshi.app/docs/browser-preview) says host-local HTTP servers are detected by `moshi-hook` and reached in-app through the active SSH-capable session without a public tunnel. +The official [Diff documentation](https://getmoshi.app/docs/diff-viewer) says Diff reads the connected host's staged, unstaged, and untracked working-tree state and keeps diff contents host-local. +The official [Chat View documentation](https://getmoshi.app/docs/chat-view) says Chat View renders the same live agent session, currently lists Claude Code, Codex CLI, OpenCode, and Pi, and requires tmux or Herdr. +The official [Hooks documentation](https://getmoshi.app/docs/hooks) says hook support is broader than Chat View support and that inbox summaries and approval routing are separate from host-local transcript, diff, source-file, and terminal traffic. + +## Supported harness review + +The supported Firstmate harness list comes from [`harness-adapters`](../../.agents/skills/harness-adapters/SKILL.md). + +| Firstmate harness | Mobile presentation | Existing Moshi hook surface | Chat View | Firstmate adapter change | +| --- | --- | --- | --- | --- | +| Claude | Applies through the shared agent contract. | Official Moshi hooks support exists; configuration is external and unchanged. | Listed by Moshi. | None. | +| Codex | Applies through the shared agent contract. | Official Moshi hooks support exists; configuration is external and unchanged. | Listed by Moshi. | None. | +| OpenCode | Applies through the shared agent contract. | Official Moshi hooks support exists; configuration is external and unchanged. | Listed by Moshi. | None. | +| Pi | Applies through the shared agent contract. | Official Moshi hooks support exists, but Firstmate's Pi has no permission system, so approval authority is not applicable. | Listed by Moshi. | None. | +| pi-signed | Applies through the shared agent contract. | It uses the Pi engine, but wrapper-specific Moshi detection is not independently verified. | Use concise numbered Firstmate chat in the same Moshi/Firstmate session unless Moshi recognizes it as Pi. | None. | +| Grok | Applies through the shared agent contract. | Official Moshi hooks support exists; configuration is external and unchanged. | Not in the current official Chat View list, so use concise numbered Firstmate chat in the same Moshi/Firstmate session. | None. | +| Kimi | Applies through the shared agent contract. | Official Moshi hooks support exists; configuration is external and unchanged. | Not in the current official Chat View list, so use concise numbered Firstmate chat in the same Moshi/Firstmate session. | None. | + +The repository's Claude, Codex, OpenCode, Pi, Grok, and Kimi hook or extension surfaces were inspected for ownership overlap. +This slice changes none of them and makes no compatibility claim about the captain's external Moshi-managed hook configuration. + +## Supported runtime backend review + +The supported spawn backend list comes from `FM_BACKEND_SPAWN` in [`bin/fm-backend.sh`](../../bin/fm-backend.sh). + +| Firstmate backend | Moshi mobile path | Applicability to this slice | +| --- | --- | --- | +| tmux | Moshi Chat View currently supports tmux; Browser Preview and Diff use the host gateway. | No backend behavior changes. | +| Herdr | Selected Firstmate mobile path; Moshi Chat View currently supports Herdr; Browser Preview and Diff use the host gateway. | No backend behavior changes. | +| Zellij | Moshi can detect Zellij for terminal context, but current Chat View requirements exclude it. | Terminal, Browser Preview, and Diff may remain usable; Chat View falls back to concise numbered Firstmate chat in the same Moshi/Firstmate session; no backend behavior changes. | +| Orca | No Firstmate-to-Moshi session integration is claimed. | Not applicable to the selected Herdr workflow; no backend behavior changes. | +| cmux | No Firstmate-to-Moshi session integration is claimed. | Not applicable to the selected Herdr workflow; no backend behavior changes. | +| Codex App | Firstmate does not accept it as a runtime backend. | Not applicable. | + +Moshi remains absent from the backend registry by design. +The mobile contract changes presentation and review handoff only, so spawn, supervision, recovery, cleanup, and backend metadata need no new branch. + +## Verification entry points + +Run the focused maintained-prose check and the repository's changed-file-selected canonical behavior entrypoint: + +```sh +bin/fm-doc-audience-check.sh +bin/fm-test-run.sh --changed +``` + +If shell surfaces change in a future extension, also run `bin/fm-lint.sh` and the relevant focused test script. +This slice changes no shell surface, but the delivery pipeline still owns its canonical lint gate. diff --git a/docs/verification/runtime-backends.md b/docs/verification/runtime-backends.md new file mode 100644 index 00000000000..65152100f4e --- /dev/null +++ b/docs/verification/runtime-backends.md @@ -0,0 +1,411 @@ +# Runtime backend verification + +Audience: maintainer verification. + +This record contains reusable version-scoped evidence for active runtime guarantees. +The backend guides own current setup, safety boundaries, and limitations. +Exact task chronology, branch names, temporary homes, local paths, process ids, thread ids, and delivery transcripts remain in private reports or PR evidence. + +## tmux + +Foreground-process behavior was verified on 2026-07-07 with tmux 3.6a on macOS. + +```sh +tmux new-session -d -s fmtest -n testwin +tmux display-message -p -t fmtest:testwin '#{pane_current_command}' +tmux send-keys -t fmtest:testwin 'sleep 30' Enter +tmux display-message -p -t fmtest:testwin '#{pane_current_command}' +tmux send-keys -t fmtest:testwin C-c +tmux display-message -p -t fmtest:testwin '#{pane_current_command}' +``` + +Observed output: + +```text +zsh +sleep +zsh +``` + +A persistent parent shell waiting for a child remained reported as the parent process, while a shell that directly execed a simple command changed identity with the process itself. +Claude, Codex, OpenCode, and Grok were observed under their own process names. +Kimi Code CLI 0.29.1 was observed under `kimi` on 2026-07-25. +Pi and pi-signed 0.82.0 were reverified on 2026-07-27 through real isolated `fm-spawn.sh` launches. + +Installed-wrapper checks: + +```sh +basename "$(command -v pi-signed)" +pi-signed --version +pi --version +``` + +Observed bounded output: + +```text +pi-signed +0.82.0 +0.82.0 +``` + +The isolated process and endpoint checks used: + +```sh +tmux display-message -p -t "$target" '#{pane_current_command}' +ps -o comm= -p "$wrapper_pid" +ps -o comm= -p "$engine_pid" +FM_HOME="$fixture_home" bin/fm-crew-state.sh "$task_id" +``` + +Observed bounded shapes: + +```text +pi-launcher +.../pi-signed +.../Pi Launcher.app/Contents/Resources/pi/pi +state: done ... +``` + +Both launches executed a submitted tool instruction and touched the generated `turn_end` marker. +The pi-signed launch retained `harness=pi-signed`, while the plain comparison retained `harness=pi`. +The exact wrapper ancestry was `pi-signed` parent to Pi engine child, and the plain Pi Launcher path also traversed the signed wrapper on this installation. +That shared plain-Pi path is retained as disconfirming evidence against using ancestry as runtime-selection authority. +Firstmate therefore sets the exact `FM_PI_HARNESS` selection marker on both worker launch paths, while an unmarked Pi-family process remains `pi`. +Both recorded runtime identities now classify the exact `pi-launcher` foreground command as `alive`. + +Backend applicability was reviewed across every spawn adapter. +Tmux needs the exact `pi-launcher`, `pi-signed`, `pi`, and `Pi` process identities for recovery-grade liveness. +Herdr uses native registered-agent state and needs no process-name branch. +Zellij has no verified recovery-grade agent process probe, while Orca and cmux do not support secondmate spawns, so those three retain their existing generic ordinary-launch semantics without a new liveness matcher. + +The structural multi-row composer reader, Kimi pointer-delivery path, and OpenCode 1.18.4 busy-queue behavior are pinned by: + +```sh +tests/fm-composer-ghost.test.sh +tests/fm-kimi-harness.test.sh +tests/fm-tmux-submit-busy.test.sh +``` + +Expected structural matrix: real text on any content row is pending; all-empty complete boxes are empty; unreadable, incomplete, or unsafe boxes are unknown; and non-bordered panes retain cursor-row compatibility. +Expected submit matrix: proven pending plus busy is accepted as queued; proven pending plus idle remains pending; ambiguous pending is never converted by the busy exception; and only a proven empty composer succeeds directly. + +### Cleanup endpoint identity + +The cleanup identity boundary was validated on 2026-07-28 with tmux 3.6a and metadata fixtures for every supported backend. + +```sh +tests/fm-teardown-endpoint-safety.test.sh +tests/fm-teardown.test.sh +tests/fm-backend-herdr.test.sh +tests/fm-backend-zellij.test.sh +tests/fm-backend-orca.test.sh +tests/fm-backend-cmux.test.sh +``` + +Bounded output from the incident regression: + +```text +ok - fm-teardown: missing, empty, malformed, ambiguous, and task-mismatched endpoints refuse before every mutation or runtime call +ok - cleanup identity: valid tmux, Herdr, Zellij, Orca, and cmux records validate while every empty backend target refuses +ok - tmux backend: direct empty target returns nonzero without invoking tmux +ok - process cleanup: creation-time PID identity removes only the exact child and preserves the control child +ok - fm-teardown: dedicated-socket invalid cleanup preserves target/control and valid cleanup removes only the exact target +``` + +The dedicated tmux cell removed ambient tmux variables, required a socket-bound wrapper, kept one target and one independent control window, and proved the wrapper was not called for invalid metadata or a direct empty target. +Valid cleanup removed only the exact task-bound target and left the control window live. +The metadata-only validation covers tmux, Herdr, Zellij, Orca, and cmux before backend dispatch. +Claude, Codex, OpenCode, Pi, pi-signed, Grok, and Kimi share that backend cleanup boundary; their harness-specific hook files and token cleanup run only after it, so no harness needs a separate endpoint parser. + +## Herdr + +The compatibility floor is protocol 14. +The latest active verification uses Herdr 0.7.5 protocol 16 on macOS aarch64, with earlier 0.7.4, protocol-14, and 0.7.3 evidence retained where they define current behavior or fallbacks. + +Core read-only probes: + +```sh +herdr --version +herdr status --json | jq -c '{client:.client.protocol,server:.server.protocol}' +herdr api schema --json | jq -c '.schemas.subscription_event["$defs"].SubscriptionEventKind.enum' +``` + +Observed current shapes: + +```text +herdr 0.7.5 +{"client":16,"server":16} +["pane.output_matched","pane.agent_status_changed","pane.scroll_changed"] +``` + +The CLI matrix was checked directly: + +| Guarantee | Command shape | Result | +| --- | --- | --- | +| Explicit session routing | `herdr <verb> ... --session <name>` | Reached the named session even while another server was running. | +| Literal send | `herdr pane send-text <pane> <text> --session <name>` | Left text unsubmitted until Enter. | +| Keys | `herdr pane send-keys <pane> enter|escape|ctrl+c --session <name>` | Enter and Escape worked; Ctrl-C interrupted foreground work. | +| Capture | `herdr pane read <pane> --source recent --lines N` | Small N could return empty below viewport height; a 200-line request plus local trim was stable. | +| Native state | `herdr agent get <pane>` | Working and done transitions were visible; long foreground tool waits required rendered-busy corroboration. | +| Restart | guarded named-session stop then start | Workspace, tab, pane, and labels persisted; the agent process and registration did not. | +| Close | `herdr pane close <pane> --session <name>` | The exact one-pane task tab closed; closing a final tab could remove the workspace. | + +All destructive verification used `bin/fm-herdr-lab.sh` with a non-default `fm-lab-` name and a byte-identical default-session tripwire. +No ambient `herdr server stop` command is a supported test operation. + +### Prune and respawn + +The real label-collision reproduction is owned by: + +```sh +HERDR_LAB_HELPER=bin/fm-herdr-lab.sh \ + tests/fm-backend-herdr-prune-safety-e2e.test.sh +``` + +Observed guarantee: a pre-existing captain-owned workspace with a seed-shaped tab was adopted for routing but its tab was never eligible for prune because the current create call did not return that seed id. + +Restart-husk replacement is owned by: + +```sh +HERDR_LAB_HELPER=bin/fm-herdr-lab.sh \ + tests/fm-backend-herdr-respawn-idem-e2e.test.sh +``` + +Observed guarantee: a restored no-agent tab was replaced create-before-close, while a registered live agent caused refusal. + +### Per-home and presentation topology + +Per-home behavior is owned by: + +```sh +HERDR_LAB_HELPER=bin/fm-herdr-lab.sh \ + tests/fm-backend-herdr-workspace-per-home-e2e.test.sh +``` + +Observed guarantee: the primary and secondmate used distinct home workspaces, a child launched by the secondmate stayed in that secondmate workspace, list-live remained home-scoped, and exact cleanup did not affect sibling homes. + +The complete projection suite ran on 2026-07-21 against Herdr 0.7.4 protocol 16: + +```sh +HERDR_LAB_HELPER=bin/fm-herdr-lab.sh \ + tests/fm-backend-herdr-presentation-e2e.test.sh +``` + +Observed guarantees included: + +```text +ok - real Herdr lab: primary and two secondmate homes each own a top-level contiguous child block +ok - real Herdr lab: concurrent primary/A/B spawns stay session-locked with zero focus drift +ok - real Herdr lab: session lock contention from a secondmate home falls back flat with no journal +ok - real Herdr lab: legacy projection labels and flat secondmate tabs are left unmigrated +ok - real Herdr lab: multi-home exact-pane teardowns restore captain focus without workspace close authority +ok - real Herdr lab validation completed on Herdr 0.7.4 with the default-session tripwire intact +``` + +The suite also covers lost or failed move responses, active-tab refusal, restart husks, missing and duplicate tokens, manual renames, concurrent cleanup, and exact focus restoration. + +The mandatory projection suite ran again on 2026-07-24 against Herdr 0.7.5 protocol 16: + +```sh +HERDR_LAB_HELPER=bin/fm-herdr-lab.sh \ + tests/fm-backend-herdr-presentation-e2e.test.sh +``` + +Observed restart-reclaim guarantees: + +```text +ok - real Herdr lab: Hi Bit and Wheelhouse-style same-identity restarts reclaim one nested space with exact focus and idempotence +ok - real Herdr lab: secondmate restart binding and reclaim stay isolated to the exact child home and parent +ok - real Herdr lab: concurrent cross-home recoveries replace exact husks under one session lock with no focus drift +ok - real Herdr lab: missing, renamed, and duplicate tokens trigger zero destructive or adoptive calls, and live duplicate risk refuses launch +ok - real Herdr lab validation completed on Herdr 0.7.5 with the default-session tripwire intact +``` + +The restored-shell session-start cleanup ran on 2026-07-24 against Herdr 0.7.5 protocol 17: + +```sh +HERDR_LAB_HELPER=bin/fm-herdr-lab.sh \ + tests/fm-herdr-session-cleanup-e2e.test.sh +``` + +Observed guarantee: one exact home-local, journal-correlated, one-tab and one-pane childless idle shell was closed after restoration while the exact non-target focus and default fleet session remained unchanged, and a repeat run was a no-op. + +### Composer and operational input + +Real captures verified these active distinctions: + +- Claude and Codex use bare `❯` and `›` agent composers. +- Pi uses content between complete separator rows and requires exact native Pi identity. +- Dim or faint suggestion text is ghost content, while normally styled text is pending input. +- Grok dark truecolor placeholders are ghost content, while bright truecolor typed input remains pending. +- A bare shell prompt has no safe agent-composer container and is unknown. + +`tests/fm-composer-ghost.test.sh`, `tests/fm-composer-lib.test.sh`, and the Herdr composer cases pin the exact captured ANSI bytes. +The U+2063 operational and routed-request separators were exercised through a real Pi-on-Herdr path; the byte-exact active regression is: + +```sh +FM_SEND_MARKER_HERDR_E2E=1 \ + tests/fm-send-secondmate-marker-herdr-e2e.test.sh +``` + +### Native blocked event + +The protocol-16 event path was measured on 2026-07-11 with Herdr 0.7.3 and Python 3.13: + +```sh +HERDR_LAB_HELPER=bin/fm-herdr-lab.sh \ + tests/fm-backend-herdr-eventwait-smoke.test.sh +``` + +Observed output: + +```text +ok - real herdr: events.subscribe capability gate passes +ok - real herdr: a driven idle->blocked transition returns the blocked record in 0.129s +ok - real herdr: the watcher fast-path enqueues a stale wake naming the task window +``` + +Polling remained active and is covered as the fallback for capability, connect, subscribe, and repeated reader failure. + +### Away-mode transport + +The Pi/Herdr return and injection path was reverified on Herdr 0.7.3 and Pi 0.80.7: + +```sh +FM_AFK_PI_HERDR_E2E=1 HERDR_LAB_HELPER=bin/fm-herdr-lab.sh \ + tests/fm-afk-pi-herdr-return-e2e.test.sh +``` + +Observed guarantees: pending composer input refused injection and raised one alert; idle Pi accepted one marked escalation; the return gate refused ordinary work while a live blocker remained; resolving the blocker allowed the return flow. +The dedicated Herdr daemon workspace topology is covered by `tests/fm-afk-launch.test.sh` and preserves the captain tab's pane count. + +## Zellij + +The current compatibility floor and latest verification are Zellij 0.44.0 with `jq` on macOS aarch64. +All real tests use a uniquely named session and `tests/zellij-test-safety.sh`; they never touch a session named `firstmate` or call all-session deletion. + +| Guarantee | Command shape | Result | +| --- | --- | --- | +| Headless session | `zellij attach -b <name>` without a TTY | Created a persistent background session and returned. | +| Session list | `zellij list-sessions --short --no-formatting` | Returned one plain name per line without starting a session. | +| Create tab | `zellij action new-tab --cwd <dir> --name <title>` | Returned a numeric tab id and focused the new tab when a client was attached. | +| Pane discovery | `zellij action list-panes --json` | Included terminal pane id, tab id, plugin flag, and top-level `pane_cwd`. | +| Literal send | `zellij action paste --pane-id <id> -- <text>` | Left text unsubmitted. | +| Keys | `send-keys --pane-id <id> Enter`, `Esc`, and one argument `Ctrl c` | All three shared operations worked. | +| Capture | `dump-screen --pane-id <id>` or `--full` | Worked with no attached client; no line-bound flag exists. | +| Close | `close-tab-by-id <id>` | Removed the live task pane and tab together. | +| Failure exit | actions against missing targets | Returned exit 0, requiring structural preflight and output-shape validation. | + +`pane_cwd` stayed frozen when a foreground subshell changed directory. +The marker-delimited `pwd` probe returned the live nested cwd and is covered by the real smoke. +The focus mitigation restored the previously active tab after `new-tab`, with the unavoidable narrow race documented in the operator guide. + +```sh +tests/fm-backend-zellij.test.sh +tests/fm-backend-zellij-smoke.test.sh +``` + +The real lifecycle smoke proved spawn, metadata, nested-subshell worktree discovery, send, capture, unlanded-work refusal, approved local landing, exact tab cleanup, and session cleanup without retaining task-specific ids or branch names here. + +## Orca + +Real readiness was verified against `/usr/local/bin/orca` with `/Applications/Orca.app` bundle version 1.4.116. + +```sh +orca status --json +``` + +Observed fields: + +```text +result.runtime.reachable=true +result.runtime.state=ready +``` + +`orca terminal create --json` returned `result.terminal.handle`. +`orca worktree create` returned `result.worktree.id` and `result.worktree.path`. +Speculative bare ids and nested terminal fields were deliberately rejected. + +```sh +tests/fm-backend-orca.test.sh +tests/fm-backend.test.sh +tests/fm-bootstrap.test.sh +``` + +The fake-Orca suite covers readiness, registration, create response parsing, metadata routing, popup-safe submit, and path-matched release refusal. + +## cmux + +The current compatibility floor is cmux 0.64, and the active live evidence uses 0.64.17 build 97 on macOS aarch64. +Real tests use only exact `fm-test-` workspaces guarded by `tests/cmux-test-safety.sh` and never quit or relaunch the captain's app. + +```sh +cmux version +cmux ping +``` + +Observed version: + +```text +cmux 0.64.17 (97) [9ed29d81a] +``` + +Source and live checks established the five control modes: + +- `off` starts no listener. +- `cmuxOnly` rejects an external Firstmate process by ancestry. +- `automation` uses an owner-only 0600 socket with no handshake. +- `password` uses the same 0600 socket plus `auth <password>`. +- `allowAll` uses a 0666 socket with no authentication. + +The live default rejection was `Access denied - only processes started inside cmux can connect`. +The live password challenge was `Authentication required - send auth <password> first`. +The app configuration writer did not retain a hand-added socket password, which is why the operator guide requires Settings and a local Firstmate password source. + +Current active CLI findings: + +| Guarantee | Command shape | Result | +| --- | --- | --- | +| Create | `new-workspace --name <title> --cwd <dir> --focus false --id-format uuids` | Created one workspace with one surface without focusing it. | +| Fresh readiness | `list-panes --workspace <id> --json --id-format uuids` | Found a brand-new surface before content existed. | +| Fresh read counterexample | `read-screen` before any write | Returned `internal_error: Failed to read terminal text`. | +| Literal send | `send --workspace <id> --surface <id> -- <text>` | Left text unsubmitted. | +| Keys | `send-key ... enter|escape|ctrl-c` | All shared key operations worked. | +| Nested cwd | `current_directory` plus foreground subshell | Structured cwd froze; the marker-delimited `pwd` probe found the live cwd. | +| Last surface | `close-surface` on the only surface | Refused with `invalid_state: Cannot close the last surface`. | +| Last workspace | `close-workspace` on the only workspace in a window | Printed success but left the workspace present. | + +The last-workspace workaround was reverified on 2026-07-10 in Automation mode. +After creating one unfocused unnamed sibling in the same window, `close-workspace` removed the exact task workspace and left only cmux's default sibling. +A selected non-last workspace closed directly, proving that window cardinality rather than selection is the trigger. + +Source inspection confirmed each workspace constructor creates a new UUID with no restored-id input. +Recovery therefore remains title-based. +The bundled Claude wrapper was observed stripping `CMUX_*` variables on its failed socket-probe path while retaining the app bundle id, supporting the macOS-only bundle-id and ancestry fallbacks. + +```sh +tests/fm-backend-cmux.test.sh +tests/fm-backend-cmux-smoke.test.sh +``` + +The real smoke proves socket access, fresh readiness, current-path probing, send and keys, bounded capture, title identity, and guarded exact cleanup. + +## Codex App host tools + +A reusable Desktop host-tool smoke ran on 2026-07-06 against Codex Desktop bundle version 26.623.101652, build 4674, bundle id `com.openai.codex`. +Local paths and task-specific ids are intentionally not retained here. + +The host-tool sequence was: + +1. list a saved project; +2. create a Desktop-owned worktree thread; +3. recover and read the thread while active and after completion; +4. verify the thread appended a Firstmate status line and wrote its report; +5. send a follow-up to the same thread; +6. read the completed follow-up; +7. archive the exact thread; +8. read the archived transcript with state `notLoaded`. + +Observed guarantee: a Desktop-owned thread can write Firstmate lifecycle files when the prompt provides an authorized absolute path, and create, send, read, and archive work at the Desktop host-tool layer. +The missing guarantee remains a supported shell-callable bridge that lets Firstmate perform those operations against the same visible Desktop endpoint. +App-server partial methods and raw socket experiments do not satisfy that bridge contract. diff --git a/docs/verification/stow-memory.md b/docs/verification/stow-memory.md new file mode 100644 index 00000000000..39e61eac4fb --- /dev/null +++ b/docs/verification/stow-memory.md @@ -0,0 +1,217 @@ +# Startup-memory `/stow` verification + +Audience: maintainer verification. + +This record supports the active bounded-memory and whole-file curation guarantees for Firstmate's internal `/stow` skill. +[`docs/configuration.md`](../configuration.md) owns the current operator-facing setting and estimate. +The internal skill owns curation and completion-receipt behavior. +Task chronology, fixture paths, and delivery evidence remain outside this record. + +## Synthetic real-agent pass + +The development-only real-agent pass ran on 2026-07-30 with Pi 0.82.0 on `openai-codex/gpt-5.6-terra` at medium thinking. +It used disposable primary and secondmate-shaped `FM_HOME` directories under the repository worktree only. +No live Firstmate memory, project data, credential content, or external system was placed in either fixture or prompt. +The following exact Bash shell body created the sanitized fixtures, invoked the model-qualified skill twice per home, and captured reports, hashes, and file modes: + +```bash +set -eu +VERIFY_ROOT=$(mktemp -d "$PWD/.stow-verification.XXXXXX") +RUNTIME_ROOT="$VERIFY_ROOT/runtime-root" +PRIMARY="$VERIFY_ROOT/primary" +SECONDMATE="$VERIFY_ROOT/secondmate" +SECONDMATE_ID=stow-verification +mkdir -p "$RUNTIME_ROOT" "$PRIMARY/config" "$PRIMARY/data" \ + "$SECONDMATE/bin" "$SECONDMATE/config" "$SECONDMATE/data" +printf '%s\n' 350 >"$PRIMARY/config/startup-memory-budget" +printf '%s\n' "$SECONDMATE_ID" >"$SECONDMATE/.fm-secondmate-home" +printf '%s\n' '# Synthetic Firstmate home' >"$SECONDMATE/AGENTS.md" + +file_mode() { + if [ "$(uname)" = Darwin ]; then + stat -f %Lp "$1" + else + stat -c %a "$1" + fi +} + +record_shared_state() { + label=$1 + path=$2 + printf '%s sha256=%s mode=%s\n' "$label" \ + "$(shasum -a 256 "$path" | awk '{print $1}')" \ + "$(file_mode "$path")" +} + +cat >"$PRIMARY/data/captain.md" <<'EOF' +# Captain + +## Current preferences + +- Prefer the simplest direct end-to-end operational path. +- Preserve unique current facts when compacting memory. +- Use plain dashes in prose. + +## Duplicate and superseded material + +- Prefer the simplest direct end-to-end operational path. +- Old policy: build a wrapper before every one-off operation. +- Old policy copy: always build a wrapper for one-off work. +- Stale tool path: `/opt/old-firstmate/bin/fm`. +- Stale release version: 0.41.0. +- Completed task: migrated the demo fixture on Monday. +- Completed task detail: checked the demo fixture again on Tuesday. +- Metric from the completed task: 47 records moved. +EOF + +cat >"$PRIMARY/data/captain-shared.md" <<'EOF' +# Shared captain preferences + +This file is main-authoritative in the main firstmate home. +In secondmate homes it is read-only in secondmate homes and must not be edited there. +Route new captain-preference discoveries to the main firstmate through marked status or a document pointer. + +- Never expose secrets or weaken an accepted safety boundary. +- Prefer the simplest direct end-to-end operational path. +- Superseded policy: secondmates may rewrite shared memory when convenient. +- Duplicate safety note: do not expose secrets. +EOF + +cat >"$PRIMARY/data/learnings.md" <<'EOF' +# Learnings + +- Stable fact: startup-memory configuration is documented in `docs/configuration.md`. +- Authoritative pointer: incident detail belongs in `data/reports/synthetic-incident.md`. +- Stable fact copy: consult `docs/configuration.md` for startup-memory configuration. +- Completed chronology: first the synthetic incident was detected, then triaged, then assigned. +- Completed chronology continued: a patch was drafted, reviewed, merged, and announced. +- Old metric: the discarded prototype used 812 estimated tokens. +- Stale path: the discarded prototype lived at `/tmp/old-memory-prototype`. +- Superseded alternative: maintain both a JSON memory database and Markdown files. +- Report-sized procedure: create a staging directory, enumerate every file, copy each file, compare every line, write a status ledger, notify all operators, archive the ledger, and repeat the entire sequence after every prompt. +EOF + +FM_HOME="$PRIMARY" bin/fm-startup-memory-budget.sh report \ + >"$VERIFY_ROOT/primary.before.report" +for file in captain.md captain-shared.md learnings.md; do + shasum -a 256 "$PRIMARY/data/$file" +done >"$VERIFY_ROOT/primary.before.sha256" + +FM_HOME="$PRIMARY" pi -p --no-session --no-extensions --no-context-files \ + --model openai-codex/gpt-5.6-terra --thinking medium \ + --skill .agents/skills/stow/SKILL.md \ + 'Invoke /stow now against only the disposable synthetic Firstmate home in $FM_HOME. There are no new session facts to file. Follow every requirement in the loaded stow skill. Run the repository-owned bin/fm-startup-memory-budget.sh report command, with the existing FM_HOME environment, before and after curation; that executable is the only permitted path outside $FM_HOME. Retain the exact before total, preserve the complete main-authoritative routing header in data/captain-shared.md, and make the completion receipt state the effective budget, exact before and after totals, an action for each of the three files, every exception, and reset safety. Inspect all three startup-memory files completely, preserve every unique current preference, authority or safety boundary, stable fact, and authoritative pointer, and consolidate the supplied duplicate, superseded, stale, chronological, metric, and report-sized material. Do not access or modify any other home, credential, project data, or external system.' \ + >"$VERIFY_ROOT/primary.pass1.out" +FM_HOME="$PRIMARY" bin/fm-startup-memory-budget.sh report \ + >"$VERIFY_ROOT/primary.after.report" +for file in captain.md captain-shared.md learnings.md; do + shasum -a 256 "$PRIMARY/data/$file" +done >"$VERIFY_ROOT/primary.after.sha256" + +FM_HOME="$PRIMARY" pi -p --no-session --no-extensions --no-context-files \ + --model openai-codex/gpt-5.6-terra --thinking medium \ + --skill .agents/skills/stow/SKILL.md \ + 'Invoke /stow now against only the disposable synthetic Firstmate home in $FM_HOME. There are no new session facts to file. Follow every requirement in the loaded stow skill. Run the repository-owned bin/fm-startup-memory-budget.sh report command, with the existing FM_HOME environment, before and after curation; that executable is the only permitted path outside $FM_HOME. Retain the exact before total, preserve the complete main-authoritative routing header in data/captain-shared.md, and make the completion receipt state the effective budget, exact before and after totals, an action for each of the three files, every exception, and reset safety. Inspect all three startup-memory files completely, preserve every unique current preference, authority or safety boundary, stable fact, and authoritative pointer, and consolidate the supplied duplicate, superseded, stale, chronological, metric, and report-sized material. Do not access or modify any other home, credential, project data, or external system.' \ + >"$VERIFY_ROOT/primary.pass2.out" +FM_HOME="$PRIMARY" bin/fm-startup-memory-budget.sh report \ + >"$VERIFY_ROOT/primary.repeat.report" +for file in captain.md captain-shared.md learnings.md; do + shasum -a 256 "$PRIMARY/data/$file" +done >"$VERIFY_ROOT/primary.repeat.sha256" + +cat >"$SECONDMATE/data/captain.md" <<'EOF' +# Secondmate captain memory + +- Current preference: report concrete blockers instead of guessing. +- Current preference copy: never guess when a concrete blocker can be reported. +- Shared overlap: never expose secrets. +- Superseded preference: silently infer missing configuration. +- Stale version: the fleet uses 0.41.0. +- Completed task: inspected the synthetic queue yesterday. +- Completed task detail: closed the synthetic queue inspection after 19 checks. +EOF + +cat >"$SECONDMATE/data/learnings.md" <<'EOF' +# Secondmate learnings + +- Unique current learning: inherited shared memory counts against the local total. +- Authoritative pointer: startup-memory behavior is documented in `docs/configuration.md`. +- Duplicate learning: include inherited shared memory in the local total. +- Stale path: `/tmp/secondmate-memory-v1`. +- Superseded alternative: copy shared facts into every local file. +- Completed chronology: opened the sample, measured it, discussed it, revised it, remeasured it, and closed it. +- Old metric: the sample once measured 604 estimated tokens. +- Report-sized procedure: take a snapshot, copy it to a ledger, annotate every old measurement, preserve every discarded alternative, append a timestamp, and repeat after each completed task. +EOF + +FM_ROOT="$RUNTIME_ROOT" +FM_HOME="$PRIMARY" +. bin/fm-ff-lib.sh +. bin/fm-config-inherit-lib.sh +validate_secondmate_home "$SECONDMATE_ID" "$SECONDMATE" +printf 'secondmate_validation=accepted id=%s home=%s\n' \ + "$SECONDMATE_ID" "$VALIDATED_HOME" >"$VERIFY_ROOT/inheritance.out" +FM_CONFIG_INHERIT_REPORT="$VERIFY_ROOT/inheritance.report" \ + propagate_secondmate_inheritance \ + "$PRIMARY" "$VALIDATED_HOME" "$PRIMARY/config" "$PRIMARY/data" +cat "$VERIFY_ROOT/inheritance.report" >>"$VERIFY_ROOT/inheritance.out" +cmp -s "$PRIMARY/data/captain-shared.md" \ + "$SECONDMATE/data/captain-shared.md" +record_shared_state inherited "$SECONDMATE/data/captain-shared.md" \ + >>"$VERIFY_ROOT/inheritance.out" + +FM_HOME="$SECONDMATE" bin/fm-startup-memory-budget.sh report \ + >"$VERIFY_ROOT/secondmate.before.report" +for file in captain.md captain-shared.md learnings.md; do + shasum -a 256 "$SECONDMATE/data/$file" +done >"$VERIFY_ROOT/secondmate.before.sha256" +record_shared_state before "$SECONDMATE/data/captain-shared.md" \ + >"$VERIFY_ROOT/secondmate.shared-state" + +FM_HOME="$SECONDMATE" pi -p --no-session --no-extensions --no-context-files \ + --model openai-codex/gpt-5.6-terra --thinking medium \ + --skill .agents/skills/stow/SKILL.md \ + 'Invoke /stow now against only the validated disposable synthetic secondmate home in $FM_HOME. There are no new session facts to file. Follow every requirement in the loaded stow skill. Run the repository-owned bin/fm-startup-memory-budget.sh report command, with the existing FM_HOME environment, before and after curation; that executable is the only permitted path outside $FM_HOME. Retain the exact before total, and make the completion receipt state the effective budget, exact before and after totals, an action for each of the three files, every exception, and reset safety. Inspect all three startup-memory files completely, keep data/captain-shared.md byte-identical and filesystem read-only because it was installed through primary-authoritative inheritance, preserve every unique current preference, stable learning, and authoritative pointer, and consolidate the supplied duplicate, superseded, stale, chronological, metric, overlap, and report-sized material in editable local memory. Do not access or modify any other home, credential, project data, or external system.' \ + >"$VERIFY_ROOT/secondmate.pass1.out" +FM_HOME="$SECONDMATE" bin/fm-startup-memory-budget.sh report \ + >"$VERIFY_ROOT/secondmate.after.report" +for file in captain.md captain-shared.md learnings.md; do + shasum -a 256 "$SECONDMATE/data/$file" +done >"$VERIFY_ROOT/secondmate.after.sha256" +record_shared_state after "$SECONDMATE/data/captain-shared.md" \ + >>"$VERIFY_ROOT/secondmate.shared-state" + +FM_HOME="$SECONDMATE" pi -p --no-session --no-extensions --no-context-files \ + --model openai-codex/gpt-5.6-terra --thinking medium \ + --skill .agents/skills/stow/SKILL.md \ + 'Invoke /stow now against only the validated disposable synthetic secondmate home in $FM_HOME. There are no new session facts to file. Follow every requirement in the loaded stow skill. Run the repository-owned bin/fm-startup-memory-budget.sh report command, with the existing FM_HOME environment, before and after curation; that executable is the only permitted path outside $FM_HOME. Retain the exact before total, and make the completion receipt state the effective budget, exact before and after totals, an action for each of the three files, every exception, and reset safety. Inspect all three startup-memory files completely, keep data/captain-shared.md byte-identical and filesystem read-only because it was installed through primary-authoritative inheritance, preserve every unique current preference, stable learning, and authoritative pointer, and consolidate the supplied duplicate, superseded, stale, chronological, metric, overlap, and report-sized material in editable local memory. Do not access or modify any other home, credential, project data, or external system.' \ + >"$VERIFY_ROOT/secondmate.pass2.out" +FM_HOME="$SECONDMATE" bin/fm-startup-memory-budget.sh report \ + >"$VERIFY_ROOT/secondmate.repeat.report" +for file in captain.md captain-shared.md learnings.md; do + shasum -a 256 "$SECONDMATE/data/$file" +done >"$VERIFY_ROOT/secondmate.repeat.sha256" +record_shared_state repeat "$SECONDMATE/data/captain-shared.md" \ + >>"$VERIFY_ROOT/secondmate.shared-state" +``` + +Bounded observed output: + +```text +secondmate_validation=accepted id=stow-verification +startup-memory-budget pushed +data/captain-shared.md pushed +inherited sha256=d08ce8e35b17c8342773d551b5c1551a5a6ded5f45ab0f7ed5b6ef91ea1d408c mode=444 +primary: 699 -> 219 estimated tokens against a 350-token budget +primary repeat: 219 -> 219; all three files byte-identical +secondmate: 518 -> 192 estimated tokens against a 350-token budget +secondmate repeat: 192 -> 192; all three files byte-identical +before sha256=d08ce8e35b17c8342773d551b5c1551a5a6ded5f45ab0f7ed5b6ef91ea1d408c mode=444 +after sha256=d08ce8e35b17c8342773d551b5c1551a5a6ded5f45ab0f7ed5b6ef91ea1d408c mode=444 +repeat sha256=d08ce8e35b17c8342773d551b5c1551a5a6ded5f45ab0f7ed5b6ef91ea1d408c mode=444 +``` + +The first pass preserved current preferences, shared-memory and safety authority, a stable operating fact, and authoritative configuration and incident-report pointers while removing duplicate, superseded, stale, and chronological material. +The secondmate fixture passed the production home validator before the existing inheritance owner installed the main-authoritative file read-only. +Both secondmate passes preserved its unique local preference and learning while leaving those inherited bytes and mode untouched. +This verifies the real instruction path consolidates to budget, reports truthful deltas, preserves the primary-owned shared boundary, and does not grow on an identical second pass. diff --git a/docs/verification/supervision.md b/docs/verification/supervision.md new file mode 100644 index 00000000000..20f415e4564 --- /dev/null +++ b/docs/verification/supervision.md @@ -0,0 +1,198 @@ +# Supervision integration verification + +Audience: maintainer verification. + +This record supports current session-start, turn-end, watcher-continuity, and wedge-alarm guarantees. +Operator behavior and active limits remain in the linked current guides. +Task-specific chronology, temporary paths, run identifiers, and delivery transcripts remain in private reports or PR evidence. + +## Native session-start delivery + +The cross-harness transport pass ran on 2026-07-17 with Codex 0.144.4, Grok 0.2.103, OpenCode 1.17.18, Pi 0.80.10, and the tracked Claude hook wiring. + +Codex command shape: + +```sh +codex exec --ephemeral --dangerously-bypass-hook-trust \ + --dangerously-bypass-approvals-and-sandbox \ + --output-last-message last.txt \ + 'Follow any SessionStart hook context before this prompt.' +``` + +Observed result: the `SessionStart` hook completed and its stdout reached model context. + +Grok command shape: + +```sh +grok --trust -p 'Follow any SessionStart hook context before this prompt.' \ + --permission-mode bypassPermissions --output-format plain +``` + +Observed result: the project hook ran, but its stdout did not reach model context. +This is the current Grok fail-open limit. + +OpenCode was checked in both headless and interactive modes. +`client.session.promptAsync` accepted the nudge in both cases; the persistent TUI completed the generated turn, while `opencode run` exited before another turn. +This is the current headless fail-open limit. + +Pi command shape: + +```sh +pi -p -e .pi/extensions/fm-primary-turnend-guard.ts \ + --no-context-files --no-session \ + 'After obeying any earlier session-start instruction, reply with exactly PI_SMOKE_DONE.' +``` + +Observed result: `PI_SMOKE_DONE`, with one session-start execution. +The earlier `sendUserMessage` counterfactual raced the positional prompt; the current non-triggering `pi.sendMessage` custom message did not. +The installed pi-signed 0.82.0 wrapper repeated the Pi primary extension and session-start path on 2026-07-27. +[`runtime-backends.md`](runtime-backends.md#tmux) owns the shared-ancestry evidence and authoritative selection-marker boundary. + +Current deterministic and live entry points: + +```sh +tests/fm-sessionstart-nudge.test.sh +FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh +FM_OPENCODE_LIVE_E2E=1 tests/fm-opencode-primary-live-e2e.test.sh +``` + +The Ahoy first-message boundary was reverified on 2026-07-22 with Pi 0.81.1 and OpenCode 1.17.18. +Marked current operational input and the two exact legacy compatibility shapes selected Bearings, while genuine near-miss captain messages remained real boundaries. +The detailed reconciliation and task chronology stay in the private audit report and PR evidence. + +## Turn-end guard + +The direct and passive mechanisms were validated across all five harnesses on 2026-07-08 through 2026-07-12, with Claude's replacement Stop-owned path revalidated on 2026-07-24. + +| Harness | Version verified | Mechanism | Observed result | +| --- | --- | --- | --- | +| Claude | 2.1.219 | Cooperative blocking `Stop` guard plus `asyncRewake` auto-arm | A fresh unsupervised session ran session start first, reclaimed a stale dead-owner lock, completed two tokenless rewake cycles with no model arm command or guard continuation, and left a competing live owner unchanged. | +| Codex | 0.142.1 | Blocking `Stop` hook | Hook process root stayed anchored to the trusted checkout and one continuation ran. | +| OpenCode | 1.17.6 | Passive `session.idle` callback | Throwing could not block, while `promptAsync` scheduled one TUI follow-up; headless remained fail-open. | +| Pi | 0.80.5 | Passive `agent_settled` callback | Exactly one guard follow-up ran for an unhealthy cycle, with no recursion across tool turns. | +| Grok | 0.2.112 native and 0.2.73 pre-native | Running-payload adaptive `Stop` | Native false-to-true continuation stayed in one process with two model turns and zero resume launches; the field-absent pre-native process launched exactly one guarded resume. | + +The Grok adaptive matrix ran on 2026-07-28 with separate scratch repositories and homes, dedicated tmux sockets, one target plus one control window, ambient tmux variables removed, and a socket-bound wrapper first in `PATH`. + +```sh +FM_GROK_STOP_LIVE_E2E=1 \ + FM_GROK_NATIVE_BIN="$native_grok_0_2_112" \ + FM_GROK_LEGACY_BIN="$official_pre_native_grok_0_2_73" \ + tests/fm-grok-stop-live-e2e.test.sh +``` + +Observed bounded output: + +```text +ok - grok 0.2.112 (9bbd559437aa) [stable] native Stop kept one session across false->true, two model turns, and zero resume processes +ok - grok 0.2.73 (9ff14c43bbe5) [stable] legacy Stop omitted capability, resumed exactly once, and stopped normally +ok - Grok adaptive Stop real-process matrix passed with exact target cleanup and control-window survival +``` + +The same run proved the Claude-compatible Stop entries stay inert under `GROK_AGENT`, the legacy resume carries `GROK_TURNEND_GUARD_ACTIVE=1`, and every replacement root is removed after exact target cleanup while its control window survives. + +The secondmate-home scope and manual-repair wake path were measured with Claude Code 2.1.207 on 2026-07-12, when a native background completion re-invoked the idle model with no human input. +The current Stop-owned main/secondmate inclusion and child-worktree exclusion are covered deterministically by `tests/fm-claude-stop-autoarm.test.sh`. +On 2026-07-28 with Claude Code 2.1.205, `fm_harness_ancestry_pid()` in `bin/fm-session-lock-lib.sh` was fixed to resolve the outermost pid of a contiguous nested-harness run instead of the first match, so the Stop auto-arm correctly reaches the session's true lock owner through Claude Code's multi-level `bg-spare` hook worker chain. + +The Claude product live path ran with Claude Code 2.1.219 on 2026-07-24: + +```sh +claude --version +FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-stop-autoarm-live-e2e.test.sh +``` + +Observed output: + +```text +2.1.219 (Claude Code) +ok - Claude 2.1.219 (Claude Code) live E2E reclaimed a stale session lock through session start, completed two tokenless Stop-owned rewake cycles, and preserved the competing-live-owner boundary +``` + +Current entry points: + +```sh +tests/fm-turnend-guard.test.sh +tests/fm-supervision-instructions.test.sh +FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh +FM_GROK_STOP_LIVE_E2E=1 FM_GROK_NATIVE_BIN="$native_grok" FM_GROK_LEGACY_BIN="$pre_native_grok" tests/fm-grok-stop-live-e2e.test.sh +``` + +## Watcher continuity + +The cross-harness evidence combines the 2026-07-17 live pass with Claude's replacement Stop-owned path revalidated on 2026-07-24, all against isolated project and home state. +No credential material was copied into a fixture. + +```text +Claude Code 2.1.219 +codex-cli 0.144.4 +OpenCode 1.17.18 +Pi 0.80.10 +grok 0.2.103 (89c3d36fb6f1) [stable] +``` + +| Harness | Exact opt-in command | Observed guarantee | +| --- | --- | --- | +| Claude | `FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-stop-autoarm-live-e2e.test.sh` | Session start reclaimed a stale owner before two Stop-owned cycles, and a competing live owner prevented arm, rewake, epoch write, or lock replacement. | +| Codex | `FM_CODEX_LIVE_E2E=1 tests/fm-codex-continuity-live-e2e.test.sh` | The one-second foreground checkpoint returned without switching to the arm wrapper. | +| OpenCode | `FM_OPENCODE_LIVE_E2E=1 tests/fm-opencode-primary-live-e2e.test.sh` | A verified successor existed before prompt handling, with no model re-arm or turn-end fallback. | +| Pi | `FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh` | One initial tool call led to extension-owned successors and clean child retirement on exit. | +| Grok | `FM_GROK_LIVE_E2E=1 tests/fm-grok-continuity-live-e2e.test.sh` | Native task completion surfaced the actionable close and the cycle ledger recorded `reason=actionable-signal`. | + +Pi 0.81.1 repeated the continuity and clean-exit lifecycle on 2026-07-23 after the Calm presentation changes. + +Pi same-process session-transition ownership was verified on 2026-07-27 against the tracked extension with a faithful in-process factory rebind (module cache retained, real arm children): + +```sh +pi --version +tests/fm-pi-watch-extension.test.sh +tests/fm-pi-primary-types.test.sh +``` + +Observed guarantee: after ordinary `session_shutdown` for `/new`, `/resume`, and `/fork`, plus same-instance shutdown-plus-start, the replacement generation armed again without a Pi restart and without the `watcher: not armed - Pi session is shutting down` refusal. +Stale prior-generation tool callbacks could not mutate the active child, repeated transitions kept exactly one live arm cycle, and terminal `quit` still refused late rearm. +Plain Pi and pi-signed share the same tracked `.pi/extensions/fm-primary-pi-watch.ts` path, so both inherit the generation owner; other primary harnesses are not applicable because they do not use this Pi extension lifecycle. + +Deterministic entry points: + +```sh +tests/fm-pi-watch-extension.test.sh +tests/fm-pi-primary-types.test.sh +tests/fm-watcher-lock.test.sh +tests/fm-subagent-pretool-check.test.sh +tests/fm-claude-stop-autoarm.test.sh +tests/fm-turnend-guard.test.sh +``` + +## Wedge-alarm channels + +The two real notification channels were bounded manually on 2026-07-10 on macOS 26.5.2 with Herdr 0.7.3. +Automated suites never execute these real notification commands. + +Argv-safe Notification Center command: + +```sh +/usr/bin/osascript \ + -e 'on run argv' \ + -e 'display notification (item 1 of argv) with title "FIRSTMATE TEST - IGNORE" sound name "Basso"' \ + -e 'end run' \ + 'FIRSTMATE TEST - IGNORE (wedge-alarm channel verification)' +``` + +Observed output: no stdout, exit 0, and one banner with the supplied body. + +Herdr command: + +```sh +herdr notification show 'FIRSTMATE TEST - IGNORE' \ + --body 'FIRSTMATE TEST - IGNORE (wedge-alarm channel verification)' \ + --sound request +``` + +Observed output: + +```json +{"id":"cli:notification:show","result":{"reason":"shown","shown":true,"type":"notification_show"}} +``` + +The safe command-channel contract is covered without a notification by `tests/fm-daemon.test.sh`: the summary reaches both `$1` and stdin, every channel is process-group bounded, and a failed channel falls through. diff --git a/docs/watcher-continuity.md b/docs/watcher-continuity.md index 7fef13b2394..52b3a9eec3a 100644 --- a/docs/watcher-continuity.md +++ b/docs/watcher-continuity.md @@ -8,6 +8,12 @@ Must-work continuity now lives above that process boundary instead of depending Pi's `.pi/extensions/fm-primary-pi-watch.ts` and OpenCode's `.opencode/plugins/fm-primary-watch-arm.js` own continuous re-arm after an actionable child close. Each adapter starts the next arm before delivering the wake prompt, checks current session-lock ownership at launch, preserves one child or scheduled retry at a time, and applies bounded exponential retry after an unexpected or failed close. A failed follow-up never cancels continuity restoration. +Pi same-process session replacement follows the generation-owner contract in `.pi/extensions/fm-primary-pi-watch.ts`. +Claude's `.claude/settings.json` Stop `asyncRewake` hook (`bin/fm-claude-stop-autoarm.sh`) owns routine tokenless re-arm. +The hook fires on every Stop, and an eligible primary with supervision need admits one home-scoped owner that foregrounds `bin/fm-watch-arm.sh` inside the hook-owned process tree. +A numeric session-lock owner that fails the shared `fm_harness_pid_alive` predicate is reclaimed through `bin/fm-lock.sh` before auto-arm state changes, while a live owner, absent lock, or malformed lock keeps the competing hook inert. +The stale-owner claim occurs only after the existing AFK and supervision-need gates pass. +While supervision is still needed and away mode remains inactive, an actionable close or typed failure wakes the idle session through exit 2. ## Actionable wake ordering @@ -18,15 +24,16 @@ When that retained arm later closes, its actual close is classified as a new sup After the configured retry bound is exhausted, it delivers the original wake with a typed continuity-restoration failure even if every successor arm hung without reporting readiness. This is deliberate Option B ordering: the fleet is protected before the model handles the wake whenever restoration succeeds, but the model is never left blind when it does not. -Claude retains its native tracked background-task completion path. -Its new PreToolUse continuity gate allows wake drain, arm recovery, and independently fail-closed teardown, but refuses other fleet commands while tasks are in flight and no identity-matched live watcher holds the home lock. -Allowing an ordinary literal teardown prevents a terminal wake from creating a recovery circle: forced or dynamically constructed teardown remains blocked, ordinary teardown itself still refuses dirty, unlanded, incomplete-scout, and unresolved-decision cases, and the turn-end guard continues to require supervision for any tasks left in flight. +Claude's Stop hook starts the successor arm at the next Stop after the handling turn, rather than before notification as Pi and OpenCode do. +The durable wake queue preserves actionable events during the residual active-turn window, and the unchanged bounded turn-end guard enforces recovery at Stop when no watcher or auto-arm claim is present. +No PreToolUse hook denies fleet commands based on watcher status. +The model no longer re-arms after ordinary wakes. +Terminal arm-output classification (`started`, `attached`, or `FAILED`) remains defense in depth for the manual recovery path. Codex retains its bounded foreground checkpoint protocol. Grok retains its tracked background-task notification protocol. No adapter starts a replacement with shell `&`. -The existing turn-end guard implementation and adapters are unchanged. -They remain the final backstop rather than the normal continuity mechanism. +The turn-end guard remains the final backstop rather than the normal continuity mechanism and cooperates with the auto-arm in its `--claude` mode. ## Arm-layer cycle contract @@ -46,49 +53,18 @@ Only the watcher process touches `state/.last-watcher-beat`; no helper process c ## Regression coverage `tests/fm-pi-watch-extension.test.sh` checks Pi's first-cycle-or-explicit-repair tool metadata and ownership-based redundant-call no-ops, then simulates actionable and empty child closes against the actual Pi and OpenCode close handlers, blocks prompt delivery to prove the successor launches first, verifies single-flight behavior, changes the session lock before close to prove ownership is rechecked, and hangs each successor arm to prove bounded fallback delivery includes the typed restoration failure. +The same suite covers ordinary same-process session replacement for `/new`, `/resume`, and `/fork`, same-instance shutdown-plus-start, stale prior-generation callbacks, repeated transitions with exactly one live cycle, disappearance of the shutting-down refusal after a valid replacement activates, and terminal quit still refusing late rearm. `tests/fm-watcher-lock.test.sh` covers verified-successor attach, the typed self-eviction failure, bounded and successor-linked lifecycle rows, and a SIGSTOP counterfactual that distinguishes a live PID from a stale beacon before classifying termination. -`tests/fm-continuity-pretool-check.test.sh` proves the Claude gate rejects only non-recovery fleet execution in the precise unhealthy state and preserves the existing Stop registration. - -## Sanitized live evidence, 2026-07-17 - -All five harnesses ran against git-initialized scratch projects and isolated `FM_HOME` state. -Existing harness-managed credentials remained in place, no credential bytes were copied into a fixture or transcript, and no account was created. -Pi used the existing shared Pi auth store with the explicit `openai-codex/gpt-5.6-sol` provider/model pin and low thinking. -Each run used the smallest prompt needed to exercise the harness-native path. - -Harness versions: - -```text -Claude Code 2.1.214 -codex-cli 0.144.4 -OpenCode 1.17.18 -Pi 0.80.10 -grok 0.2.103 (89c3d36fb6f1) [stable] -``` - -Claude ran an arm fixture through its native tracked background option, observed background completion, allowed the wake drain, and refused the next unrelated fleet command before its body executed. -The captured system message exactly named `[watcher-continuity]`, `bin/fm-wake-drain.sh`, tracked Claude re-arm through `bin/fm-watch-arm.sh`, and the blocked `fm-crew-state.sh` command. -Command: `FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-continuity-live-e2e.test.sh`. -Observed result: `ok - Claude 2.1.214 (Claude Code) live E2E refused only the post-completion fleet command with exact re-arm guidance`. - -Codex ran the real one-second foreground watcher checkpoint and returned `checkpoint: no actionable wake within 1s` without switching to the arm wrapper. -Command: `FM_CODEX_LIVE_E2E=1 tests/fm-codex-continuity-live-e2e.test.sh`. -Observed result: `ok - codex-cli 0.144.4 live E2E preserved the one-second foreground checkpoint path`. - -OpenCode ran its persistent TUI plugin, established the first watcher from `session.idle`, received an actionable close, and ledger-linked a live successor before the model handled the wake. -The model executed no watcher-arm command and the turn-end backstop did not fire. -Command: `FM_OPENCODE_LIVE_E2E=1 tests/fm-opencode-primary-live-e2e.test.sh`. -Observed result: `ok - OpenCode 1.17.18 live E2E auto-started one successor before prompt handling without a model re-arm`. - -Pi loaded the tracked extensions in its interactive TUI, called `fm_watch_arm_pi` once, received an actionable close, and ledger-linked a successor before the handling turn ended. -The turn-end backstop did not fire, and `/quit` removed both the watcher and arm child. -Command: `FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh`. -Observed result: `ok - Pi 0.80.10 live E2E used shared Codex auth, auto-started one successor before turn end, and cleaned up`. - -Grok ran the real arm wrapper through `run_terminal_command` with its tracked background option, surfaced its native task-completion notification after the actionable close, and recorded `reason=actionable-signal` in the cycle ledger. -No shell ampersand was used. -Command: `FM_GROK_LIVE_E2E=1 tests/fm-grok-continuity-live-e2e.test.sh`. -Observed result: `ok - grok 0.2.103 (89c3d36fb6f1) [stable] live E2E preserved tracked background completion and shared ledger classification`. - -The goal is continuity with fewer supervision tokens and no Pi/OpenCode model-memory re-arm step. -No zero-latency guarantee is claimed; lock verification, watcher startup, and bounded retry delays remain deliberate safety work. +`tests/fm-subagent-pretool-check.test.sh` proves Claude retains only the non-status Bash seatbelts. +`tests/fm-claude-stop-autoarm.test.sh` covers the auto-arm's scope, stale and live session owners, unchanged AFK and need boundaries, single-flight, and exit-2 translation. +`FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-stop-autoarm-live-e2e.test.sh` starts with the reproduced stale-lock state, runs session start first, completes two tokenless cycles, and checks the competing-live-owner negative control. +`tests/fm-turnend-guard.test.sh` covers the cooperative `--claude` guard. + +## Active limits and verification + +The goal is continuity without a Pi or OpenCode model-memory re-arm step. +No zero-latency guarantee is claimed because lock verification, watcher startup, and bounded retry delays remain deliberate safety work. +OpenCode support targets persistent TUI sessions rather than headless `opencode run`. +Claude depends on the Stop `asyncRewake` rewake, Grok retains native background-completion notifications, and Codex retains bounded foreground checkpoints. + +[`verification/supervision.md`](verification/supervision.md#watcher-continuity) records the current five-harness live evidence, the 2026-07-24 Stop-owned Claude auto-arm results, and exact opt-in commands. diff --git a/docs/wedge-alarm.md b/docs/wedge-alarm.md index b8a3bf27912..cfee3784b42 100644 --- a/docs/wedge-alarm.md +++ b/docs/wedge-alarm.md @@ -1,87 +1,39 @@ -# Away-mode injection wedge alarm - active alert channels +# Away-mode injection wedge alarm -The away-mode sub-supervisor (`bin/fm-supervise-daemon.sh`) buffers escalations and injects them into firstmate's own pane. -When injection cannot confirm a submit past `FM_MAX_DEFER_SECS` (the pane is genuinely busy or wedged, or its Enter is swallowed), `inject_wedge_alarm` raises a loud, rate-limited alarm so the stall never stays invisible. - -## Why an active channel beyond the status-line flash - -Before this change the only ACTIVE signal `inject_wedge_alarm` sent was a tmux `display-message` status-line flash, guarded by `if [ "$backend" = tmux ]`. -That flash is a client-side OSD with no cross-backend equivalent, so on every non-tmux supervisor backend it was skipped entirely. -On 2026-07-10 a `claude`-on-`herdr` primary wedged past max-defer overnight: the tmux flash was skipped, and only the passive `state/.subsuper-inject-wedged` marker was written. -Nothing surfaces that marker until the next fleet action, so 20 escalations sat buffered for roughly 8.5 hours with no active alert. -The classifier-side half of that incident shipped separately (PR #429); this is the alarm-channel half. - -`inject_wedge_alarm` now also calls `wedge_alarm_notify`, a configurable active alert that does not depend on any pane or its backend status-line. -The durable marker and the tmux flash are unchanged; the active alert is added alongside them. +The away-mode sub-supervisor (`bin/fm-supervise-daemon.sh`) buffers escalations and injects them into Firstmate's own pane. +When injection cannot confirm a submit past `FM_MAX_DEFER_SECS`, `inject_wedge_alarm` raises a loud, rate-limited alarm so the stall never stays invisible. +The active alert is pane-independent because a tmux status-line flash has no cross-backend equivalent and cannot reach an unattended captain reliably. +The durable marker and tmux flash remain as additional signals. ## Channels -`config/wedge-alarm` (local, gitignored) lists channel directives, one per non-empty, non-comment line; every listed non-`off` channel fires, best-effort. -`FM_WEDGE_ALARM_CHANNEL` overrides the file with a single directive (used by the tests). - -- `off` - position-independent kill switch that disables every active alert; the marker and tmux flash remain. -- `auto` / `default` - platform default. macOS resolves to `osascript`; other platforms have no built-in OS channel, so `auto` there fires nothing and logs that the durable marker is the only signal (configure a `command:` directive instead). -- `osascript` - a macOS Notification Center banner via `osascript`. OS-level, so it reaches the captain even when every pane and its status-line is unreadable. -- `herdr` - a herdr UI notification via `herdr notification show`. herdr's own surface, separate from the pane and its status-line. -- `command:<cmd>` - run `<cmd>` via `sh -c`, with the alarm summary passed as `$1` and on stdin. Lets the alert reach a phone or pager (ntfy, Slack, SMS) even when the captain is away from the machine entirely. - -An absent `config/wedge-alarm` behaves as `auto`, i.e. default-on on macOS. -Default-on is deliberate: the alarm's entire purpose is that a wedged away-mode primary is never silent, so the reachable OS channel fires unless the captain explicitly disables it. -The alarm is rate-limited to at most once per max-defer window, and fires only after a genuine wedge past max-defer, so the default-on banner is rare and never chatty. - -Each channel is best-effort: a missing binary or a non-zero exit logs a warning and the alarm falls through to the next channel, never crashing the daemon loop. -Every invocation is also process-group bounded by `FM_WEDGE_ALARM_TIMEOUT_SECS` (10 seconds by default), including `command:`, `osascript`, `herdr`, and an `FM_WEDGE_ALARM_EXEC` override. -On timeout or daemon shutdown, its watchdog terminates the notifier group, logs the timeout when applicable, and continues to the next configured channel. -The AppleScript passes the summary as an `argv` item rather than interpolating it into the script source, so summary text can never break the notification. -See `docs/examples/wedge-alarm` for a copyable starting config. - -## Test safety: no test posts a real notification - -Every notifier channel (`osascript`, `herdr`, and `command:`) routes through a single seam, `FM_WEDGE_ALARM_EXEC`: when it is set, the daemon hands the fixed channel category and summary to that command instead of the real notifier (`wedge_alarm_emit` in `bin/fm-supervise-daemon.sh`). -This makes it structurally impossible for a test to post a real desktop notification, and impossible for a future test author to forget to stub: - -- The daemon is only ever sourced (not executed) by tests - production `bin/fm-afk-start.sh` execs it. - Whenever the daemon is sourced, its library-mode guard defaults `FM_WEDGE_ALARM_EXEC` to `discard`, which fires nothing. - A real daemon a test later spawns inherits that default through the environment. -- `tests/wake-helpers.sh` upgrades the default to an on-disk recorder that logs `<channel>\t<summary>` to `$FM_WEDGE_ALARM_LOG`, so the daemon and wake suites can assert channel selection without any real notifier. -- Production leaves `FM_WEDGE_ALARM_EXEC` unset, so the real channels fire. - -Because of this seam, the automated tests verify channel selection and summary propagation only. -The real `osascript`/`herdr` invocation form is verified once by the single bounded manual run below, never from a suite. - -## Verification (macOS, darwin) - -Recorded 2026-07-10T12:41-0700 on macOS 26.5.2 (build 25F84), `osascript` at `/usr/bin/osascript`, `herdr` 0.7.3. -This is the single bounded manual verification (two invocations, one per OS channel), labelled "FIRSTMATE TEST - IGNORE" so the banners are unmistakably harmless. -These are the only verification commands that fire real notifications, and they are never run inside a test suite. - -### osascript channel (the exact argv-safe form the daemon runs) - -``` -$ /usr/bin/osascript -e 'on run argv' \ - -e 'display notification (item 1 of argv) with title "FIRSTMATE TEST - IGNORE" sound name "Basso"' \ - -e 'end run' "FIRSTMATE TEST - IGNORE (wedge-alarm channel verification)" -$ echo $? -0 -``` +`config/wedge-alarm` is local and gitignored. +It lists channel directives, one per non-empty, non-comment line, and every listed non-`off` channel fires best-effort. +`FM_WEDGE_ALARM_CHANNEL` overrides the file with one directive for focused testing. -Exit 0; a Notification Center banner titled "FIRSTMATE TEST - IGNORE" was posted with the label as its body. -In production the title is "firstmate: away-mode escalations WEDGED" and the body is the `<age>s undelivered - see <marker>` summary. +- `off` disables every active alert while retaining the durable marker and tmux flash. +- `auto` or `default` resolves to `osascript` on macOS. + Other platforms have no built-in OS channel, so configure `command:` when a durable marker alone is insufficient. +- `osascript` posts a macOS Notification Center banner outside the terminal pane. +- `herdr` calls `herdr notification show` outside the supervised pane. +- `command:<cmd>` runs `<cmd>` through `sh -c` with the alarm summary as `$1` and on stdin, allowing delivery to a phone or pager service. -### herdr channel +An absent `config/wedge-alarm` behaves as `auto`, which is default-on on macOS. +This is deliberate because the alarm fires only after a genuine max-defer wedge and is rate-limited to at most once per max-defer window. -``` -$ herdr notification show "FIRSTMATE TEST - IGNORE" \ - --body "FIRSTMATE TEST - IGNORE (wedge-alarm channel verification)" --sound request -{"id":"cli:notification:show","result":{"reason":"shown","shown":true,"type":"notification_show"}} -$ echo $? -0 -``` +Each channel is best-effort. +A missing binary or non-zero exit logs a warning and continues to the next channel without crashing the daemon loop. +Every invocation is process-group bounded by `FM_WEDGE_ALARM_TIMEOUT_SECS`, which defaults to 10 seconds, including `command:`, `osascript`, `herdr`, and the test seam. +On timeout or daemon shutdown, the notifier process group is terminated and the next configured channel may run. +AppleScript receives the summary as an argv item rather than interpolated source, so summary text cannot alter the script. +See [`examples/wedge-alarm`](examples/wedge-alarm) for a copyable config. -Exit 0; herdr reported `"shown":true`. -The daemon redirects this stdout to `/dev/null` and treats a zero exit as success. +## Test safety -### command channel dispatch (summary on $1 and stdin) +Every notifier routes through `FM_WEDGE_ALARM_EXEC` in `wedge_alarm_emit`. +When the daemon is sourced as a library, that seam defaults to `discard`, so a test cannot accidentally post a real notification. +`tests/wake-helpers.sh` replaces it with a recorder when a suite needs to assert channel selection and summary propagation. +Production leaves the seam unset and uses the configured real channels. -The `command:` channel runs `sh -c "<cmd>" fm-wedge-alarm "<summary>"` with the summary also piped on stdin. -`test_wedge_alarm_command_channel_receives_summary` deliberately unsets the seam for a safe file-writing command to verify this dispatch contract without a notification. +`tests/fm-daemon.test.sh` covers directive parsing, rate limiting, timeout and process-group cleanup, argv-safe dispatch, channel fallback, and safe `command:` summary delivery. +[`verification/supervision.md`](verification/supervision.md#wedge-alarm-channels) records the bounded manual macOS and Herdr channel proof. diff --git a/docs/zellij-backend.md b/docs/zellij-backend.md index 697686c77dc..367da98ebbd 100644 --- a/docs/zellij-backend.md +++ b/docs/zellij-backend.md @@ -1,224 +1,109 @@ -# Zellij runtime backend (experimental) +# Zellij runtime backend -This document records the empirical verification behind `bin/backends/zellij.sh`, the zellij session-provider adapter added in P3 of the runtime-backend abstraction. -It is the zellij equivalent of the tmux facts recorded in the `harness-adapters` skill and of `docs/herdr-backend.md`'s herdr facts. - -Zellij is [a terminal multiplexer](https://zellij.dev) with a CLI action interface (`zellij action <subcommand>`) for scripted control of sessions, tabs, and panes. -Verified against the real installed binary: zellij 0.44.0, macOS aarch64. -All real-zellij verification in this document and in `tests/fm-backend-zellij-smoke.test.sh` uses isolated, uniquely-named sessions (via `FM_ZELLIJ_SESSION`) plus the guarded teardown helper in `tests/zellij-test-safety.sh` - never the real `firstmate` session name a live fleet would use, and never `kill-all-sessions`/`delete-all-sessions`. +Zellij is an experimental explicit-only session backend. +It provides the terminal session while Treehouse continues to provide task worktrees. +[`configuration.md`](configuration.md#runtime-backend-configbackend--fm_backend) owns shared selection and metadata semantics. ## Setup -Pick zellij if you already use it as your terminal multiplexer and want firstmate crew windows there instead of tmux; it has no per-home container split, so it is simpler than herdr for a single-home fleet. +Pick Zellij when you already use it as a terminal multiplexer and accept its current focus, liveness, and polling limits. Prerequisites: -- `zellij` itself, version 0.44 or newer (installed 0.44.0 verified) - see [zellij.dev](https://zellij.dev) for install instructions. -- `jq`, required to parse zellij's JSON output: `brew install jq` (or your platform's package manager). -- The universal firstmate prerequisites - a verified crew harness plus the required toolchain, owned by [`docs/configuration.md`](configuration.md) ("Harness support", "Toolchain"); treehouse still provides the worktree, zellij only provides the session. - -Select zellij by putting `zellij` in a local `config/backend` file - the durable way to pick it - or by exporting `FM_BACKEND=zellij` when you launch your harness for a one-off session; telling the first mate in chat to use zellij also works. -Unlike tmux and herdr, zellij is **never** auto-detected - it always requires an explicit choice. -A zellij spawn refuses loudly before creating a session container or acquiring a ship/scout worktree if `zellij` or `jq` is missing or the installed zellij is older than 0.44. -For `--secondmate` launches, secondmate home sync and inherited local-material propagation happen before this spawn-time backend gate. - -No first-run provisioning is needed beyond having `zellij` and `jq` on `PATH`; firstmate creates the session and tab it needs on first spawn. - -Watching and attaching: firstmate uses one shared session (default name `firstmate`, overridable with `FM_ZELLIJ_SESSION`) with one tab per task. -The tab's caller-facing label is always `fm-<id>`, but its actual visible title is home-scoped - `fm-<home-label>-<id>`, e.g. `fm-firstmate-a1b2c3d4-fix-login-k3` - so that two firstmate homes sharing this one session (a primary plus a secondmate, two secondmates, or two independent primary installations on the same machine) never collide on the tab bar even if their task ids happen to match; see "Home-scoped tab titles" below. -Attach to the selected `FM_ZELLIJ_SESSION` (or the default `firstmate` session) with `zellij attach <name>` to see every task, primary or secondmate, as a tab in that one tab bar. -You do not need to attach for routine supervision: from an active firstmate session, `bin/fm-peek.sh fm-<id>` reads a task's pane without attaching, and `FM_HOME=<this-firstmate-home> bin/fm-send.sh fm-<id> "<text>"` steers it unless `FM_HOME` is already set to the active firstmate home. - -Verify it works by spawning a trivial task with `--backend zellij` and confirming the task's meta records `backend=zellij` plus `zellij_session=`, `zellij_tab_id=`, and `zellij_pane_id=`; attaching to the session should show the new home-scoped tab title, such as `fm-firstmate-<8hex>-<id>`. - -Limitations: zellij is experimental, has no per-home workspace split (all tasks share one tab bar, unlike herdr), has no verified agent-process liveness classifier for the session-start secondmate sweep, and still carries the known gaps documented below (no native busy-state signal and a narrow focus-steal race on tab creation) - see "Known gaps left for a follow-up" at the end of this document. - -## Status: experimental - -Zellij is experimental, exactly like every non-tmux backend in this design. -Select it by putting `zellij` in a local `config/backend` file, by exporting `FM_BACKEND=zellij`, or by telling the first mate in chat to use zellij. -Unlike tmux and herdr, zellij is **never** selected by runtime auto-detection: the design report's Open Question #2 recommends starting with a dedicated background session for predictability rather than reusing whatever zellij session firstmate itself might be running inside, and empirical verification below (see "Focus-steal on new-tab") confirms that recommendation was correct - reusing an ambient session a human might be attached to would risk yanking their view on every spawn. -Absent `backend=` in a task's meta always means `tmux`; only a zellij task ever carries an explicit `backend=zellij` line. -A zellij spawn refuses loudly if `zellij` or `jq` is missing, or if the installed zellij's version is older than the verified minimum, 0.44 (`fm_backend_zellij_version_check`). - -## Worktree provider stays treehouse - -Zellij is a session provider only (D3, `data/fm-backend-design-d7/herdr-addendum.md`, restated for zellij in the same task). -Treehouse remains the worktree provider, exactly as it is for tmux and herdr. - -## Task container shape: one session, one tab per task - -Per the design report's "Zellij implementation choices" #1, unchanged by empirical verification: firstmate uses **one** zellij session (default name `firstmate`, overridable via `FM_ZELLIJ_SESSION` for test isolation - mirrors herdr's `HERDR_SESSION`) and **one tab per task**, whose caller-facing label is `fm-<id>` (its actual, home-scoped tab title is described in "Home-scoped tab titles" below). -This is deliberately simpler than herdr's later workspace-per-firstmate-home refinement (`docs/herdr-backend.md` "Task container shape"): zellij has no workspace concept at all, only sessions/tabs/panes, so there is no analogous per-home container to split - primary and secondmate tasks share the one `firstmate` session's tab bar, distinguished only by their tab titles, exactly as the original P1/P2 tmux-parity shape worked before herdr's per-home split existed. -No empirical evidence surfaced during verification that forces a different container shape; the report's original choice stands - only the tab TITLE gained a per-home discriminator, not the container. - -## Home-scoped tab titles (cross-home collision fix) - -Because every task in every firstmate home - primary or secondmate - shares this ONE session's tab bar with no per-home container split, and zellij enforces no tab-name uniqueness at all (verified: two tabs can share a name), two firstmate homes whose task ids happen to collide could send/peek/close each other's tabs. -This is the exact gap a captain-directed no-mistakes review gate caught for the cmux backend (`docs/cmux-backend.md` "Task container shape") - cmux's fix was ported here for the identical reason, sharing its tag-derivation code (`bin/fm-backend-hometag-lib.sh`). - -The caller-facing task label stays `fm-<id>` in meta and briefs; task-selector resolution is the shared contract owned by [`docs/configuration.md`](configuration.md) ("Runtime backend"). -The actual zellij tab title a NEW task's tab is created with is home-scoped: `fm-<home-label>-<id>`. -`<home-label>` is `firstmate` for the primary home, or `2ndmate-<id>` when `$FM_HOME/.fm-secondmate-home` contains a secondmate id, plus a short stable hash of the resolved `FM_ROOT` path - the same identity scheme as cmux's home label (`docs/cmux-backend.md` "Task container shape"), so e.g. `fm-firstmate-a1b2c3d4-fix-login-k3` or `fm-2ndmate-sm1-9f8e7d6c-fix-login-k3`. -The path hash means even two independent PRIMARY installations on one machine (each with no `.fm-secondmate-home` marker, so both would otherwise resolve to the same `firstmate` prefix) still get distinct tags. -`fm_backend_zellij_create_task` creates every new tab with this scoped title and checks for a duplicate against the scoped title, never the bare label. -Every list/find/recover/kill path (`fm_backend_zellij_target_ready`'s and `fm_backend_zellij_kill`'s expected-label verification, `fm_backend_zellij_list_live`'s recovery sweep, `fm_backend_zellij_resolve_bare_selector`'s ad hoc lookup) is scoped the same way: it checks the home-scoped title first and never trusts a bare, unscoped title match against another home's tab. - -**Migration posture for tasks spawned before this change.** A tab created before this home-scoping shipped still carries its old, untagged bare title (`fm-<id>`, no home tag). -Rather than silently orphaning every already-running zellij task, the adapter's label-verification path (`fm_backend_zellij_tab_matches_label`, used by both `target_ready` and `kill`) falls back to an exact untagged bare-title match - but ONLY when that bare title is unambiguous: exactly one live tab in the whole session carries it. -If 2+ live tabs share the same untagged bare title (this home's own pre-migration tab plus, say, a same-named tab from a different firstmate home sharing this session), the match refuses loudly rather than guessing which one is "ours". -A task already reachable through its recorded `window=` meta therefore keeps working unmodified after an upgrade to this fix, with no manual re-tagging step, as long as its title is not itself ambiguous; a genuinely ambiguous legacy collision (rare - it requires two homes to have independently generated the exact same task id before this fix shipped) surfaces as a loud refusal rather than a silent misdirect, and is resolved the same way any other stuck task is: tear down and respawn, which always gets the new home-scoped title. -`fm_backend_zellij_list_live`'s bulk recovery sweep deliberately does NOT attempt this legacy bare-title fallback (telling apart "our own pre-migration tab" from "another home's same-shaped bare title" in a sweep with no numeric id already in hand is not something this adapter can do safely); a pre-migration task stays reachable through its meta's `window=` field instead. - -**Moving/relocating a firstmate installation** changes its resolved `FM_ROOT` path and therefore its tag; tabs titled under the old tag simply stop matching new lookups. -This is accepted, exactly as it is for cmux: a task's own recorded worktree path in `state/<id>.meta` does not survive a repo relocation either, so this is consistent with an existing, already-accepted limitation, not a new one. - -## Target string and meta fields - -A zellij task's `window=` meta field holds `<zellij-session>:<pane-id>`, for example `firstmate:7`. -The pane id is a bare non-negative integer with no embedded colon (simpler than herdr's own pane-id shape, which itself contains a colon), so splitting on the first colon is trivially correct. -This mirrors tmux's `session:window` and herdr's `session:pane` target shapes closely enough that `fm_backend_resolve_selector` (`bin/fm-backend.sh`) needed no zellij-specific logic at all. -When the shared selector contract routes a zellij caller through firstmate metadata, it also supplies the expected caller-facing tab label `fm-<id>` to the zellij adapter, which internally checks it against the home-scoped title (falling back to the unambiguous-untagged legacy match described above). -That label check prevents a stale numeric pane id from being trusted after an external session deletion/recreation, or from being trusted for a different firstmate home's same-named tab; explicit raw `session:pane` targets remain a pane-existence-only escape hatch because there is no metadata label to verify. - -Zellij tasks additionally record: - -- `zellij_session=` - the named zellij session this task's tab lives in. -- `zellij_tab_id=` - the task's tab id. -- `zellij_pane_id=` - the task's terminal pane id, the fast-path operational target (same value as the `window=` field's second component). - -## Verified CLI facts - -| Operation | Verified zellij call | What was verified | -|---|---|---| -| Version gate | `zellij --version` -> `"zellij 0.44.0"` | Session-independent; no server needs to be running. | -| Headless session start | `zellij attach -b <name>` with stdin redirected from `/dev/null` and no controlling TTY | Creates the session and returns promptly (cannot actually attach without a TTY, so it exits after creating). The session persists with zero attached clients - `dump-screen`, `list-panes`, etc. all work against it. Running it again against an EXISTING session prints `"Session already exists"` and exits 1 - harmless, since existence is checked first via `list-sessions` and the launch call's own exit status is never inspected. | -| Session existence check | `zellij list-sessions --short --no-formatting` | Plain one-name-per-line output, safe to `grep -qxF`. Passive - never starts a session (unlike herdr's `target_ready`, which DOES auto-start: a herdr server restart is non-destructive and recovers persisted state, but zellij's `kill-session` is destructive, so auto-recreating under an unexpected name would silently orphan whatever the caller meant to reach). | -| Duplicate task check | `zellij action list-tabs --json`, match by home-scoped `.name` | Zellij does NOT enforce tab-name uniqueness itself (verified: two tabs can share a name, same as herdr's tabs). The adapter's own duplicate check is required, and it checks the home-scoped title such as `fm-firstmate-a1b2c3d4-<id>` (see "Home-scoped tab titles" above), never the bare `fm-<id>` label. | -| Create task tab | `zellij action new-tab --cwd <dir> --name <scoped-title>` | Returns the created tab's bare integer id on stdout, exactly as documented (resolves report gap #3). No `--no-focus`-equivalent flag exists at all - see "Focus-steal on new-tab" below. The caller passes `fm-<id>`, but the adapter creates `fm-<home-label>-<id>`. | -| Pane discovery | `zellij action list-panes --json`, filter `.tab_id == <id> and .is_plugin == false` | `tab_id`, `id` (the pane's own bare integer id), `is_plugin`, and `pane_cwd` are ALL present in the default `--json` output with no extra flags (`--tab`/`--geometry`/`--state`/`--command` add more fields but are not needed here). Terminal (non-plugin) pane ids are globally unique across a session's whole tab set - a SEPARATE incrementing namespace from plugin panes, which is why a plugin pane and a terminal pane can share the same bare `id` (the CLI's own `--pane-id` contract, `"3 (equivalent to terminal_3)"`, already documents this split). | -| Worktree-path discovery | marked active cwd probe + capture-scrape (`fm_backend_zellij_current_path`), NOT `.pane_cwd` | `.pane_cwd` reflects a `cd` run directly in the pane's own top-level shell, but does NOT follow a NESTED SUBSHELL's own `cd` (exactly what `treehouse get` does) - see "Worktree-path discovery: pane_cwd does not track a subshell" below. This directly contradicts the design report's assumption that passive `pane_cwd` polling would be "acceptable for tmux and zellij" (report gap #4 is NOT cleanly resolved as originally framed; the adapter works around it instead). | -| Send literal (unsubmitted) | `zellij action paste --pane-id <id> -- <text>` | Uses bracketed paste mode, does NOT auto-submit. Verified directly: a marker sent this way sits unexecuted at the prompt until a separate Enter. Behaves like tmux's `send-keys -l` / herdr's `pane send-text`. Chosen over `write-chars` per the design report's recommendation for popup-safety parity with the other backends. The `--` separator keeps option-shaped text such as `--help` literal. | -| Send key | `zellij action send-keys --pane-id <id> <key>` | Verified names: `"Enter"` (also `"enter"`) works; `"Esc"`/`"esc"` work but `"Escape"`/`"escape"` are REJECTED with "Invalid key"; Ctrl-C must be the SINGLE shell argument `"Ctrl c"` (a two-word key expression as ONE argv entry) - `"C-c"`, `"Ctrl+c"`, and passing `Ctrl`/`c` as two SEPARATE argv words all fail. Resolves report gap #2. | -| Send + submit, composed | `paste` then `send-keys --pane-id <id> Enter` | Zellij has no single-call atomic "type and submit" primitive (unlike tmux's `send-keys ... Enter` or herdr's `pane run`); `fm_backend_zellij_send_text_line` composes the two calls, which is the only form this adapter has for that operation. | -| Bounded capture | `zellij action dump-screen --pane-id <id>` for 40 lines or fewer; `zellij action dump-screen --pane-id <id> --full` above that threshold | Works for a background session with NO attached client (resolves report gap #1). No `--lines`-style bound flag exists at all (unlike herdr's buggy small-N `--lines`, there is simply no flag). Routine watcher-sized reads use zellij's viewport-only dump to avoid unbounded scrollback reads; larger explicit peeks request `--full` and trim to the caller's requested line count locally with `tail`. The tradeoff: on a very short terminal viewport, a 40-line routine read can see fewer than 40 lines and miss content above the visible screen. | -| Busy state | *(no native primitive)* | D5 (`herdr-addendum.md`): zellij has no agent-state API. `fm_backend_busy_state`'s dispatcher (`bin/fm-backend.sh`) falls through to `unknown` for zellij via its wildcard case, exactly like tmux - the watcher's existing pane-hash + regex path is the only busy-state source for this backend. | -| Agent liveness | *(no verified primitive)* | `fm_backend_agent_state` reports `unverified` for zellij, so `bin/fm-bootstrap.sh`'s session-start secondmate liveness sweep never auto-respawns a zellij secondmate endpoint, conservatively avoiding a false-dead reading that would create a duplicate secondmate supervisor in one home. | -| Kill | `zellij action close-tab-by-id <id>` (tab id resolved fresh from the pane id when possible; teardown can pass recorded `zellij_tab_id` plus the expected caller-facing `fm-<id>` label when the pane is already gone) | Unlike herdr (where closing a tab's only pane also closes the tab), closing a zellij pane with `close-pane` does NOT close the now-empty tab - it survives as an empty "ghost" entry in `list-tabs`. `close-tab-by-id` on a LIVE tab (with its pane still running) verified to cleanly remove both pane and tab in one call. Kill resolves the owning tab and closes by tab id; if teardown supplies an expected label, the tab id must still match it through the home-scoped-title or unambiguous legacy-title check before it is closed, including the recorded `zellij_tab_id` ghost-tab fallback. Best-effort (`\|\| true`), matching tmux's `kill-window` and herdr's `pane close` contract. | -| Recovery / list-live | `zellij action list-tabs --json`, filter names starting with this home's own `fm-<home-label>-` prefix | Name-based, never trusts a stored pane id blindly - the same posture herdr's `list_live` takes. Scoped to this installation's own home-scoped prefix (see "Home-scoped tab titles" above), so it never lists another firstmate home's tabs; the adapter strips the tag back off and reports the plain `fm-<id>` label. Does not attempt the legacy untagged-title fallback (that fallback is for a single already-known tab, not a bulk sweep). | -| Session cleanup (test-only) | `zellij delete-session <name> --force` | The single-call kill-and-delete form, gated behind `tests/zellij-test-safety.sh`'s guard (refuses an empty name, the literal `"firstmate"` default name, or a name not currently listed). Never `kill-all-sessions`/`delete-all-sessions` - see "Session safety" below. | - -## Worktree-path discovery: `pane_cwd` does not track a subshell (report gap #4, contradicted) - -The design report assumed passive `pane_cwd` polling would be "acceptable for tmux and zellij" (mirroring tmux's proven `pane_current_path`). -This was verified WRONG for the exact case that matters most: `treehouse get`, which opens a nested interactive subshell inside the pane. - -Verified against the real binary, step by step: - -1. A plain `cd /tmp` typed directly into a pane's own top-level shell updates `list-panes --json`'s `pane_cwd` within one sub-second poll - this is what an earlier, narrower verification pass mistakenly generalized from. -2. Running `treehouse get` in the same pane, waiting for its "Entered worktree at ..." banner, and even typing `pwd` INSIDE the now-interactive treehouse subshell (confirming on-screen that the shell truly is in the acquired worktree) - `pane_cwd` stays **frozen** at the ORIGINAL project directory the whole time. It never updates once a subshell has taken over as the pane's foreground process. -3. `list-panes --json --all` was checked for any pid or alternate live-cwd field (mirroring herdr's `foreground_cwd`) - none exists. Zellij's CLI exposes `pane_command` (the last-invoked command string, e.g. `"treehouse get"`) and `pane_cwd` (frozen at that command's invocation time), but no per-pane process id and no live-tracking cwd field at all. - -This is a genuinely worse gap than herdr's frozen-cwd trap: herdr at least exposes `foreground_cwd` as the fix (`docs/herdr-backend.md`); zellij's CLI has no equivalent primitive to reach for. - -**Workaround, `fm_backend_zellij_current_path`:** actively probe instead of passively reading JSON. -Submit a short begin marker, `pwd`, and a short end marker into the pane via the same `send_text_line` primitive used for `treehouse get` itself, briefly settle, capture the pane, and concatenate only the visual lines between the two markers. -This works because `pwd` reads from the current foreground shell no matter how many subshells deep the pane is, sidestepping the need for any structured field at all. -The begin/end markers avoid false matches from absolute-path prompts, previous scrollback, and treehouse's own `~`-prefixed "Entered worktree at ..." banner (`tests/fm-backend-zellij.test.sh` pins prompt-path, banner, and wrapped-path cases). -Concatenating the marked block also handles a long worktree path that zellij's visual screen dump soft-wraps across multiple terminal rows. -Verified against the real binary in both shapes: a direct `cd` in the pane's own shell, AND a nested subshell's own `cd` (`bash -c` spawned and cd'd inside it) - the load-bearing case matching `treehouse get`'s actual shape (`tests/fm-backend-zellij-smoke.test.sh`'s two `current_path` assertions). - -This op is scoped to `fm-spawn.sh`'s own worktree-discovery poll loop, the only caller - injecting a harmless extra cwd probe into the pane's scrollback before the harness ever launches is an acceptable trade for a reliable answer, and does not affect the interactive session the crewmate later runs in. - -## Focus-steal on new-tab (report gap #5, confirmed - and mitigated) - -Verified against the real binary with a genuinely attached pty client (`script -q /dev/null zellij attach <session>`): `zellij action new-tab` unconditionally focuses the newly created tab for every attached client, and **there is no flag to suppress this** - `new-tab --help` lists no `--no-focus` equivalent at all (unlike herdr's `--no-focus`, verified in `docs/herdr-backend.md`, or tmux's `new-window -d`). -Before the client attached, the freshly created tab showed `"active": false` in `list-tabs --json`; after attaching a real pty client and creating another tab, that new tab immediately showed `"active": true` and the client's live view moved to it. - -**Mitigation**, implemented in `fm_backend_zellij_create_task`: capture the session's previously-active tab id (`list-tabs --json`, `.active == true`) *before* calling `new-tab`, then call `go-to-tab-by-id <that-id>` afterward to restore it. -Verified empirically: this correctly moves an attached client's view back to where it was, and is a safe, silent no-op (`go-to-tab-by-id` against a session with zero attached clients returns exit 0 doing nothing observable) for the common unattended-spawn case where no client is attached at all. -This is the one place this adapter deviates from a flag-based solution the other backends have, because zellij genuinely does not expose one; the mitigation is a best-effort second call, not a suppression flag, so there is a narrow window between tab creation and the restore call during which an attached client's view is briefly on the new tab. - -## Unconditional exit code 0 (un-anticipated, load-bearing finding) +- Zellij 0.44 or newer. +- `jq` for JSON responses. +- The universal harness and toolchain requirements in [`configuration.md`](configuration.md#toolchain). -Not called out in the original design report, and the single most important operational caveat for this adapter: **every `zellij action <subcommand>` call exits 0 unconditionally**, regardless of whether the target actually exists. +Select it with local `config/backend` containing `zellij`, `FM_BACKEND=zellij` for one launch, or an explicit request to Firstmate. +It is never auto-detected. +A spawn stops before creating a session or acquiring a worktree when Zellij or `jq` is missing or Zellij is below 0.44. -Verified three ways against the real binary: +Firstmate uses one shared session named `firstmate` by default. +`FM_ZELLIJ_SESSION` can select another name for isolated verification. +Attach with: -- Against a **nonexistent session**: every action subcommand tried (`list-panes`, `paste`, `new-tab`) printed the live session list to stdout and an error (`"Session '<name>' not found..."`) to stderr, but exited **0**. -- Against a **live session but a nonexistent pane id**: `send-keys --pane-id 999 Enter` produced **no output on either stream** and exited **0**. -- `dump-screen --pane-id 999 --full` against a live session but dead pane returned **empty output** (a single newline) with exit **0** - a soft, not hard, signal (a genuinely blank pane could also read this way). +```sh +zellij attach <session-name> +``` -This means the exit code can **never** be trusted to detect a bad target on this backend - a meaningful difference from tmux, which does return a nonzero exit and a clear error for a truly nonexistent target. +Routine supervision does not require attachment. +Use `bin/fm-peek.sh <id>` and `FM_HOME=<home> bin/fm-send.sh <id> '<text>'` against the metadata-routed endpoint. -**Mitigation, in two layers:** +Verify setup by spawning a small task and confirming metadata contains `backend=zellij`, `zellij_session=`, `zellij_tab_id=`, and `zellij_pane_id=`. -1. Send, capture, and cwd operations call `fm_backend_zellij_target_ready` first, which verifies session existence via the passive `list-sessions` check and verifies the specific terminal pane via `list-panes --json` filtered to `.id == <pane>` and `.is_plugin == false`. - When the caller reached the pane through a recorded firstmate task, `target_ready` also resolves the pane's owning tab and checks it against the expected caller-facing `fm-<id>` label through the home-scoped-title or unambiguous legacy-title check before sending, capturing, or reading cwd. - This catches a whole session gone (killed externally, or a stale meta from a prior run), the normal stale-pane case, and stale numeric pane ids reused by an unrelated recreated session. - Explicit raw `session:pane` targets keep the pane-only check because they intentionally have no recorded `fm-<id>` ownership context. - Kill checks the session, resolves the tab from the pane when possible, and uses teardown's recorded `zellij_tab_id` fallback when the pane is already gone only after `list-tabs --json` proves the tab still matches the expected caller-facing `fm-<id>` label through that same title check. -2. Output-**shape** validation rejects the "session not found" text fallback structurally: `fm_backend_zellij_create_task` requires `new-tab`'s stdout to parse as a bare integer (the colored session-list text does not), and every `list-panes`/`list-tabs` consumer pipes through `jq`, which fails to parse the plain-text fallback as JSON. +## Task shape and home isolation -**Accepted residual gaps**: a pane can still die in the brief window between `fm_backend_zellij_target_ready`'s ownership check and the operation's own `zellij action` call. -That remaining race degrades to "the operation quietly did nothing" - the same class of gap firstmate already tolerates for an unverified send on any backend, caught downstream by `fm-spawn.sh`'s worktree-discovery poll timing out after 60s, `fm_backend_zellij_send_text_submit`'s preflight or content-diff retry loop (which reports `send-failed`, `pending`, or `unknown` rather than a false "sent" for these cases), or the watcher's stale-pane detection eventually noticing a pane that never changes. -An explicit raw `session:pane` target can also still address a reused pane id if an operator deliberately bypasses firstmate metadata; that path is kept as an escape hatch, not as the normal task routing path. +Every task receives one tab in the shared Zellij session. +The caller-facing label remains `fm-<id>`, while the visible title is home-scoped as `fm-<home-label>-<id>`. +The home label is `firstmate` or `2ndmate-<id>` plus a short stable hash of the resolved Firstmate root. +This prevents task-id collisions between a primary, secondmates, and separate Firstmate installations sharing one session. -## Every pane op needs an EXPLICIT `--pane-id` (un-anticipated finding) +Zellij does not enforce tab-name uniqueness, so the adapter performs its own duplicate check against the scoped title. +Create, recover, list, and cleanup paths all use the same scoped title owner in `bin/fm-backend-hometag-lib.sh`. +Moving a Firstmate installation changes its path hash and leaves old titles unmatched, consistent with worktree paths also becoming stale after a move. -A fresh zellij session auto-opens a floating "About Zellij"/release-notes **plugin** pane in tab 0 that starts **focused** and visually on top of the real terminal pane. -A pane-targeting call made WITHOUT an explicit `--pane-id` (relying on the "focused pane" default) silently goes to this plugin pane instead of the terminal - verified directly: `write-chars 'echo hello'` with no `--pane-id` produced no visible effect in the terminal pane at all. -Every op in this adapter passes an explicit `--pane-id` (a bare integer is confirmed equivalent to `terminal_<n>`, never ambiguous with a plugin pane of the same bare number) for exactly this reason; there is no default-target code path anywhere in `bin/backends/zellij.sh`. +A pre-home-tag task remains reachable through its recorded metadata only when exactly one live tab has the old unscoped title. +Multiple old tabs with the same title cause a refusal rather than a guess. +Bulk recovery never adopts unscoped legacy tabs because it has no safe home identity for them. -## Tab-name duplication is not enforced (un-anticipated, but expected finding) +```text +backend=zellij +window=<session>:<pane-id> +zellij_session=<session> +zellij_tab_id=<tab-id> +zellij_pane_id=<pane-id> +``` -Same as herdr's tabs and unlike tmux's own window-name uniqueness: `zellij action new-tab --name <label>` happily creates a second tab sharing an existing name. -`fm_backend_zellij_create_task`'s own `list-tabs`-based duplicate check is therefore required, mirroring both prior adapters - and, because this session's tab bar is shared by every firstmate home with no per-home container split, that check is against the home-scoped title (see "Home-scoped tab titles" above), not the bare `fm-<id>` label, so it cannot be fooled into refusing (or worse, silently reusing) another home's same-id tab. +Recorded pane ids are numeric and are never trusted alone after a session recreation. +Metadata-routed operations also verify the owning tab's expected scoped or unambiguous legacy title. +An explicit raw `session:pane` target remains a pane-existence-only operator escape hatch. -## Closing a pane does not close its tab (un-anticipated finding) +## Current operation and safety -Unlike herdr (where closing a tab's only root pane also closes the tab), zellij's `close-pane --pane-id <id>` leaves an empty "ghost" tab behind in `list-tabs --json` - verified: the tab entry persists with zero panes until explicitly closed. -`close-tab-by-id <id>` on a still-LIVE tab (pane running normally) was separately verified to cleanly remove both the pane and the tab in one call, needing no `close-pane` first. -This is why `fm_backend_zellij_kill` resolves the owning tab id from the pane when possible, accepts teardown's recorded `zellij_tab_id` as a fallback when the pane has already gone, verifies the expected caller-facing `fm-<id>` label through the home-scoped-title or unambiguous legacy-title check when teardown provides it, and calls `close-tab-by-id`, rather than mirroring herdr's simpler "close the pane, the tab follows" contract. +Zellij's CLI action commands return exit 0 even for missing sessions or panes. +The adapter therefore verifies session, terminal pane, and expected title before an operation and validates JSON or integer response shapes afterward. +A pane can still disappear between verification and the operation; downstream submit, worktree-discovery, and stale detection report that narrow race rather than treating exit 0 as success. -## Composer verification: delta-based +Every pane operation passes an explicit `--pane-id` because a new session can focus its release-notes plugin pane, whose numeric plugin id is in a separate namespace from terminal pane ids. -Zellij's CLI exposes no cursor-row/ANSI-only capture primitive (like tmux's), so `fm_backend_zellij_send_text_submit` still uses a content-diff strategy: capture the pane right after typing (the unsubmitted "typed" baseline), then after each Enter attempt capture again - unchanged means retry, changed means submitted. -This is now zellij-specific; the herdr adapter moved away from content-diff after the 2026-07-03 grok slash-submit incident and now confirms normal idle-baseline submits through native agent-state, retaining structural composer-state for the affirmative-empty injection guard and submit fallback. -All implemented submit-verifying backends expose the identical caller-facing verdict vocabulary (`empty`, `pending`, `unknown`, `send-failed`), so `fm-send.sh` needs no backend-specific branching. +`pane_cwd` follows a top-level shell `cd` but not the foreground subshell opened by `treehouse get`. +Worktree discovery therefore sends begin and end markers around `pwd`, captures the marked block, and joins wrapped path lines. +This active probe is scoped to spawn-time worktree discovery and is not advertised as a general live-cwd API. -## Session safety +`new-tab` has no no-focus flag and temporarily focuses the created tab in attached clients. +The adapter records the previously active tab and immediately restores it with `go-to-tab-by-id`. +There is a narrow visible race between those calls that no current Zellij flag can remove. -`zellij kill-session <name>` and `zellij delete-session <name>` both take an explicit, required name - there is no ambient "whatever session is running" command shape like herdr's `server stop` that caused two live-fleet kills (`docs/herdr-backend.md` "Session targeting"). -The realistic risk for this backend is instead a test accidentally reusing (and then deleting) the real `firstmate` session name, or reaching for the fleet-wide `kill-all-sessions`/`delete-all-sessions` commands. -`tests/zellij-test-safety.sh`'s `zellij_refuse_if_unsafe` guards against both: it refuses an empty name, the literal `"firstmate"` default, or a name not currently listed as active, before `zellij_safe_delete` is allowed to run `delete-session --force`. -Every real-zellij test in this document and its accompanying test files uses a uniquely-named session (`fm-backend-smoke-$$`, or similar) and this guarded cleanup path exclusively. +Literal send uses bracketed paste followed by a separate explicit Enter. +The adapter supports `Enter`, `Esc`, and the one-argument key expression `Ctrl c` through the shared key vocabulary. +Zellij exposes no cursor-row, ANSI composer style, or native agent-state signal, so submit acknowledgement remains content-delta based. +This can distinguish no change from a changed screen but is less precise than tmux's structural box reader or Herdr's native state plus structural classifier. -## End-to-end verification (spawn -> steer -> peek -> done -> merge -> teardown) +Viewport capture has no line-bound option. +Routine reads use `dump-screen` and larger peeks use `dump-screen --full`, followed by local trimming. +A short viewport may expose fewer lines than requested. -Beyond the fake-CLI unit tests (`tests/fm-backend-zellij.test.sh`) and the real-CLI smoke tests (`tests/fm-backend-zellij-smoke.test.sh`), the full firstmate lifecycle was driven end to end against a real `claude` crewmate through this branch's own scripts, in a scratch `FM_HOME`, a scratch `local-only` git project, and an isolated `FM_ZELLIJ_SESSION` (never the real `firstmate` session name): +Closing a pane leaves an empty tab. +Cleanup resolves and verifies the owning tab, then uses `close-tab-by-id` so both the task pane and tab disappear. +Real test cleanup uses only an isolated non-`firstmate` session and the guard in `tests/zellij-test-safety.sh`; it never calls all-session deletion commands. -1. `FM_HOME=<scratch> FM_BACKEND=zellij FM_ZELLIJ_SESSION=<isolated> bin/fm-spawn.sh zellij-e2e-t1 projects/scratch-e2e-project claude` - spawned successfully, printing `window=<session>:<pane>` in the summary and writing `backend=zellij`, `zellij_session=`, `zellij_tab_id=`, `zellij_pane_id=` to the task's meta. The worktree-discovery poll correctly resolved the real treehouse worktree path using the active `pwd`-probe workaround. -2. `FM_HOME=<scratch> FM_ZELLIJ_SESSION=<isolated> bin/fm-peek.sh fm-zellij-e2e-t1` - showed the live claude trust dialog ("Quick safety check: Is this a project you created or one you trust?"). -3. `FM_HOME=<scratch> FM_ZELLIJ_SESSION=<isolated> bin/fm-send.sh fm-zellij-e2e-t1 --key Enter` - accepted the trust dialog. -4. `FM_HOME=<scratch> FM_ZELLIJ_SESSION=<isolated> bin/fm-peek.sh fm-zellij-e2e-t1` again - showed claude actively working through the brief (verifying isolation, then implementing). -5. `FM_HOME=<scratch> FM_ZELLIJ_SESSION=<isolated> bin/fm-send.sh fm-zellij-e2e-t1 "captain says: proceed as planned, this is a trivial verification task"` - a plain-text steer while claude was mid-turn, exercising the delta-based send-and-verify path; the send completed without a `pending`/`send-failed` error. -6. The crewmate appended `done: ready in branch fm/zellij-e2e-t1` to its status file, and its commit (`add hello.txt`, message `add hello.txt`) was confirmed present on branch `fm/zellij-e2e-t1` in the project's git history, with `hello.txt` containing exactly the expected line. -7. `bin/fm-teardown.sh zellij-e2e-t1` **REFUSED**, exactly as required: `REFUSED: local-only worktree ... has work not yet merged into main and not on any remote.` -8. `bin/fm-merge-local.sh zellij-e2e-t1` - fast-forwarded local `main` to the crewmate's commit (`02c9dd2 -> ba41f90`). -9. `bin/fm-teardown.sh zellij-e2e-t1` now succeeded: terminated the lingering worktree processes, returned the treehouse worktree, closed the zellij tab (confirmed gone via `list-tabs --json` - only the default `Tab #1` remained), and removed all of the task's `state/` files. +## Active limits -The one real bug this pass caught - the `pane_cwd`-does-not-track-a-subshell gap (see "Worktree-path discovery" above) - was found and fixed during this E2E run itself: the FIRST attempt refused to launch with "did not yield an isolated worktree" because `current_path`'s original (JSON-only) implementation never saw `treehouse get`'s subshell move away from the project directory, so the 60-second poll's own comparison collapsed to "same path" and the isolation guard correctly (if confusingly) refused. After the `pwd`-probe fix, the identical flow spawned cleanly on the very next attempt. +- Zellij is experimental and explicit-only. +- All homes share one session and tab bar; scoped titles prevent cross-home identity collisions but do not create per-home visual containers. +- There is no native busy or push-event signal, so supervision uses capture/hash and busy-regex polling. +- There is no verified agent-process liveness signal, so a dead Zellij secondmate is reported inconclusive rather than auto-respawned. +- New-tab focus restoration has a narrow visible race. +- CLI exit status is not meaningful; a target can still disappear after structural readiness checks. +- Worktree cwd discovery requires the spawn-time marker probe. +- An ambiguous unscoped legacy title requires manual cleanup and respawn. -The isolated zellij session and the scratch `FM_HOME`/project were fully torn down after this run (`zellij delete-session <isolated> --force`, `rm -rf` on the scratch root); the real `firstmate` session name and the live tmux/herdr fleet were never touched at any point. +## Regression entry points -## Known gaps left for a follow-up +```sh +tests/fm-backend-zellij.test.sh +tests/fm-backend-zellij-smoke.test.sh +``` -- **No event push at all**, not even herdr's semantic busy-state (D5): zellij has no analogue to herdr's `agent.get`, so `fm-watch.sh`'s existing pane-hash + `FM_BUSY_REGEX` poll loop is the ONLY event source for this backend, identical to the tmux path. This is the expected, designed-for outcome (D5 explicitly calls for "the poll-based capture/hash/busy-regex path, same vocabulary as tmux"), not a shortfall relative to the report. -- **No verified agent-process liveness classifier.** The session-start secondmate liveness sweep therefore receives `unverified` from `fm_backend_agent_state` for zellij and reports that the recovery classifier is unverified instead of killing or respawning the endpoint. - This leaves a dead zellij secondmate for manual recovery, but avoids the worse failure mode of duplicating a live supervisor. -- **The focus-steal mitigation has a narrow race window.** Between `new-tab` (which steals focus immediately) and the follow-up `go-to-tab-by-id` restore call, an attached client's view is briefly on the new tab. No flag-based suppression exists to close this window entirely (see "Focus-steal on new-tab" above); a future zellij release may add one. -- **A pane can still die after `target_ready` succeeds and before the operation runs.** Metadata-routed operations now verify the expected caller-facing `fm-<id>` label through the home-scoped-title or unambiguous legacy-title check, as well as the pane id up front, but zellij's unconditional exit 0 still leaves this narrow time-of-check/time-of-use race for one-shot operations (see "Unconditional exit code 0" above). -- **The `pwd`-probe workaround for worktree-path discovery is scoped to `fm-spawn.sh`'s own poll loop only** (see "Worktree-path discovery" above). It is not a general-purpose live-cwd primitive; a future caller needing a live cwd read for a zellij pane outside that narrow spawn-time context would need the same active-probe approach, not a passive JSON field. -- **No per-home container split**, unlike herdr's later P3 refinement (`docs/herdr-backend.md` "Default task container shape"). This is a deliberate simplicity choice per the locked captain decision (D2: "zellij, content unchanged from the report"), not an oversight; if a captain later runs many concurrent secondmates on the zellij backend and wants per-home visual separation in the tab bar, that would be a natural follow-up mirroring herdr's workspace-per-home pass. Note this is a CONTAINER-level (visual tab-bar grouping) gap only - the cross-home NAME-collision gap this shared container shape used to carry (two homes' same-id tabs sending/peeking/closing each other) is closed by home-scoped tab titles, "Home-scoped tab titles" above. -- **The untagged-legacy migration fallback has one residual ambiguity gap.** `fm_backend_zellij_tab_matches_label`'s bare-title fallback (for a tab spawned before home-scoping shipped) refuses rather than guesses when 2+ live tabs share the exact same untagged bare title - but a genuinely ambiguous case then requires manual intervention (tear down and respawn to get a new home-scoped title) rather than an automatic resolution. This is accepted as the honest trade for not needing a one-time re-tagging migration step; see "Home-scoped tab titles" above. +The real smoke test uses a unique session and guarded deletion. +[`verification/runtime-backends.md`](verification/runtime-backends.md#zellij) records the active CLI matrix and lifecycle evidence. diff --git a/tests/fixtures/quota-array-dispatch/cases.json b/tests/fixtures/quota-array-dispatch/cases.json new file mode 100644 index 00000000000..c6fc3c3a867 --- /dev/null +++ b/tests/fixtures/quota-array-dispatch/cases.json @@ -0,0 +1,394 @@ +{ + "cases": [ + { + "id": "higher-raw-ahead-vs-lower-raw-sustainable", + "expect": "B", + "reason": "prefer sustainable pace over higher raw headroom with conservation pressure", + "candidates": [ + { + "id": "A", + "harness": "claude", + "model": "strong-a", + "effort": "high", + "fit": "comparable", + "reasoningClass": "strong", + "tight": false, + "rawHeadroom": 80, + "paceStatus": "ahead", + "aheadWindowIds": ["weekly"], + "worstReserve": -12.0, + "unknownPace": false, + "paceAvailable": true + }, + { + "id": "B", + "harness": "codex", + "model": "strong-b", + "effort": "high", + "fit": "comparable", + "reasoningClass": "strong", + "tight": false, + "rawHeadroom": 55, + "paceStatus": "behind", + "aheadWindowIds": [], + "worstReserve": 18.0, + "unknownPace": false, + "paceAvailable": true + } + ] + }, + { + "id": "mixed-effective-with-ahead-bound", + "expect": "B", + "reason": "mixed with aheadWindowIds is conservation pressure", + "candidates": [ + { + "id": "A", + "harness": "claude", + "model": "mixed-a", + "effort": "high", + "fit": "comparable", + "reasoningClass": "strong", + "tight": false, + "rawHeadroom": 75, + "paceStatus": "mixed", + "aheadWindowIds": ["seven_day"], + "worstReserve": -8.0, + "unknownPace": false, + "paceAvailable": true + }, + { + "id": "B", + "harness": "codex", + "model": "steady-b", + "effort": "high", + "fit": "comparable", + "reasoningClass": "strong", + "tight": false, + "rawHeadroom": 60, + "paceStatus": "on_pace", + "aheadWindowIds": [], + "worstReserve": 0.0, + "unknownPace": false, + "paceAvailable": true + } + ] + }, + { + "id": "both-ahead-least-negative-reserve", + "expect": "B", + "reason": "among pressured candidates prefer least-negative worst reserve", + "candidates": [ + { + "id": "A", + "harness": "claude", + "model": "pressured-a", + "effort": "high", + "fit": "comparable", + "reasoningClass": "strong", + "tight": false, + "rawHeadroom": 50, + "paceStatus": "ahead", + "aheadWindowIds": ["weekly"], + "worstReserve": -22.0, + "unknownPace": false, + "paceAvailable": true + }, + { + "id": "B", + "harness": "codex", + "model": "pressured-b", + "effort": "high", + "fit": "comparable", + "reasoningClass": "strong", + "tight": false, + "rawHeadroom": 48, + "paceStatus": "ahead", + "aheadWindowIds": ["weekly"], + "worstReserve": -5.0, + "unknownPace": false, + "paceAvailable": true + } + ] + }, + { + "id": "ahead-bounding-window-overrides-neutral-effective-summary", + "expect": "B", + "reason": "an ahead applicable bounding window creates conservation pressure even when the effective summary is neutral", + "candidates": [ + { + "id": "A", + "harness": "claude", + "model": "bounded-a", + "effort": "high", + "fit": "comparable", + "reasoningClass": "strong", + "tight": false, + "rawHeadroom": 72, + "paceStatus": "on_pace", + "aheadWindowIds": [], + "boundingWindows": [ + { + "id": "weekly", + "paceStatus": "ahead", + "reservePercentPoints": -9.0 + } + ], + "worstReserve": -9.0, + "unknownPace": false, + "paceAvailable": true + }, + { + "id": "B", + "harness": "codex", + "model": "steady-b", + "effort": "high", + "fit": "comparable", + "reasoningClass": "strong", + "tight": false, + "rawHeadroom": 58, + "paceStatus": "behind", + "aheadWindowIds": [], + "boundingWindows": [ + { + "id": "weekly", + "paceStatus": "behind", + "reservePercentPoints": 7.0 + } + ], + "worstReserve": 7.0, + "unknownPace": false, + "paceAvailable": true + } + ] + }, + { + "id": "known-sustainable-vs-unknown", + "expect": "A", + "reason": "prefer known sustainable evidence over unknown pace", + "candidates": [ + { + "id": "A", + "harness": "codex", + "model": "known-a", + "effort": "medium", + "fit": "comparable", + "reasoningClass": "standard", + "tight": false, + "rawHeadroom": 40, + "paceStatus": "behind", + "aheadWindowIds": [], + "worstReserve": 10.0, + "unknownPace": false, + "paceAvailable": true + }, + { + "id": "B", + "harness": "claude", + "model": "unknown-b", + "effort": "medium", + "fit": "comparable", + "reasoningClass": "standard", + "tight": false, + "rawHeadroom": 42, + "paceStatus": "unknown", + "aheadWindowIds": [], + "worstReserve": null, + "unknownPace": true, + "paceAvailable": true + } + ] + }, + { + "id": "select-pi-xai-before-authentication", + "expect": "pi-xai", + "reason": "an unauthenticated standalone Grok candidate cannot block selected authenticated Pi/xAI", + "candidates": [ + { + "id": "pi-xai", + "harness": "pi", + "model": "xai/grok-4.5", + "provider": "xai", + "authenticationSurface": "Pi xAI OAuth", + "authAvailable": true, + "fit": "comparable", + "reasoningClass": "strong", + "tight": false, + "rawHeadroom": 55, + "paceStatus": "behind", + "aheadWindowIds": [], + "worstReserve": 15.0, + "unknownPace": false, + "paceAvailable": true + }, + { + "id": "standalone-grok", + "harness": "grok", + "model": "grok-4.5", + "provider": "grok", + "authenticationSurface": "Grok Build CLI", + "authAvailable": false, + "authFailure": "Grok Build CLI login missing", + "fit": "comparable", + "reasoningClass": "strong", + "tight": false, + "rawHeadroom": 80, + "paceStatus": "ahead", + "aheadWindowIds": ["weekly"], + "worstReserve": -12.0, + "unknownPace": false, + "paceAvailable": true + } + ] + }, + { + "id": "all-tight-strongest-reasoning", + "expect": "A", + "reason": "preserve strongest-reasoning class when every candidate is tight", + "requiredReasoningClass": "strong", + "candidates": [ + { + "id": "A", + "harness": "claude", + "model": "strong-tight", + "effort": "high", + "fit": "comparable", + "reasoningClass": "strong", + "tight": true, + "rawHeadroom": 8, + "paceStatus": "behind", + "aheadWindowIds": [], + "worstReserve": 2.0, + "unknownPace": false, + "paceAvailable": true + }, + { + "id": "B", + "harness": "codex", + "model": "weaker-roomier", + "effort": "medium", + "fit": "comparable", + "reasoningClass": "standard", + "tight": true, + "rawHeadroom": 25, + "paceStatus": "behind", + "aheadWindowIds": [], + "worstReserve": 12.0, + "unknownPace": false, + "paceAvailable": true + } + ] + }, + { + "id": "genuine-tie-captain-choice", + "expectError": "genuine tie requires captain choice", + "reason": "report genuine ties instead of selecting by array order or harness identity", + "candidates": [ + { + "id": "A", + "harness": "claude", + "model": "same-model", + "effort": "high", + "fit": "comparable", + "reasoningClass": "strong", + "tight": false, + "rawHeadroom": 50, + "paceStatus": "on_pace", + "aheadWindowIds": [], + "worstReserve": 0.0, + "unknownPace": false, + "paceAvailable": true + }, + { + "id": "B", + "harness": "codex", + "model": "same-model", + "effort": "high", + "fit": "comparable", + "reasoningClass": "strong", + "tight": false, + "rawHeadroom": 50, + "paceStatus": "on_pace", + "aheadWindowIds": [], + "worstReserve": 0.0, + "unknownPace": false, + "paceAvailable": true + } + ] + }, + { + "id": "genuine-tie-reversed-array-order", + "expectError": "genuine tie requires captain choice", + "reason": "reversing a genuine tie must still require captain choice", + "candidates": [ + { + "id": "B", + "harness": "codex", + "model": "same-model", + "effort": "high", + "fit": "comparable", + "reasoningClass": "strong", + "tight": false, + "rawHeadroom": 50, + "paceStatus": "on_pace", + "aheadWindowIds": [], + "worstReserve": 0.0, + "unknownPace": false, + "paceAvailable": true + }, + { + "id": "A", + "harness": "claude", + "model": "same-model", + "effort": "high", + "fit": "comparable", + "reasoningClass": "strong", + "tight": false, + "rawHeadroom": 50, + "paceStatus": "on_pace", + "aheadWindowIds": [], + "worstReserve": 0.0, + "unknownPace": false, + "paceAvailable": true + } + ] + }, + { + "id": "schema-v2-absent-pace", + "expect": "A", + "reason": "absent pace degrades to raw headroom without fabricating pace health", + "candidates": [ + { + "id": "A", + "harness": "codex", + "model": "legacy-a", + "effort": "medium", + "fit": "comparable", + "reasoningClass": "standard", + "tight": false, + "rawHeadroom": 70, + "paceStatus": null, + "aheadWindowIds": [], + "worstReserve": null, + "unknownPace": false, + "paceAvailable": false + }, + { + "id": "B", + "harness": "claude", + "model": "legacy-b", + "effort": "medium", + "fit": "comparable", + "reasoningClass": "standard", + "tight": false, + "rawHeadroom": 40, + "paceStatus": null, + "aheadWindowIds": [], + "worstReserve": null, + "unknownPace": false, + "paceAvailable": false + } + ] + } + ] +} diff --git a/tests/fixtures/quota-array-dispatch/schema-v3-shape.json b/tests/fixtures/quota-array-dispatch/schema-v3-shape.json new file mode 100644 index 00000000000..a79f86aca32 --- /dev/null +++ b/tests/fixtures/quota-array-dispatch/schema-v3-shape.json @@ -0,0 +1,103 @@ +{ + "schemaVersion": 3, + "generatedAt": "1970-01-01T00:00:00.000Z", + "providers": [ + { + "provider": "claude", + "label": "Claude", + "source": "test", + "plan": "test", + "windows": [ + { + "id": "five_hour", + "label": "session", + "kind": "session", + "percentUsed": 20, + "percentRemaining": 80, + "windowSeconds": 18000, + "pace": { + "status": "behind", + "timeRemainingPercent": 40.0, + "elapsedPercent": 60.0, + "reservePercentPoints": 40.0 + } + }, + { + "id": "seven_day", + "label": "week", + "kind": "weekly", + "percentUsed": 55, + "percentRemaining": 45, + "windowSeconds": 604800, + "pace": { + "status": "ahead", + "timeRemainingPercent": 60.0, + "elapsedPercent": 40.0, + "reservePercentPoints": -15.0 + } + } + ], + "quotaSemantics": { + "status": "known", + "description": "sanitized representative schemaVersion 3 shape", + "effectiveAvailability": [ + { + "scope": "all_models", + "status": "known", + "effectivePercentRemaining": 45, + "boundedBy": ["five_hour", "seven_day"], + "limitingWindowIds": ["seven_day"], + "pace": { + "status": "mixed", + "aheadWindowIds": ["seven_day"], + "behindWindowIds": ["five_hour"], + "worstReservePercentPoints": -15.0, + "worstReserveWindowId": "seven_day" + } + } + ] + } + }, + { + "provider": "codex", + "label": "Codex", + "source": "test", + "plan": "test", + "windows": [ + { + "id": "weekly", + "label": "week", + "kind": "weekly", + "percentUsed": 30, + "percentRemaining": 70, + "windowSeconds": 604800, + "pace": { + "status": "behind", + "timeRemainingPercent": 50.0, + "elapsedPercent": 50.0, + "reservePercentPoints": 20.0 + } + } + ], + "quotaSemantics": { + "status": "known", + "description": "sanitized representative schemaVersion 3 shape", + "effectiveAvailability": [ + { + "scope": "all_models", + "status": "known", + "effectivePercentRemaining": 70, + "boundedBy": ["weekly"], + "limitingWindowIds": ["weekly"], + "pace": { + "status": "behind", + "behindWindowIds": ["weekly"], + "worstReservePercentPoints": 20.0, + "worstReserveWindowId": "weekly" + } + } + ] + } + } + ] +} diff --git a/tests/fm-afk-launch.test.sh b/tests/fm-afk-launch.test.sh index 8075d7067ac..b65bd9cdc26 100755 --- a/tests/fm-afk-launch.test.sh +++ b/tests/fm-afk-launch.test.sh @@ -70,6 +70,54 @@ unit_clear_stale() { rm -rf "$st" } +unit_relative_paths_are_absolute_before_daemon_launch() { + local root home state out status linked_home + root=$(mktemp -d "${TMPDIR:-/tmp}/fm-afk-relative-home.XXXXXX") + mkdir -p "$root/home/state" "$root/cdpath/home/state" + home=$(cd "$root/home" && pwd -P) + state="$home/state" + out=$( + cd "$root" || exit 1 + CDPATH="$root/cdpath" FM_HOME=home FM_STATE_OVERRIDE=home/state \ + bash -c '. "$1"; printf "%s\n%s\n" "$FM_HOME" "$FM_AFK_LAUNCH_STATE"' _ "$LAUNCH" + ) + if [ "$out" = "$home"$'\n'"$state" ]; then + pass "launcher paths: relative home and state ignore CDPATH before daemon command construction" + else + fail "launcher paths: relative home or state remained cwd-dependent ($out)" + fi + linked_home="$root/home-link" + ln -s "$root/home" "$linked_home" + out=$(FM_HOME="$linked_home" FM_STATE_OVERRIDE="$linked_home/state" \ + bash -c '. "$1"; printf "%s\n%s\n" "$FM_HOME" "$FM_AFK_LAUNCH_STATE"' _ "$LAUNCH") + if [ "$out" = "$linked_home"$'\n'"$linked_home/state" ]; then + pass "launcher paths: absolute symlink spellings are preserved" + else + fail "launcher paths: absolute symlink spelling changed ($out)" + fi + out=$( + cd "$root" || exit 1 + FM_HOME=missing-home "$LAUNCH" help 2>&1 + ) + status=$? + if [ "$status" -ne 0 ] && printf '%s\n' "$out" | grep -F "FM_HOME directory cannot be resolved: missing-home" >/dev/null; then + pass "launcher paths: unresolved relative FM_HOME fails loudly" + else + fail "launcher paths: unresolved relative FM_HOME did not name the bad input ($out)" + fi + out=$( + cd "$root" || exit 1 + FM_HOME=home FM_STATE_OVERRIDE=missing-state "$LAUNCH" help 2>&1 + ) + status=$? + if [ "$status" -ne 0 ] && printf '%s\n' "$out" | grep -F "FM_STATE_OVERRIDE directory cannot be resolved: missing-state" >/dev/null; then + pass "launcher paths: unresolved relative FM_STATE_OVERRIDE fails loudly" + else + fail "launcher paths: unresolved relative FM_STATE_OVERRIDE did not name the bad input ($out)" + fi + rm -rf "$root" +} + # --------------------------------------------------------------------------- # UNIT 2: a FRESH entry clears; a REFRESH (daemon already alive) preserves the # current session's buffered escalations. @@ -861,6 +909,7 @@ e2e_tmux() { } unit_clear_stale +unit_relative_paths_are_absolute_before_daemon_launch unit_fresh_vs_refresh unit_stop_ordering unit_stop_rejects_reused_pid diff --git a/tests/fm-arm-pretool-check.test.sh b/tests/fm-arm-pretool-check.test.sh index 0bc6cfac282..5ba750aea09 100755 --- a/tests/fm-arm-pretool-check.test.sh +++ b/tests/fm-arm-pretool-check.test.sh @@ -439,95 +439,6 @@ test_allow_is_silent_both_modes() { # --- harness wiring: each adapter invokes the shared checker ----------------- -test_grok_pretool_hook_wired() { - local settings command - settings="$ROOT/.grok/hooks/fm-primary-pretool-check.json" - [ -f "$settings" ] || fail "tracked grok primary PreToolUse hook config is missing" - command=$(jq -r '.hooks.PreToolUse[0].hooks[0].command // empty' "$settings") - [ -n "$command" ] || fail "PreToolUse hook command is missing from grok primary hook config" - assert_contains "$command" 'GROK_WORKSPACE_ROOT' "grok pretool hook must anchor from GROK_WORKSPACE_ROOT" - assert_contains "$command" 'fm-arm-pretool-check.sh' "grok pretool hook must invoke the shared checker" - assert_contains "$command" 'exec "${GROK_WORKSPACE_ROOT:-}/bin/fm-arm-pretool-check.sh"' "grok pretool hook must forward its stdin payload unchanged to the checker" - # shellcheck disable=SC2016 # single quotes are deliberate: a literal needle string, not an expansion - assert_not_contains "$command" 'root=${GROK_WORKSPACE_ROOT' "grok pretool hook must not assign a bare \$root var (breaks grok's own \${VAR} pre-substitution; see docs/arm-pretool-check.md)" - local matcher - matcher=$(jq -r '.hooks.PreToolUse[0].matcher // empty' "$settings") - [ "$matcher" = "Bash" ] || fail "grok pretool hook must matcher-scope to Bash, got: $matcher" - pass ".grok primary hook: PreToolUse hook invokes the shared checker" -} - -test_grok_turnend_hook_uses_safe_var_pattern() { - local settings command - settings="$ROOT/.grok/hooks/fm-primary-turnend-guard.json" - [ -f "$settings" ] || fail "tracked grok primary Stop hook config is missing" - command=$(jq -r '.hooks.Stop[0].hooks[0].command // empty' "$settings") - # shellcheck disable=SC2016 # single quotes are deliberate: literal needle strings, not expansions - assert_not_contains "$command" 'root=${GROK_WORKSPACE_ROOT' "grok Stop hook must not assign a bare \$root var either (regression fixed 2026-07-09, docs/arm-pretool-check.md)" - # shellcheck disable=SC2016 - assert_contains "$command" '${GROK_WORKSPACE_ROOT:-}' "grok Stop hook must reference GROK_WORKSPACE_ROOT with an inline default every time" - pass ".grok primary hook: Stop hook uses the \${VAR:-} pattern throughout (no bare \$root)" -} - -test_claude_settings_pretool_hook_wired() { - local settings command - settings="$ROOT/.claude/settings.json" - [ -f "$settings" ] || fail "tracked claude primary settings are missing" - command=$(jq -r '.hooks.PreToolUse[0].hooks[0].command // empty' "$settings") - [ -n "$command" ] || fail "PreToolUse hook command is missing from claude primary settings" - assert_contains "$command" 'CLAUDE_PROJECT_DIR' "claude pretool hook must anchor via CLAUDE_PROJECT_DIR" - assert_contains "$command" 'fm-arm-pretool-check.sh' "claude pretool hook must invoke the shared checker" - assert_contains "$command" '--claude' "claude pretool hook must pass --claude so stdout stays empty on deny" - [ "$command" = '"$CLAUDE_PROJECT_DIR"/bin/fm-arm-pretool-check.sh --claude' ] \ - || fail "claude pretool hook must forward stdin directly with only --claude, got: $command" - local matcher - matcher=$(jq -r '.hooks.PreToolUse[0].matcher // empty' "$settings") - [ "$matcher" = "Bash" ] || fail "claude pretool hook must matcher-scope to Bash, got: $matcher" - pass ".claude/settings.json: PreToolUse hook invokes the shared checker with --claude" -} - -test_codex_hooks_pretool_wired() { - local settings command - settings="$ROOT/.codex/hooks.json" - [ -f "$settings" ] || fail "tracked codex primary hooks are missing" - command=$(jq -r '.hooks.PreToolUse[0].hooks[0].command // empty' "$settings") - [ -n "$command" ] || fail "PreToolUse hook command is missing from codex primary hooks" - assert_contains "$command" 'fm-arm-pretool-check.sh' "codex pretool hook must invoke the shared checker" - assert_contains "$command" 'pwd -P' "codex pretool hook must anchor to the hook process root like the Stop hook does" - assert_contains "$command" 'printf "%s" "$payload" | "$root/bin/fm-arm-pretool-check.sh"' "codex pretool hook must forward the exact captured payload to the checker" - local matcher - matcher=$(jq -r '.hooks.PreToolUse[0].matcher // empty' "$settings") - [ "$matcher" = "Bash" ] || fail "codex pretool hook must matcher-scope to Bash, got: $matcher" - pass ".codex/hooks.json: PreToolUse hook invokes the shared checker" -} - -test_opencode_pretool_plugin_wired() { - local plugin content - plugin="$ROOT/.opencode/plugins/fm-primary-pretool-check.js" - [ -f "$plugin" ] || fail "tracked opencode primary pretool plugin is missing" - content=$(cat "$plugin") - assert_contains "$content" 'tool.execute.before' "opencode pretool plugin must hook tool.execute.before" - assert_contains "$content" 'fm-arm-pretool-check.sh' "opencode pretool plugin must invoke the shared checker" - assert_contains "$content" 'const command = output?.args?.command;' "opencode must extract output.args.command exactly" - assert_contains "$content" '["--command", command]' "opencode must forward the exact command as one CLI argument" - assert_contains "$content" 'if (result.code !== 2) return;' "opencode must throw only for checker exit 2" - assert_contains "$content" 'throw new Error' "opencode pretool plugin must throw to block the tool call" - pass ".opencode primary plugin: tool.execute.before invokes the shared checker and blocks by throwing" -} - -test_pi_extension_carries_pretool_check() { - local ext content - ext="$ROOT/.pi/extensions/fm-primary-turnend-guard.ts" - [ -f "$ext" ] || fail "tracked pi primary extension is missing" - content=$(cat "$ext") - assert_contains "$content" 'tool_call' "pi extension must hook tool_call for the pretool seatbelt" - assert_contains "$content" 'fm-arm-pretool-check.sh' "pi extension must invoke the shared checker" - assert_contains "$content" 'String((event.input as { command?: unknown })?.command ?? "")' "pi must extract and string-coerce event.input.command exactly" - assert_contains "$content" 'const result = await runPretoolCheck(command);' "pi must forward the exact command to the checker" - assert_contains "$content" 'if (result.code !== 2) return {};' "pi must block only for checker exit 2" - assert_contains "$content" 'block: true' "pi extension must return block:true to deny" - pass ".pi primary extension: tool_call handler invokes the shared checker and can block" -} - # --- shellcheck (belt-and-suspenders; CI/CONTRIBUTING.md also runs this) ----- test_shellcheck_clean() { @@ -553,10 +464,4 @@ test_failopen_missing_node test_claude_mode_stdout_empty_on_deny test_default_mode_stdout_has_grok_json_on_deny test_allow_is_silent_both_modes -test_grok_pretool_hook_wired -test_grok_turnend_hook_uses_safe_var_pattern -test_claude_settings_pretool_hook_wired -test_codex_hooks_pretool_wired -test_opencode_pretool_plugin_wired -test_pi_extension_carries_pretool_check test_shellcheck_clean diff --git a/tests/fm-ask-user-authority.test.sh b/tests/fm-ask-user-authority.test.sh index c05d84946fd..89ec517fa1b 100644 --- a/tests/fm-ask-user-authority.test.sh +++ b/tests/fm-ask-user-authority.test.sh @@ -1,124 +1,13 @@ #!/usr/bin/env bash -# Scenario regressions for ask-user authority. -# -# Hi Bit PR 148 is motivating evidence only: yolo approved 31 ask-user finding -# groups, and a later audit classified 14 of 32 rounds as over-engineered after -# checkpoint-based gameplay verification expanded into continuous adversarial -# 60 Hz browser proof. -# The tests below enforce the general contract boundary without naming that -# project in the runtime policy. -# shellcheck disable=SC2016 +# Behavioral regressions for ask-user authority instructions generated by fm-brief. set -u # shellcheck source=tests/lib.sh . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" -AGENTS="$ROOT/AGENTS.md" -OWNER="$ROOT/.agents/skills/ask-user-authority/SKILL.md" BRIEF="$ROOT/bin/fm-brief.sh" -SECONDMATE="$ROOT/.agents/skills/secondmate-provisioning/SKILL.md" TMP_ROOT=$(fm_test_tmproot fm-ask-user-authority) -approval_contract() { - awk ' - /^### Selected delivery path and approval authority$/ { found = 1; next } - found && /^### Validate$/ { exit } - found { print } - ' "$AGENTS" -} - -test_owner_and_always_loaded_boundary() { - local contract trigger_count - contract=$(approval_contract) - - assert_contains "$contract" "only within the captain's original request and accepted task criteria" \ - "standing authority lost the accepted-contract boundary" - assert_contains "$contract" 'never approves an ask-user Fix that would materially expand that product or engineering contract' \ - "standing authority lost the contract-expansion exception" - assert_contains "$contract" 'destructive, irreversible, and security-sensitive choices remain stronger captain boundaries' \ - "contract expansion weakened stronger captain boundaries" - assert_contains "$contract" 'Complexity alone is not expansion' \ - "standing authority incorrectly treats complexity as expansion" - assert_contains "$contract" 'load `ask-user-authority`' \ - "standing authority lost the detailed-procedure trigger" - assert_contains "$contract" 'implementation worker never answers its own finding' \ - "implementation worker can answer its own finding" - - assert_present "$OWNER" "ask-user authority owner is missing" - assert_grep 'name: ask-user-authority' "$OWNER" "ask-user authority skill has the wrong name" - assert_grep 'user-invocable: false' "$OWNER" "ask-user authority skill must be agent-only" - assert_grep 'single owner of the decision procedure for ask-user findings' "$OWNER" \ - "ask-user authority skill does not declare ownership" - assert_grep 'With `yolo` off, every ask-user finding belongs to the captain' "$OWNER" \ - "detailed procedure permits autonomous ask-user decisions with yolo off" - trigger_count=$(grep -Fc -- '- `ask-user-authority` -' "$AGENTS") - [ "$trigger_count" -eq 1 ] || fail "ask-user-authority must have exactly one section 13 trigger, found $trigger_count" - assert_no_grep 'Hi Bit' "$AGENTS" "AGENTS.md encoded an incident-specific authority rule" - assert_no_grep 'Hi Bit' "$OWNER" "authority owner encoded an incident-specific rule" - pass "ask-user authority has one conditional owner and a concise always-loaded boundary" -} - -test_concrete_required_defect_stays_autonomous() { - assert_grep 'genuinely necessary to satisfy the accepted contract' "$OWNER" \ - "required concrete corrections no longer stay within standing authority" - assert_grep 'Fixing a concrete defect that violates an original acceptance criterion stays within `yolo` authority' "$OWNER" \ - "concrete acceptance-criterion defect scenario is missing" - pass "required concrete defect correction stays within yolo authority" -} - -test_continuous_monitoring_expansion_escalates() { - assert_grep 'continuous-monitoring requirement' "$OWNER" \ - "continuous monitoring is not classified as a possible contract expansion" - assert_grep 'continuous frame-by-frame monitoring when the accepted criterion requested checkpoint proof expands the contract' "$OWNER" \ - "checkpoint-to-continuous-monitoring escalation scenario is missing" - pass "continuous frame-by-frame proof escalates when only checkpoints were requested" -} - -test_repeated_same_theme_escalates_before_another_round() { - assert_grep 'Repeated same-theme findings require escalation before another Fix' "$OWNER" \ - "same-theme findings do not stop another autonomous fix round" - assert_grep 'preserving a questionable abstraction rather than closing independent defects' "$OWNER" \ - "same-theme escalation lost its causal distinction" - pass "repeated abstraction-preserving findings escalate before another fix round" -} - -test_stronger_security_boundary_survives() { - assert_grep 'genuinely security-sensitive choices always escalate' "$OWNER" \ - "security-sensitive choices no longer use the stronger captain boundary" - assert_grep 'genuinely security-sensitive action requires the captain under the stronger existing boundary' "$OWNER" \ - "security-sensitive scenario is missing" - pass "genuinely security-sensitive action still escalates" -} - -test_explicit_complex_architecture_stays_in_scope() { - assert_grep 'complex architecture that the captain explicitly requested' "$OWNER" \ - "explicitly requested complex architecture is not protected from complexity-only escalation" - assert_grep 'does not escalate merely because it is complex' "$OWNER" \ - "complexity alone still triggers escalation" - pass "explicitly requested complex architecture stays autonomous" -} - -test_reviewer_labels_are_evidence_not_authority() { - for label in correctness security fail-closed high-risk required; do - assert_grep "$label" "$OWNER" "reviewer-label evidence rule is missing '$label'" - done - assert_grep 'never as authority to broaden the task' "$OWNER" \ - "reviewer labels can still broaden the accepted contract" - pass "reviewer risk labels remain evidence rather than expansion authority" -} - -test_captain_escalation_is_decision_ready() { - for phrase in \ - 'original requirement or accepted task criterion' \ - 'proposed product or engineering contract expansion' \ - 'smallest alternative that complies with the accepted contract' \ - 'consequences of accepting and declining the expansion' \ - 'recommendation with the reason'; do - assert_grep "$phrase" "$OWNER" "captain-facing escalation lost '$phrase'" - done - pass "contract-expansion escalation carries all five decision elements" -} - test_primary_and_secondmate_instruction_generation() { local home ship charter home="$TMP_ROOT/home" @@ -139,23 +28,12 @@ test_primary_and_secondmate_instruction_generation() { FM_HOME="$home" FM_ROOT_OVERRIDE="$ROOT" FM_SECONDMATE_CHARTER='Handle sample work.' \ "$BRIEF" authority-mate --secondmate --no-projects >/dev/null 2>&1 charter="$home/data/authority-mate/brief.md" + # shellcheck disable=SC2016 # Backticks are literal generated Markdown. assert_grep 'The local `AGENTS.md` is your job description' "$charter" \ "generated secondmate charter does not load the tracked authority boundary" - assert_grep 'purely local fast-forward of tracked files' "$SECONDMATE" \ - "secondmate update owner no longer carries tracked instructions into homes" - assert_grep 'AGENTS.md re-read' "$SECONDMATE" \ - "running secondmates are not told to re-read updated tracked authority" assert_no_grep 'continuous frame-by-frame monitoring' "$charter" \ "generated secondmate charter duplicated the detailed authority procedure" - pass "primary workers and secondmates receive the authority rule through their normal instruction owners" + pass "primary workers and secondmates receive the authority rule through generated instructions" } -test_owner_and_always_loaded_boundary -test_concrete_required_defect_stays_autonomous -test_continuous_monitoring_expansion_escalates -test_repeated_same_theme_escalates_before_another_round -test_stronger_security_boundary_survives -test_explicit_complex_architecture_stays_in_scope -test_reviewer_labels_are_evidence_not_authority -test_captain_escalation_is_decision_ready test_primary_and_secondmate_instruction_generation diff --git a/tests/fm-backend-cmux.test.sh b/tests/fm-backend-cmux.test.sh index daa6e5c4f97..152f78eb7c4 100755 --- a/tests/fm-backend-cmux.test.sh +++ b/tests/fm-backend-cmux.test.sh @@ -387,7 +387,7 @@ test_ping_state_down() { dir="$TMP_ROOT/ping-down"; mkdir -p "$dir/responses" fb=$(make_cmux_fakebin "$dir") out=$( PATH="$fb:$PATH" FM_CMUX_LOG="$dir/log" FM_CMUX_RESPONSES="$dir/responses" FM_CMUX_FAKE_PING_EXIT=1 \ - FM_CMUX_FAKE_PING="Error: Socket not found at /Users/x/.local/state/cmux/cmux.sock" \ + FM_CMUX_FAKE_PING="Error: Socket not found at /home/x/.local/state/cmux/cmux.sock" \ bash -c '. "$0/bin/backends/cmux.sh"; fm_backend_cmux_ping_state' "$ROOT" ) [ "$out" = down ] || fail "ping_state should report down when the socket does not exist yet, got '$out'" pass "fm_backend_cmux_ping_state: reports 'down' when the app is not running yet" @@ -657,11 +657,11 @@ test_current_path_probes_with_marker() { cmux_panes_response "$dir" 2 "bbbbbbbb-1111-1111-1111-111111111111" cmux_panes_response "$dir" 4 "bbbbbbbb-1111-1111-1111-111111111111" cmux_panes_response "$dir" 6 "bbbbbbbb-1111-1111-1111-111111111111" - cmux_read_screen_response "$dir" 7 $'/tmp/proj\n❯ printf marker\n__FM_CMUX_CWD_BEGIN__\n/Users/kunchen/.treehouse/fake-worktree\n__FM_CMUX_CWD_END__\n/Users/kunchen/.treehouse/fake-worktree ❯' + cmux_read_screen_response "$dir" 7 $'/tmp/proj\n❯ printf marker\n__FM_CMUX_CWD_BEGIN__\n/home/fixture/.treehouse/fake-worktree\n__FM_CMUX_CWD_END__\n/home/fixture/.treehouse/fake-worktree ❯' fb=$(make_cmux_fakebin "$dir") out=$( PATH="$fb:$PATH" FM_CMUX_LOG="$dir/log" FM_CMUX_RESPONSES="$dir/responses" \ bash -c '. "$0/bin/backends/cmux.sh"; fm_backend_cmux_current_path "aaaaaaaa-0000-0000-0000-000000000000:bbbbbbbb-1111-1111-1111-111111111111"' "$ROOT" ) - [ "$out" = "/Users/kunchen/.treehouse/fake-worktree" ] || fail "current_path should read only the marked cwd line, got '$out'" + [ "$out" = "/home/fixture/.treehouse/fake-worktree" ] || fail "current_path should read only the marked cwd line, got '$out'" assert_contains "$(cat "$dir/log")" "__FM_CMUX_CWD_BEGIN__" "current_path did not send the cwd begin marker" assert_contains "$(cat "$dir/log")" "pwd;" "current_path did not send the pwd probe" assert_contains "$(cat "$dir/log")" $'\x1f''send-key'$'\x1f''--workspace'$'\x1f''aaaaaaaa-0000-0000-0000-000000000000'$'\x1f''--surface'$'\x1f''bbbbbbbb-1111-1111-1111-111111111111'$'\x1f''enter' \ diff --git a/tests/fm-backend-herdr-presentation-e2e.test.sh b/tests/fm-backend-herdr-presentation-e2e.test.sh index 194d2053ce9..fb44305e99e 100755 --- a/tests/fm-backend-herdr-presentation-e2e.test.sh +++ b/tests/fm-backend-herdr-presentation-e2e.test.sh @@ -866,7 +866,7 @@ touch "$SECOND_HOME_A/state/.last-watcher-beat" "$SECOND_HOME_B/state/.last-watc # may write config/herdr-presentation-spaces. git -C "$SECOND_HOME_A" init -q git -C "$SECOND_HOME_B" init -q -printf 'config/herdr-presentation-spaces\nconfig/crew-harness\nconfig/crew-dispatch.json\nconfig/backlog-backend\n' \ +printf 'config/herdr-presentation-spaces\nconfig/crew-harness\nconfig/crew-dispatch.json\nconfig/backlog-backend\nconfig/backend\nconfig/startup-memory-budget\n' \ > "$SECOND_HOME_A/.gitignore" cp "$SECOND_HOME_A/.gitignore" "$SECOND_HOME_B/.gitignore" git -C "$SECOND_HOME_A" add .gitignore diff --git a/tests/fm-backend-herdr.test.sh b/tests/fm-backend-herdr.test.sh index b2e980d9644..92ecbdcb925 100755 --- a/tests/fm-backend-herdr.test.sh +++ b/tests/fm-backend-herdr.test.sh @@ -1217,30 +1217,6 @@ test_presentation_session_lock_path_rejects_malformed_socket() { pass "herdr presentation lock: null and missing socket paths fail closed" } -test_presentation_lock_malformed_socket_falls_back() { - local dir log resp fb out status lock_source - dir="$TMP_ROOT/presentation-malformed-socket-fallback"; mkdir -p "$dir/responses" - log="$dir/log"; resp="$dir/responses"; : > "$log" - printf '%s\n' '{"sessions":[{"name":"fmtest","running":true,"socket_path":null}]}' > "$resp/1.out" - fb=$(make_herdr_fakebin "$dir") - lock_source=$(sed -n '/^spawn_herdr_presentation_order_lock_acquire()/,/^spawn_herdr_presentation_order_lock_release()/p' "$ROOT/bin/fm-spawn.sh" | sed '$d') - out=$(PATH="$fb:$PATH" FM_HERDR_LOG="$log" FM_HERDR_RESPONSES="$resp" \ - LOCK_SOURCE="$lock_source" \ - bash -c ' - . "$0/bin/backends/herdr.sh" - eval "$LOCK_SOURCE" - if spawn_herdr_presentation_order_lock_acquire fmtest; then - printf "%s" acquired - else - printf "%s" flat - fi - ' "$ROOT" 2>&1) - status=$? - [ "$status" -eq 0 ] || fail "malformed socket fallback must not fail the spawn path: $out" - [ "$out" = flat ] || fail "malformed socket_path must fall back flat, got '$out'" - pass "herdr presentation lock: malformed socket metadata degrades to flat" -} - test_projection_order_rejects_malformed_socket() { local dir log resp fb mover out status dir="$TMP_ROOT/projection-order-malformed-socket"; mkdir -p "$dir/responses" @@ -1267,117 +1243,6 @@ SH pass "herdr presentation ordering: malformed socket metadata is warning-only and read-only" } -test_presentation_lock_insecure_namespace_falls_back() { - local dir log resp fb bad out status lock_source - dir="$TMP_ROOT/presentation-insecure-lock"; mkdir -p "$dir/responses" "$dir/sockdir" - log="$dir/log"; resp="$dir/responses"; : > "$log" - : > "$dir/sockdir/fmtest.sock" - bad="$dir/insecure"; mkdir -m 755 "$bad" - printf '%s\n' "{\"sessions\":[{\"name\":\"fmtest\",\"running\":true,\"socket_path\":\"$dir/sockdir/fmtest.sock\"}]}" > "$resp/1.out" - fb=$(make_herdr_fakebin "$dir") - lock_source=$(sed -n '/^spawn_herdr_presentation_order_lock_acquire()/,/^spawn_herdr_presentation_order_lock_release()/p' "$ROOT/bin/fm-spawn.sh" | sed '$d') - out=$(PATH="$fb:$PATH" FM_HERDR_LOG="$log" FM_HERDR_RESPONSES="$resp" \ - BAD_NAMESPACE="$bad" LOCK_SOURCE="$lock_source" \ - bash -c ' - . "$0/bin/backends/herdr.sh" - eval "$LOCK_SOURCE" - fm_backend_herdr_presentation_lock_namespace() { printf "%s" "$BAD_NAMESPACE"; } - if spawn_herdr_presentation_order_lock_acquire fmtest; then - printf "%s" acquired - else - printf "%s" flat - fi - ' "$ROOT" 2>&1) - status=$? - [ "$status" -eq 0 ] || fail "an insecure lock namespace must not fail the spawn path: $out" - [ "$out" = flat ] || fail "an insecure lock namespace must fall back flat, got '$out'" - pass "herdr presentation lock: insecure shared namespace refuses acquisition for flat fallback" -} - -test_spawn_task_lock_covers_all_backend_creation_and_metadata_publication() { - local source wake_source acquire_pattern backend_pattern meta_pattern acquire_line backend_line meta_line - source=$(cat "$ROOT/bin/fm-spawn.sh") - wake_source=". \"\$SCRIPT_DIR/fm-wake-lib.sh\"" - acquire_pattern="fm_lock_try_acquire \"\$SPAWN_TASK_LOCK\"" - backend_pattern="^case \"\$BACKEND\" in" - meta_pattern="} > \"\$STATE/\$ID.meta\"" - assert_contains "$source" "$wake_source" \ - "fm-spawn does not load the shared lock implementation" - acquire_line=$(grep -n "$acquire_pattern" "$ROOT/bin/fm-spawn.sh" | head -1 | cut -d: -f1) - backend_line=$(grep -n "$backend_pattern" "$ROOT/bin/fm-spawn.sh" | tail -1 | cut -d: -f1) - meta_line=$(grep -n "$meta_pattern" "$ROOT/bin/fm-spawn.sh" | tail -1 | cut -d: -f1) - [ -n "$acquire_line" ] && [ -n "$backend_line" ] && [ -n "$meta_line" ] \ - || fail "could not locate the spawn lock, backend creation, and metadata publication" - [ "$acquire_line" -lt "$backend_line" ] && [ "$backend_line" -lt "$meta_line" ] \ - || fail "the task lock does not span backend creation through metadata publication" - pass "fm-spawn: one task lock spans every backend creation path through metadata publication" -} - -test_projected_spawn_disarms_cleanup_before_ambiguous_launch_submission() { - local literal_pattern disarm_pattern release_pattern enter_pattern literal_line disarm_line release_line enter_line - # These are literal source patterns for grep, so shell expansion would invalidate the assertion. - # shellcheck disable=SC2016 - literal_pattern='spawn_send_literal "$T" "$LAUNCH"' - # shellcheck disable=SC2016 - disarm_pattern='HERDR_PROJECTION_ABORT_CLEANUP=0' - release_pattern='spawn_herdr_presentation_order_lock_release' - # shellcheck disable=SC2016 - enter_pattern='spawn_send_key "$T" Enter' - literal_line=$(grep -nF "$literal_pattern" "$ROOT/bin/fm-spawn.sh" | tail -1 | cut -d: -f1) - disarm_line=$(grep -nF "$disarm_pattern" "$ROOT/bin/fm-spawn.sh" | tail -1 | cut -d: -f1) - release_line=$(grep -nF "$release_pattern" "$ROOT/bin/fm-spawn.sh" | tail -1 | cut -d: -f1) - enter_line=$(grep -nF "$enter_pattern" "$ROOT/bin/fm-spawn.sh" | tail -1 | cut -d: -f1) - [ -n "$literal_line" ] && [ -n "$disarm_line" ] && [ -n "$release_line" ] && [ -n "$enter_line" ] \ - || fail "could not locate the projected launch cleanup boundary" - [ "$literal_line" -lt "$disarm_line" ] \ - && [ "$disarm_line" -lt "$release_line" ] \ - && [ "$release_line" -lt "$enter_line" ] \ - || fail "projected spawn must disarm cleanup before releasing its lock and submitting ambiguous Enter" - pass "fm-spawn: projected cleanup disarms before lock release and ambiguous launch submission" -} - -test_projected_abort_cleanup_holds_presentation_lock() { - local dir lock started proceed function_source owner_pid status - dir="$TMP_ROOT/projection-abort-lock"; mkdir -p "$dir" - lock="$dir/presentation.lock" - started="$dir/cleanup-started" - proceed="$dir/cleanup-proceed" - function_source=$(sed -n '/^spawn_abort_cleanup()/,/^trap spawn_abort_cleanup EXIT/p' "$ROOT/bin/fm-spawn.sh" | sed '$d') - ROOT="$ROOT" LOCK="$lock" STARTED="$started" PROCEED="$proceed" FUNCTION_SOURCE="$function_source" bash -c ' - . "$ROOT/bin/fm-wake-lib.sh" - eval "$FUNCTION_SOURCE" - fm_backend_herdr_projection_cleanup_exact() { - : > "$STARTED" - while [ ! -e "$PROCEED" ]; do sleep 0.01; done - } - fm_lock_try_acquire "$LOCK" || exit 1 - HERDR_PRESENTATION_ORDER_LOCK_HELD=1 - HERDR_PRESENTATION_ORDER_LOCK=$LOCK - HERDR_PROJECTION_ABORT_CLEANUP=1 - HERDR_PROJECTION_ABORT_SESSION=fmtest - HERDR_PROJECTION_ABORT_TASK_PANE=w9:p2 - HERDR_PROJECTION_ABORT_SEEDED_PANE=w9:p1 - ORCA_ABORT_CLEANUP=0 - SPAWN_TASK_LOCK_HELD=0 - spawn_abort_cleanup - ' & - owner_pid=$! - while [ ! -e "$started" ] && kill -0 "$owner_pid" 2>/dev/null; do sleep 0.01; done - [ -e "$started" ] || fail "projected abort cleanup did not start" - if LOCK="$lock" ROOT="$ROOT" bash -c '. "$ROOT/bin/fm-wake-lib.sh"; fm_lock_try_acquire "$LOCK"'; then - : > "$proceed" - wait "$owner_pid" || true - fail "concurrent presentation work acquired the lock during abort cleanup" - fi - : > "$proceed" - wait "$owner_pid" - status=$? - [ "$status" -eq 0 ] || fail "projected abort cleanup owner failed" - LOCK="$lock" ROOT="$ROOT" bash -c '. "$ROOT/bin/fm-wake-lib.sh"; fm_lock_try_acquire "$LOCK"' \ - || fail "presentation lock remained held after abort cleanup" - pass "fm-spawn: projected abort cleanup remains serialized by the presentation lock" -} - test_projection_reclaim_refusal_matrix_is_non_mutating() { local dir state home other_home home_real journal legacy token label out mutation_log dir="$TMP_ROOT/projection-reclaim-refusals"; state="$dir/state"; home="$dir/home"; other_home="$dir/other-home" @@ -2647,23 +2512,6 @@ EOF pass "fm_backend_herdr_workspace_prune_seeded_default_tab: refuses to close the seeded default tab when its pane reports a working agent (defense in depth)" } -# test_no_jq_reserved_keyword_arg_names: regression guard for the -# workspace-leak root cause (a jq `--arg`/`--argjson` named after a jq -# reserved keyword, e.g. `label`, is a compile error on jq <= 1.6; this -# adapter discards jq's stderr, so the error silently becomes an empty -# result instead of a visible failure). Greps every bin/ script for the -# pattern so a future filter reintroducing it fails loudly here instead of -# silently misbehaving on an older jq. -test_no_jq_reserved_keyword_arg_names() { - local reserved='and|as|catch|def|elif|else|end|foreach|if|import|include|label|module|or|reduce|then|try' - local hits - hits=$(grep -rnE -- "--arg(json)?[[:space:]]+($reserved)\b" "$ROOT/bin" 2>/dev/null) - if [ -n "$hits" ]; then - fail "a jq --arg/--argjson variable is named after a jq reserved keyword (compile error on jq <= 1.6, silently swallowed by 2>/dev/null):"$'\n'"$hits" - fi - pass "no bin/ jq filter names a --arg/--argjson variable after a jq reserved keyword" -} - # --- native event push: normalize / policy-routing / dedupe / wait ---------- # # These exercise the herdr subscriber (fm_backend_herdr_wait_transition and its @@ -2991,7 +2839,6 @@ test_repeated_cycles_reuse_one_workspace_no_orphans test_adopted_workspace_never_prunes_default_tab test_label_collision_startup_workspace_leaves_live_tab_alone test_prune_refuses_a_working_agent_pane_defense_in_depth -test_no_jq_reserved_keyword_arg_names test_create_task_refuses_duplicate_label test_create_task_refuses_duplicate_label_when_agent_live test_create_task_refuses_when_any_duplicate_label_is_live @@ -3025,12 +2872,7 @@ test_projection_order_foreign_new_child_before_parent_is_read_only test_projection_order_missing_parent_is_read_only test_presentation_session_lock_path_is_shared_across_homes test_presentation_session_lock_path_rejects_malformed_socket -test_presentation_lock_malformed_socket_falls_back test_projection_order_rejects_malformed_socket -test_presentation_lock_insecure_namespace_falls_back -test_spawn_task_lock_covers_all_backend_creation_and_metadata_publication -test_projected_spawn_disarms_cleanup_before_ambiguous_launch_submission -test_projected_abort_cleanup_holds_presentation_lock test_projection_reclaim_refusal_matrix_is_non_mutating test_projection_reclaim_replaces_only_exact_husk_and_advances_binding test_projection_recovery_is_read_only_and_refuses_live_duplicate_risk diff --git a/tests/fm-backend-orca.test.sh b/tests/fm-backend-orca.test.sh index 66c3dd36535..a54e448d108 100755 --- a/tests/fm-backend-orca.test.sh +++ b/tests/fm-backend-orca.test.sh @@ -702,7 +702,7 @@ test_peek_send_and_crew_state_route_through_orca_meta() { fm_git_init_commit "$wt" state="$TMP_ROOT/io-state"; mkdir -p "$state" fm_write_meta "$state/$id.meta" \ - "window=fm-$id" "terminal=term-io" "worktree=$wt" "project=$wt" "harness=claude" "kind=scout" "backend=orca" + "window=fm-$id" "endpoint_task_id=$id" "terminal=term-io" "worktree=$wt" "project=$wt" "harness=claude" "kind=scout" "backend=orca" touch "$state/.last-watcher-beat" orca_case io-path neutral=$(neutral_fm_root "$CASE_DIR/neutral") @@ -739,7 +739,7 @@ test_peek_and_crew_state_fail_closed_on_orca_error_json() { fm_git_init_commit "$wt" state="$TMP_ROOT/read-error-state"; mkdir -p "$state" fm_write_meta "$state/$id.meta" \ - "window=fm-$id" "terminal=term-stale" "worktree=$wt" "project=$wt" "harness=claude" "kind=scout" "backend=orca" + "window=fm-$id" "endpoint_task_id=$id" "terminal=term-stale" "worktree=$wt" "project=$wt" "harness=claude" "kind=scout" "backend=orca" touch "$state/.last-watcher-beat" orca_case read-error-json neutral=$(neutral_fm_root "$CASE_DIR/neutral") @@ -785,7 +785,7 @@ test_scout_teardown_removes_orca_worktree_via_helper() { printf 'report\n' > "$data/$id/report.md" touch "$state/.last-watcher-beat" fm_write_meta "$state/$id.meta" \ - "window=fm-$id" "terminal=term-teardown" "worktree=$wt" "project=$proj" \ + "window=fm-$id" "endpoint_task_id=$id" "terminal=term-teardown" "worktree=$wt" "project=$proj" \ "harness=claude" "kind=scout" "mode=no-mistakes" "yolo=off" \ "backend=orca" "orca_worktree_id=wt-teardown" \ "decisions_reviewed=1" "decision_keys=" @@ -822,7 +822,7 @@ test_scout_teardown_refuses_orca_id_path_mismatch() { printf 'report\n' > "$data/$id/report.md" touch "$state/.last-watcher-beat" fm_write_meta "$state/$id.meta" \ - "window=fm-$id" "terminal=term-scout-mismatch" "worktree=$wt" "project=$proj" \ + "window=fm-$id" "endpoint_task_id=$id" "terminal=term-scout-mismatch" "worktree=$wt" "project=$proj" \ "harness=claude" "kind=scout" "mode=no-mistakes" "yolo=off" \ "backend=orca" "orca_worktree_id=wt-scout-mismatch" \ "decisions_reviewed=1" "decision_keys=" @@ -858,7 +858,7 @@ test_teardown_removes_orca_worktree_when_path_missing() { printf 'report\n' > "$data/$id/report.md" touch "$state/.last-watcher-beat" fm_write_meta "$state/$id.meta" \ - "window=fm-$id" "terminal=term-missing-path" "worktree=$wt" "project=$proj" \ + "window=fm-$id" "endpoint_task_id=$id" "terminal=term-missing-path" "worktree=$wt" "project=$proj" \ "harness=claude" "kind=scout" "mode=no-mistakes" "yolo=off" \ "backend=orca" "orca_worktree_id=wt-missing-path" \ "decisions_reviewed=1" "decision_keys=" @@ -891,12 +891,13 @@ test_teardown_preserves_metadata_when_orca_remove_error_json() { printf 'report\n' > "$data/$id/report.md" touch "$state/.last-watcher-beat" fm_write_meta "$state/$id.meta" \ - "window=fm-$id" "worktree=$wt" "project=$proj" \ + "window=fm-$id" "endpoint_task_id=$id" "terminal=term-remove-error" "worktree=$wt" "project=$proj" \ "harness=claude" "kind=scout" "mode=no-mistakes" "yolo=off" \ "backend=orca" "orca_worktree_id=wt-remove-error" \ "decisions_reviewed=1" "decision_keys=" orca_case remove-error-teardown - printf '{"ok":false,"error":{"code":"worktree_not_removed","message":"worktree not removed"}}\n' > "$RESP/1.out" + printf '{"ok":true,"result":{}}\n' > "$RESP/1.out" + printf '{"ok":false,"error":{"code":"worktree_not_removed","message":"worktree not removed"}}\n' > "$RESP/2.out" neutral=$(neutral_fm_root "$CASE_DIR/neutral") set +e out=$( PATH="$FB:$PATH" FM_ORCA_LOG="$LOG" FM_ORCA_RESPONSES="$RESP" \ @@ -921,7 +922,7 @@ test_scout_teardown_refuses_orca_missing_report_when_path_missing() { mkdir -p "$data/$id" "$state" "$config" touch "$state/.last-watcher-beat" fm_write_meta "$state/$id.meta" \ - "window=fm-$id" "terminal=term-missing-report" "worktree=$wt" "project=$proj" \ + "window=fm-$id" "endpoint_task_id=$id" "terminal=term-missing-report" "worktree=$wt" "project=$proj" \ "harness=claude" "kind=scout" "mode=no-mistakes" "yolo=off" \ "backend=orca" "orca_worktree_id=wt-missing-report" orca_case missing-report @@ -951,7 +952,7 @@ test_ship_teardown_refuses_orca_missing_worktree_path() { mkdir -p "$data/$id" "$state" "$config" touch "$state/.last-watcher-beat" fm_write_meta "$state/$id.meta" \ - "window=fm-$id" "terminal=term-missing-ship" "worktree=$wt" "project=$proj" \ + "window=fm-$id" "endpoint_task_id=$id" "terminal=term-missing-ship" "worktree=$wt" "project=$proj" \ "harness=claude" "kind=ship" "mode=no-mistakes" "yolo=off" \ "backend=orca" "orca_worktree_id=wt-missing-ship" orca_case missing-ship-path @@ -982,7 +983,7 @@ test_ship_teardown_removes_orca_worktree_when_id_path_matches() { mkdir -p "$data/$id" "$state" "$config" touch "$state/.last-watcher-beat" fm_write_meta "$state/$id.meta" \ - "window=fm-$id" "terminal=term-ship-match" "worktree=$wt" "project=$proj" \ + "window=fm-$id" "endpoint_task_id=$id" "terminal=term-ship-match" "worktree=$wt" "project=$proj" \ "harness=claude" "kind=ship" "mode=local-only" "yolo=off" \ "backend=orca" "orca_worktree_id=wt-ship-match" orca_case ship-match @@ -1017,7 +1018,7 @@ test_ship_teardown_refuses_orca_unresolvable_worktree_id() { mkdir -p "$data/$id" "$state" "$config" touch "$state/.last-watcher-beat" fm_write_meta "$state/$id.meta" \ - "window=fm-$id" "terminal=term-ship-unresolved" "worktree=$wt" "project=$proj" \ + "window=fm-$id" "endpoint_task_id=$id" "terminal=term-ship-unresolved" "worktree=$wt" "project=$proj" \ "harness=claude" "kind=ship" "mode=local-only" "yolo=off" \ "backend=orca" "orca_worktree_id=wt-ship-unresolved" orca_case ship-unresolved @@ -1056,7 +1057,7 @@ test_ship_teardown_refuses_orca_id_path_mismatch() { mkdir -p "$data/$id" "$state" "$config" touch "$state/.last-watcher-beat" fm_write_meta "$state/$id.meta" \ - "window=fm-$id" "terminal=term-ship-mismatch" "worktree=$wt" "project=$proj" \ + "window=fm-$id" "endpoint_task_id=$id" "terminal=term-ship-mismatch" "worktree=$wt" "project=$proj" \ "harness=claude" "kind=ship" "mode=local-only" "yolo=off" \ "backend=orca" "orca_worktree_id=wt-ship-mismatch" orca_case ship-mismatch @@ -1094,7 +1095,7 @@ test_teardown_refuses_orca_missing_worktree_id() { printf 'report\n' > "$data/$id/report.md" touch "$state/.last-watcher-beat" fm_write_meta "$state/$id.meta" \ - "window=fm-$id" "terminal=term-missing-id" "worktree=$wt" "project=$proj" \ + "window=fm-$id" "endpoint_task_id=$id" "terminal=term-missing-id" "worktree=$wt" "project=$proj" \ "harness=claude" "kind=scout" "mode=no-mistakes" "yolo=off" "backend=orca" \ "decisions_reviewed=1" "decision_keys=" orca_case missing-id @@ -1112,7 +1113,7 @@ test_teardown_refuses_orca_missing_worktree_id() { pass "fm-teardown.sh backend=orca: refuses missing worktree ids before cleanup" } -test_teardown_removes_orca_worktree_without_terminal_handle() { +test_teardown_refuses_orca_worktree_without_terminal_handle() { local proj wt data state config id out rc neutral id="orcanotermz0" proj="$TMP_ROOT/no-terminal-project" @@ -1125,12 +1126,11 @@ test_teardown_removes_orca_worktree_without_terminal_handle() { printf 'report\n' > "$data/$id/report.md" touch "$state/.last-watcher-beat" fm_write_meta "$state/$id.meta" \ - "window=fm-$id" "worktree=$wt" "project=$proj" \ + "window=fm-$id" "endpoint_task_id=$id" "worktree=$wt" "project=$proj" \ "harness=claude" "kind=scout" "mode=no-mistakes" "yolo=off" \ "backend=orca" "orca_worktree_id=wt-no-terminal" \ "decisions_reviewed=1" "decision_keys=" orca_case no-terminal - printf '{"ok":true,"result":{"worktree":{"id":"wt-no-terminal","path":"%s"}}}\n' "$wt" > "$RESP/1.out" neutral=$(neutral_fm_root "$CASE_DIR/neutral") set +e out=$( PATH="$FB:$PATH" FM_ORCA_LOG="$LOG" FM_ORCA_RESPONSES="$RESP" \ @@ -1138,13 +1138,11 @@ test_teardown_removes_orca_worktree_without_terminal_handle() { "$ROOT/bin/fm-teardown.sh" "$id" 2>&1 ) rc=$? set -e - expect_code 0 "$rc" "Orca teardown should remove a worktree even when no terminal was ever recorded"$'\n'"$out" - assert_contains "$(cat "$LOG")" $'orca\x1f''worktree'$'\x1f''rm'$'\x1f''--worktree'$'\x1f''id:wt-no-terminal'$'\x1f''--force'$'\x1f''--json' \ - "teardown did not remove the partial Orca worktree" - assert_not_contains "$(cat "$LOG")" $'orca\x1f''terminal'$'\x1f''close' \ - "teardown should not close a terminal when no terminal handle is recorded" - assert_absent "$state/$id.meta" "successful partial cleanup should remove task metadata" - pass "fm-teardown.sh backend=orca: removes partial worktree-only metadata" + [ "$rc" -ne 0 ] || fail "Orca teardown accepted metadata without a terminal handle" + assert_contains "$out" "missing terminal" "teardown did not explain the incomplete Orca endpoint" + [ ! -s "$LOG" ] || fail "teardown dispatched to Orca before rejecting the incomplete endpoint" + assert_present "$state/$id.meta" "missing-terminal refusal removed task metadata" + pass "fm-teardown.sh backend=orca: refuses incomplete worktree-only endpoint metadata before runtime dispatch" } test_secondmate_force_teardown_removes_orca_child_via_orca() { @@ -1164,7 +1162,8 @@ test_secondmate_force_teardown_removes_orca_child_via_orca() { printf '%s\n' "- domain - Orca child cleanup (home: $subhome; scope: orca cleanup; projects: alpha; added 2026-07-03)" \ > "$home/data/secondmates.md" fm_write_meta "$subhome/state/$child_id.meta" \ - "window=fm-$child_id" "terminal=term-child-cleanup" "worktree=$childwt" "project=$childproj" \ + "window=fm-$child_id" "endpoint_task_id=$child_id" \ + "terminal=term-child-cleanup" "worktree=$childwt" "project=$childproj" \ "harness=claude" "kind=ship" "mode=no-mistakes" "yolo=off" \ "backend=orca" "orca_worktree_id=wt-child-cleanup" orca_case secondmate-child-cleanup @@ -1206,7 +1205,8 @@ test_secondmate_force_teardown_refuses_orca_child_id_path_mismatch() { printf '%s\n' "- domain - Orca child cleanup (home: $subhome; scope: orca cleanup; projects: alpha; added 2026-07-03)" \ > "$home/data/secondmates.md" fm_write_meta "$subhome/state/$child_id.meta" \ - "window=fm-$child_id" "terminal=term-child-mismatch" "worktree=$childwt" "project=$childproj" \ + "window=fm-$child_id" "endpoint_task_id=$child_id" \ + "terminal=term-child-mismatch" "worktree=$childwt" "project=$childproj" \ "harness=claude" "kind=ship" "mode=no-mistakes" "yolo=off" \ "backend=orca" "orca_worktree_id=wt-child-mismatch" orca_case secondmate-child-mismatch @@ -1229,7 +1229,7 @@ test_secondmate_force_teardown_refuses_orca_child_id_path_mismatch() { pass "fm-teardown.sh --force: refuses Orca child id/path mismatches" } -test_secondmate_force_teardown_removes_partial_orca_child() { +test_secondmate_force_teardown_refuses_partial_orca_child() { local home subhome childproj childwt child_id neutral out rc home="$TMP_ROOT/orca-partial-child-parent" subhome="$TMP_ROOT/orca-partial-child-secondmate" @@ -1246,11 +1246,11 @@ test_secondmate_force_teardown_removes_partial_orca_child() { printf '%s\n' "- domain - Orca partial child cleanup (home: $subhome; scope: orca cleanup; projects: alpha; added 2026-07-03)" \ > "$home/data/secondmates.md" fm_write_meta "$subhome/state/$child_id.meta" \ - "window=fm-$child_id" "worktree=$childwt" "project=$childproj" \ + "window=fm-$child_id" "endpoint_task_id=$child_id" \ + "worktree=$childwt" "project=$childproj" \ "harness=claude" "kind=ship" "mode=no-mistakes" "yolo=off" \ "backend=orca" "orca_worktree_id=wt-partial-child" orca_case secondmate-partial-child-cleanup - printf '{"ok":true,"result":{"worktree":{"id":"wt-partial-child","path":"%s"}}}\n' "$childwt" > "$RESP/1.out" add_tmux_fake "$FB" neutral=$(neutral_fm_root "$CASE_DIR/neutral") set +e @@ -1258,13 +1258,12 @@ test_secondmate_force_teardown_removes_partial_orca_child() { FM_ROOT_OVERRIDE="$neutral" FM_HOME="$home" "$ROOT/bin/fm-teardown.sh" domain --force 2>&1 ) rc=$? set -e - expect_code 0 "$rc" "forced secondmate teardown should remove partial Orca child state"$'\n'"$out" - assert_contains "$(cat "$LOG")" $'orca\x1f''worktree'$'\x1f''rm'$'\x1f''--worktree'$'\x1f''id:wt-partial-child'$'\x1f''--force'$'\x1f''--json' \ - "partial child cleanup did not remove the Orca worktree through orca worktree rm" - assert_not_contains "$(cat "$LOG")" $'orca\x1f''terminal'$'\x1f''close' \ - "partial child cleanup should not close a terminal when no terminal handle is recorded" - assert_absent "$home/state/domain.meta" "parent metadata should be removed after forced partial cleanup" - pass "fm-teardown.sh --force: removes partial Orca secondmate children" + [ "$rc" -ne 0 ] || fail "forced secondmate teardown accepted a child with no terminal identity" + assert_contains "$out" "missing terminal" "partial child refusal did not explain the incomplete endpoint" + [ ! -s "$LOG" ] || fail "partial child refusal dispatched to Orca or tmux" + assert_present "$home/state/domain.meta" "partial child refusal removed parent metadata" + assert_present "$subhome/state/$child_id.meta" "partial child refusal removed child metadata" + pass "fm-teardown.sh --force: refuses partial Orca secondmate children before runtime dispatch" } test_dispatcher_sources_orca_and_routes_primitives() { @@ -1323,7 +1322,7 @@ test_ship_teardown_removes_orca_worktree_when_id_path_matches test_ship_teardown_refuses_orca_unresolvable_worktree_id test_ship_teardown_refuses_orca_id_path_mismatch test_teardown_refuses_orca_missing_worktree_id -test_teardown_removes_orca_worktree_without_terminal_handle +test_teardown_refuses_orca_worktree_without_terminal_handle test_secondmate_force_teardown_removes_orca_child_via_orca test_secondmate_force_teardown_refuses_orca_child_id_path_mismatch -test_secondmate_force_teardown_removes_partial_orca_child +test_secondmate_force_teardown_refuses_partial_orca_child diff --git a/tests/fm-backend-zellij.test.sh b/tests/fm-backend-zellij.test.sh index e7a8f143d03..ae4be257bdf 100755 --- a/tests/fm-backend-zellij.test.sh +++ b/tests/fm-backend-zellij.test.sh @@ -664,18 +664,18 @@ test_current_path_probes_with_marker_and_ignores_prompt_paths() { zellij_pane_response "$dir" 4 7 3 zellij_pane_response "$dir" 6 7 3 printf '%s\n' 'scratch-e2e-project HEAD' \ - '/Users/kunchen/src/project ❯ printf marker' \ + '/home/fixture/src/project ❯ printf marker' \ '__FM_ZELLIJ_CWD_BEGIN__' \ - '/Users/kunchen/.treehouse/fake-' \ + '/home/fixture/.treehouse/fake-' \ 'worktree' \ '__FM_ZELLIJ_CWD_END__' \ - '/Users/kunchen/.treehouse/fake-worktree ❯' \ + '/home/fixture/.treehouse/fake-worktree ❯' \ > "$dir/responses/7.out" fb=$(make_zellij_fakebin "$dir") out=$( PATH="$fb:$PATH" FM_ZELLIJ_LOG="$dir/log" FM_ZELLIJ_RESPONSES="$dir/responses" \ FM_ZELLIJ_SESSION_LIST="firstmate" \ bash -c '. "$0/bin/backends/zellij.sh"; fm_backend_zellij_current_path firstmate:7' "$ROOT" ) - [ "$out" = "/Users/kunchen/.treehouse/fake-worktree" ] || fail "current_path should read only the marked cwd line, got '$out'" + [ "$out" = "/home/fixture/.treehouse/fake-worktree" ] || fail "current_path should read only the marked cwd line, got '$out'" zellij_assert_call_order "$dir/log" $'\x1f''list-panes'$'\x1f''--json' $'\x1f''paste' \ "current_path did not verify the pane before the cwd probe paste" zellij_assert_call_order "$dir/log" $'\x1f''list-panes'$'\x1f''--json' $'\x1f''dump-screen' \ @@ -695,13 +695,13 @@ test_current_path_ignores_tilde_prefixed_banner_lines() { zellij_pane_response "$dir" 4 7 3 zellij_pane_response "$dir" 6 7 3 printf '%s\n' "🌳 Entered worktree at ~/.treehouse/scratch-e2e-project/1. Type 'exit' to return." \ - 'scratch-e2e-project HEAD' '__FM_ZELLIJ_CWD_BEGIN__' '/Users/kunchen/.treehouse/real-worktree' '__FM_ZELLIJ_CWD_END__' '❯' \ + 'scratch-e2e-project HEAD' '__FM_ZELLIJ_CWD_BEGIN__' '/home/fixture/.treehouse/real-worktree' '__FM_ZELLIJ_CWD_END__' '❯' \ > "$dir/responses/7.out" fb=$(make_zellij_fakebin "$dir") out=$( PATH="$fb:$PATH" FM_ZELLIJ_LOG="$dir/log" FM_ZELLIJ_RESPONSES="$dir/responses" \ FM_ZELLIJ_SESSION_LIST="firstmate" \ bash -c '. "$0/bin/backends/zellij.sh"; fm_backend_zellij_current_path firstmate:7' "$ROOT" ) - [ "$out" = "/Users/kunchen/.treehouse/real-worktree" ] || fail "current_path should skip the ~-prefixed banner line and read the marked cwd output, got '$out'" + [ "$out" = "/home/fixture/.treehouse/real-worktree" ] || fail "current_path should skip the ~-prefixed banner line and read the marked cwd output, got '$out'" pass "fm_backend_zellij_current_path: never picks up a ~-prefixed banner line as the answer" } @@ -796,8 +796,11 @@ test_teardown_passes_recorded_tab_id_to_zellij_kill() { printf 'report\n' > "$data/zghost/report.md" fm_write_meta "$state/zghost.meta" \ "window=firstmate:7" \ + "endpoint_task_id=zghost" \ "backend=zellij" \ + "zellij_session=firstmate" \ "zellij_tab_id=3" \ + "zellij_pane_id=7" \ "worktree=$dir/missing-worktree" \ "project=$project" \ "kind=scout" \ @@ -827,7 +830,11 @@ test_forced_secondmate_teardown_kills_zellij_children_with_child_home_tag() { printf 'smz\n' > "$home/.fm-secondmate-home" fm_write_meta "$state/smz.meta" \ "window=firstmate:99" \ + "endpoint_task_id=smz" \ "backend=zellij" \ + "zellij_session=firstmate" \ + "zellij_tab_id=99" \ + "zellij_pane_id=99" \ "worktree=$home" \ "project=$home" \ "kind=secondmate" \ @@ -835,8 +842,11 @@ test_forced_secondmate_teardown_kills_zellij_children_with_child_home_tag() { "home=$home" fm_write_meta "$home/state/childz.meta" \ "window=firstmate:7" \ + "endpoint_task_id=childz" \ "backend=zellij" \ + "zellij_session=firstmate" \ "zellij_tab_id=4" \ + "zellij_pane_id=7" \ "worktree=$dir/missing-child-worktree" \ "project=$project" \ "kind=scout" diff --git a/tests/fm-backend.test.sh b/tests/fm-backend.test.sh index 05ab1bdb436..7ac873f0878 100755 --- a/tests/fm-backend.test.sh +++ b/tests/fm-backend.test.sh @@ -12,7 +12,10 @@ # binaries and fixtures as the REFACTORED versions in this checkout, then # diffs the two command logs byte-for-byte - the report's P1 checklist # item "run current main scripts and refactored scripts against the same -# fake tools and compare command logs". +# fake tools and compare command logs". The teardown old-vs-new case also +# overlays a content-historical permissive tmux kill fixture: after the +# exact-selector change lands on the default branch, merge-base with main +# collapses to HEAD and can no longer supply that baseline. # 3. Asserts the `--backend`/`FM_BACKEND` selection refuses unknown backends # and the blocked `codex-app` backend loudly. # @@ -80,6 +83,9 @@ SH } # The commit this branch started from - the P1 "current main" baseline. +# Suitable for byte-identical old-vs-new checks while a branch still diverges +# from main. After a squash lands, merge-base(HEAD, main) collapses to HEAD, so +# callers that need a true pre-change fixture must not rely on this alone. resolve_base_ref() { local ref base for ref in main refs/heads/main origin/main refs/remotes/origin/main origin/HEAD refs/remotes/origin/HEAD; do @@ -95,6 +101,30 @@ resolve_base_ref() { BASE_REF=$(resolve_base_ref) \ || fail "fm-backend baseline requires local main or origin/main; fetch the default branch before running this test" +# Newest first-parent revision whose bin/backends/tmux.sh still uses the +# pre-exact permissive kill-window target. Content-addressed from history so the +# fixture stays historical on default-branch CI and on branches cut after the +# exact-selector change, where merge-base with main is self-referential. +resolve_permissive_tmux_kill_ref() { + local commit body + while IFS= read -r commit; do + [ -n "$commit" ] || continue + body=$(git -C "$ROOT" show "$commit:bin/backends/tmux.sh" 2>/dev/null) || continue + # shellcheck disable=SC2016 + case "$body" in + *'tmux kill-window -t "=$session:=$window"'*) continue ;; + esac + # shellcheck disable=SC2016 + case "$body" in + *'tmux kill-window -t "$1"'*|*'tmux kill-window -t "$target"'*) + printf '%s\n' "$commit" + return 0 + ;; + esac + done < <(git -C "$ROOT" log --first-parent --format='%H' HEAD -- bin/backends/tmux.sh) + return 1 +} + # --- shared: a pre-refactor bin/ shim -------------------------------------- # # build_old_bin echoes a directory whose bin/ subdir holds the PRE-REFACTOR @@ -108,10 +138,9 @@ BASE_REF=$(resolve_base_ref) \ # fm-backend.sh (and its bin/backends/ adapters) is the dispatcher every one # of the five REFACTORED scripts sources; it must be a real, reachable file in # the old bin/ too or `. "$SCRIPT_DIR/fm-backend.sh"` aborts under set -eu - -# hence it is a copied sibling, not an extracted-from-BASE_REF file: for a -# tmux-only conformance run the tmux adapter's behavior is what is under test, -# and that is unchanged by any later (e.g. non-tmux backend) addition to -# fm-backend.sh's own dispatch surface. +# hence the dispatcher is a copied sibling, while the tmux adapter is extracted +# from BASE_REF so conformance tests retain the exact historical behavior even +# when this branch changes tmux dispatch semantics. OLD_BIN_UNCHANGED_SIBLINGS="fm-gate-refuse-lib.sh fm-guard.sh fm-lock-lib.sh fm-tasks-axi-lib.sh fm-pr-lib.sh fm-tangle-lib.sh fm-tmux-lib.sh fm-composer-lib.sh fm-wake-lib.sh fm-classify-lib.sh fm-supervision-lib.sh fm-ff-lib.sh fm-config-inherit-lib.sh fm-project-mode.sh fm-harness.sh fm-crew-state.sh fm-decision-hold.sh fm-backend.sh fm-operational-input.sh" # A pull-request merge may add a new main-only dependency that the branch's older baseline does not have yet. OLD_BIN_OPTIONAL_SIBLINGS="fm-pending-reply-lib.sh" @@ -130,6 +159,7 @@ build_old_bin() { # <name> -> echoes root dir (root/bin/<script> is the entry p cp "$ROOT/bin/$f" "$bin/$f" done cp -R "$ROOT/bin/backends" "$bin/backends" + git -C "$ROOT" show "$BASE_REF:bin/backends/tmux.sh" > "$bin/backends/tmux.sh" for f in $OLD_BIN_REFACTORED; do git -C "$ROOT" show "$BASE_REF:bin/$f" > "$bin/$f" chmod +x "$bin/$f" @@ -284,7 +314,7 @@ test_backend_detect_cmux_fallback_ancestry_pid_match() { # $$ is this test script's own pid - the walk starts there. The cmux app # pid (66666) is matched via the lsappinfo bundle-id resolution, with a # deliberately non-standard install path so only the pid can match. - printf '%s\t77777\t/bin/zsh\n77777\t66666\t/usr/bin/login\n66666\t1\t/Users/x/Custom.app/Contents/MacOS/custom\n' "$$" > "$table" + printf '%s\t77777\t/bin/zsh\n77777\t66666\t/usr/bin/login\n66666\t1\t/home/x/Custom.app/Contents/MacOS/custom\n' "$$" > "$table" ( unset TMUX HERDR_ENV CMUX_WORKSPACE_ID __CFBundleIdentifier @@ -304,7 +334,7 @@ test_backend_detect_cmux_fallback_ancestry_comm_match() { # lsappinfo resolves nothing (empty output, like the real one for a # non-running or non-GUI-visible app); the bundle-shaped comm path is the # remaining match, at a non-/Applications install location on purpose. - printf '%s\t77777\t/bin/zsh\n77777\t66666\t/usr/bin/login\n66666\t1\t/Users/x/Applications/cmux.app/Contents/MacOS/cmux\n' "$$" > "$table" + printf '%s\t77777\t/bin/zsh\n77777\t66666\t/usr/bin/login\n66666\t1\t/home/x/Applications/cmux.app/Contents/MacOS/cmux\n' "$$" > "$table" ( unset TMUX HERDR_ENV CMUX_WORKSPACE_ID __CFBundleIdentifier FM_FAKE_LSAPPINFO_OUT @@ -618,9 +648,23 @@ set -u case "${1:-}" in send-keys) exit 0 ;; display-message) - for a in "$@"; do case "$a" in *cursor_y*) printf '0\n'; exit 0 ;; esac; done + for a in "$@"; do case "$a" in *cursor_y*) printf '1\n'; exit 0 ;; esac; done printf 'fakepane\n'; exit 0 ;; - capture-pane) printf '\xe2\x94\x82 \xe2\x94\x82\n'; exit 0 ;; + capture-pane) + start= end= + while [ $# -gt 0 ]; do + case "$1" in + -S) start=$2; shift 2 ;; + -E) end=$2; shift 2 ;; + *) shift ;; + esac + done + if [ "$start" = 1 ] && [ "$end" = 1 ]; then + printf '│ │\n' + else + printf '╭────╮\n│ │\n╰────╯\n' + fi + exit 0 ;; list-windows) exit 0 ;; esac exit 0 @@ -916,9 +960,20 @@ run_teardown_case() { } test_teardown_conformance_old_vs_new() { - local old_bin fb proj wt id + local old_bin fb proj wt id old_tmux_ref saved_base_ref local state_old state_new config_old config_new data log_old log_new out_old out_new rc_old rc_new + # Force the post-squash topology inside this case: merge-base with main may + # equal HEAD on default-branch CI, and that must not make the legacy kill + # fixture self-referential. build_old_bin still uses BASE_REF for entrypoints; + # only the tmux kill adapter is pinned to the content-historical permissive ref. + saved_base_ref=$BASE_REF + BASE_REF=$(git -C "$ROOT" rev-parse HEAD) + old_tmux_ref=$(resolve_permissive_tmux_kill_ref) \ + || { BASE_REF=$saved_base_ref; fail "unable to locate a historical bin/backends/tmux.sh with permissive kill-window selectors"; } old_bin=$(build_old_bin teardown-old) + git -C "$ROOT" show "$old_tmux_ref:bin/backends/tmux.sh" > "$old_bin/bin/backends/tmux.sh" \ + || { BASE_REF=$saved_base_ref; fail "could not materialize historical tmux adapter from $old_tmux_ref"; } + BASE_REF=$saved_base_ref proj="$TMP_ROOT/teardown-project"; wt="$TMP_ROOT/teardown-wt" id="teardownconform1" fm_git_worktree "$proj" "$wt" "fm/$id" @@ -948,14 +1003,21 @@ test_teardown_conformance_old_vs_new() { expect_code 0 "$rc_old" "old fm-teardown.sh (scout, report present) should succeed"$'\n'"$out_old" expect_code 0 "$rc_new" "new fm-teardown.sh (scout, report present) should succeed"$'\n'"$out_new" - diff -u "$log_old" "$log_new" > "$TMP_ROOT/teardown-diff.txt" 2>&1 \ - || fail "fm-teardown.sh: tmux+treehouse command log differs old vs new"$'\n'"$(cat "$TMP_ROOT/teardown-diff.txt")" assert_contains "$(cat "$log_new")" "treehouse"$'\x1f''return'$'\x1f''--force'$'\x1f'"$wt" \ "teardown did not call treehouse return --force <worktree>" - assert_contains "$(cat "$log_new")" "tmux"$'\x1f''kill-window'$'\x1f''-t'$'\x1f'"firstmate:fm-$id" \ - "teardown did not call tmux kill-window -t <window>" - - pass "fm-teardown.sh: treehouse return + tmux kill-window command log is byte-identical old vs new for a scout task" + # The legacy fixture's adapter comes from BASE_REF, so its selector form is + # whatever the merge-base carried: permissive while the exact-selector change + # was still on a branch, exact for every branch cut after it landed on main. + # Pinning the old form here would make this case pass once and then fail + # forever, so the '=' exactness markers are normalized away and the legacy run + # is only required to have reached tmux window cleanup for this task. The + # exact-selector contract belongs to the current script, asserted below. + assert_contains "$(tr -d '=' < "$log_old")" "tmux"$'\x1f''kill-window'$'\x1f''-t'$'\x1f'"firstmate:fm-$id" \ + "legacy teardown fixture did not exercise tmux window cleanup for the task" + assert_contains "$(cat "$log_new")" "tmux"$'\x1f''kill-window'$'\x1f''-t'$'\x1f'"=firstmate:=fm-$id" \ + "teardown did not call tmux kill-window with exact session and window selectors" + + pass "fm-teardown.sh: treehouse return remains compatible while tmux cleanup uses exact selectors" } # --- backend selection loudly refuses an unknown backend -------------------- diff --git a/tests/fm-bearings-snapshot.test.sh b/tests/fm-bearings-snapshot.test.sh index f8afefa2551..31c27a677ae 100755 --- a/tests/fm-bearings-snapshot.test.sh +++ b/tests/fm-bearings-snapshot.test.sh @@ -1861,35 +1861,6 @@ EOF pass "main and secondmate captain actionability use the same blocker readiness" } -# The /bearings skill is the one owner of the four-section chat-response contract. -# Assert it states exactly the four fixed sections in order, each with its explicit -# empty-state sentence, documents the At Anchor exclusion, and mandates a chat that is -# materially shorter than and links to the report file. -test_chat_contract_four_sections() { - local skill body headings report_headings expected - skill="$ROOT/.agents/skills/bearings/SKILL.md" - [ -f "$skill" ] || fail "bearings SKILL.md missing at $skill" - body=$(awk '/^## Chat-response contract$/{capture=1; next} capture && /^## /{exit} capture' "$skill") - headings=$(printf '%s\n' "$body" | sed -nE "s/^[0-9]+\. \*\*([^*]+)\*\*.*/\1/p") - expected=$(printf '%s\n' "Captain's Call" "Recently Landed" "Underway" "Charted Next") - [ "$headings" = "$expected" ] || fail "chat contract must contain exactly four numbered sections in fixed order, got: $headings" - assert_contains "$body" "Nothing needs your action right now" "Captain's Call empty-state sentence" - assert_contains "$body" "No recent completions are in the current baseline" "Recently Landed empty-state sentence" - assert_contains "$body" "Nothing is underway" "Underway empty-state sentence" - assert_contains "$body" "Nothing is queued" "Charted Next empty-state sentence" - report_headings=$(sed -nE 's/^ - \*\*(Captain.s Call|Recently Landed|Underway|Charted Next)\*\*.*/\1/p' "$skill") - [ "$report_headings" = "$expected" ] || fail "detailed report contract must contain the same four complete sections, got: $report_headings" - grep -Eq 'since the (prior|last) report|Nothing has landed since|unchanged delta' "$skill" \ - && fail "bearings contract still contains prior-report delta wording" - # shellcheck disable=SC2016 # Backticks are literal Markdown in the expected text. - assert_contains "$(cat "$skill")" 'Never read an earlier `data/status-report-*.md`' "prior reports must not influence current output" - assert_contains "$(cat "$skill")" "bounded current recent-completions baseline" "Recently Landed must be a current baseline" - assert_contains "$body" "no At Anchor section" "the At Anchor exclusion must be documented" - assert_contains "$body" "materially shorter" "the chat must be materially shorter than the report file" - assert_contains "$body" "links to" "the chat must link to the report file" - pass "the /bearings skill states the four-section chat contract in order, with empty-states and the At Anchor exclusion" -} - test_domain_alpha_stale_parent_event_does_not_become_current_work test_gnu_stat_uses_file_formats_without_bsd_fallback_pollution test_parent_activity_evidence_is_bounded_and_disclosed @@ -1920,7 +1891,6 @@ test_main_unstructured_current_is_disclosed_with_structured_sibling test_main_orphan_counterfactual_meta_clears_inventory_warning test_mixed_secondmate_roles_partial_state_and_captain_readiness test_main_captain_readiness_matches_secondmate_projection -test_chat_contract_four_sections test_completed_scout_report_not_pending test_open_decision_surfaces_end_to_end test_report_pointers_surface diff --git a/tests/fm-bootstrap.test.sh b/tests/fm-bootstrap.test.sh index dbcd8ace458..f4e5739fe79 100755 --- a/tests/fm-bootstrap.test.sh +++ b/tests/fm-bootstrap.test.sh @@ -655,6 +655,7 @@ make_routine_bootstrap_fixture() { printf '%s\n' '.fm-secondmate-home' printf '%s\n' 'config/crew-harness' printf '%s\n' 'config/crew-dispatch.json' + printf '%s\n' 'config/startup-memory-budget' } > "$root/.gitignore" printf '%s\n' 'instructions' > "$root/AGENTS.md" mkdir -p "$root/bin" "$root/.agents/skills" @@ -676,7 +677,13 @@ make_routine_bootstrap_fixture() { cat > "$fakebin/tmux" <<'SH' #!/usr/bin/env bash case "${1:-}" in - display-message) printf '%s\n' codex ;; + display-message) + case "$*" in + *'#{cursor_y}'*) printf '%s\n' 0 ;; + *) printf '%s\n' codex ;; + esac + ;; + capture-pane) printf '\n' ;; list-windows) printf '%s\n' fm-sm ;; esac exit 0 @@ -712,19 +719,6 @@ test_routine_bootstrap_contract_runs_under_system_bash() { pass "bootstrap routine contract runs under system /bin/bash" } -test_bootstrap_info_is_no_load_and_actionable_lines_trigger() { - local trigger - # shellcheck disable=SC2016 # The backtick-delimited skill names are literal Markdown. - trigger=$(sed -n '/- `bootstrap-diagnostics`/,/- `diagnostic-reasoning`/p' "$ROOT/AGENTS.md") - assert_contains "$trigger" "actionable diagnostic line" "bootstrap-diagnostics trigger should be action-scoped" - assert_contains "$trigger" "BOOTSTRAP_INFO:" "bootstrap-diagnostics trigger should classify BOOTSTRAP_INFO as no-load" - assert_not_contains "$trigger" "TASKS_AXI:" "tasks-axi availability must not trigger diagnostics loading" - assert_not_contains "$trigger" "CREW_HARNESS_OVERRIDE:" "harness override confirmation must not trigger diagnostics loading" - assert_not_contains "$trigger" "CREW_DISPATCH: active" "active dispatch confirmation must not trigger diagnostics loading" - assert_not_contains "$trigger" "already-live" "already-live secondmate liveness must not trigger diagnostics loading" - pass "bootstrap diagnostics trigger excludes benign lines and keeps actionable prefixes" -} - test_crew_dispatch_active_rules_are_verbose_bootstrap_info() { local case_dir fakebin out expect case_dir="$TMP_ROOT/dispatch-active" @@ -775,7 +769,10 @@ unsupported codex max effort is flagged^{"rules":[{"when":"big feature","use":{" unsupported grok max effort is flagged^{"rules":[{"when":"deep current work","use":{"harness":"grok","model":"grok-4","effort":"max"}}]}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - invalid effort: grok:max unsupported grok xhigh effort is flagged^{"rules":[{"when":"deep current work","use":{"harness":"grok","model":"grok-4","effort":"xhigh"}}]}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - invalid effort: grok:xhigh pi max effort is accepted^{"rules":[{"when":"deep coding","use":{"harness":"pi","model":"openai-codex/gpt-5.6-sol","effort":"max"}}]}^empty^ +pi-signed max effort is accepted^{"rules":[{"when":"signed coding","use":{"harness":"pi-signed","model":"openai-codex/gpt-5.6-sol","effort":"max"}}]}^empty^ unsupported opencode effort is flagged^{"rules":[{"when":"opencode work","use":{"harness":"opencode","model":"anthropic/claude-sonnet-4-5","effort":"high"}}]}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - invalid effort: opencode:high +kimi model profile is accepted^{"rules":[{"when":"kimi work","use":{"harness":"kimi","model":"kimi-code/k3"}}]}^empty^ +unsupported kimi effort is flagged^{"rules":[{"when":"kimi work","use":{"harness":"kimi","model":"kimi-code/k3","effort":"high"}}]}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - invalid effort: kimi:high array use with quota-balanced is accepted^{"rules":[{"when":"big feature","use":[{"harness":"claude","model":"claude-sonnet-5","effort":"high"},{"harness":"codex","model":"gpt-5.5","effort":"high"}],"select":"quota-balanced"}]}^empty^ array use without select is accepted^{"rules":[{"when":"big feature","use":[{"harness":"claude"},{"harness":"codex"}]}]}^empty^ one-element array use is accepted^{"rules":[{"when":"focused feature","use":[{"harness":"claude"}]}]}^empty^ @@ -812,6 +809,5 @@ test_fleet_sync_timeout_empty_override_uses_default test_fleet_sync_timeout_is_computed_before_launch test_routine_bootstrap_confirmations_are_silent test_routine_bootstrap_contract_runs_under_system_bash -test_bootstrap_info_is_no_load_and_actionable_lines_trigger test_crew_dispatch_active_rules_are_verbose_bootstrap_info test_crew_dispatch_validation diff --git a/tests/fm-brief.test.sh b/tests/fm-brief.test.sh index 74690eab423..0199311824b 100755 --- a/tests/fm-brief.test.sh +++ b/tests/fm-brief.test.sh @@ -1,14 +1,18 @@ #!/usr/bin/env bash # Behavior tests for bin/fm-brief.sh. # -# Regression coverage for the heredoc-in-command-substitution parse bug (issue -# #166): each ship-mode branch builds its Definition-of-done text with -# `VAR=$(cat <<EOF ... EOF)`. Bash's lexer tracks quote state through the -# heredoc body while it scans for the matching `)` of the command -# substitution, so a single unescaped apostrophe anywhere in that body breaks -# parsing of the *entire rest of the script* - `bash -n` fails, not just the -# generated brief. A plain `cat > file <<EOF ... EOF` (not wrapped in `$(...)`) -# is unaffected, so the secondmate charter block does not need this guard. +# Regression coverage for the heredoc-in-command-substitution parse bug (issues +# #166, #958, #1069). Building a variable with `VAR=$(cat <<EOF ... EOF)` is +# unsafe on Bash 3.2 (macOS /bin/bash): the lexer scans for the matching `)` of +# the command substitution textually and tracks quote state through the heredoc +# body, so a single apostrophe, unbalanced quote, or unbalanced paren anywhere +# in that body breaks parsing of the *entire rest of the script* - `bash -n` +# fails, not just the generated brief. The DOD and Herdr-section builders now +# use `IFS= read -r -d '' VAR <<EOF || true` instead, which removes the `$(...)` +# wrapper and eliminates the whole defect class regardless of future prose. +# test_no_heredoc_in_command_substitution guards that structure directly. +# Ambient `bash -n` here is Bash 5 and cannot see the bug, so the real +# cross-version enforcement lives in the macos-stock-bash CI job. set -u # shellcheck source=tests/lib.sh @@ -18,9 +22,10 @@ TMP_ROOT=$(fm_test_tmproot fm-brief) BRIEF_HOME="$TMP_ROOT/home" mkdir -p "$BRIEF_HOME/data" -# The script itself must always parse. This is the direct regression test for -# issue #166: a stray apostrophe in any of the three DOD heredoc bodies -# (no-mistakes/direct-PR/local-only) breaks `bash -n` on the whole file. +# The script itself must always parse under the ambient bash. That is Bash 5 in +# CI and locally, where the issue #958/#1069 parser bug does not fire, so this +# is a weak guard on its own; test_no_heredoc_in_command_substitution and the +# macos-stock-bash CI job carry the real cross-version enforcement. test_script_parses() { local out rc out=$(bash -n "$ROOT/bin/fm-brief.sh" 2>&1); rc=$? @@ -29,6 +34,142 @@ test_script_parses() { pass "fm-brief.sh: bash -n succeeds" } +# Structural class guard (issues #166, #958, #1069): never build a variable by +# wrapping a heredoc in a command substitution (`VAR=$(cat <<EOF ... EOF)`). +# That construct is what breaks Bash 3.2 parsing, and pinning one historical +# apostrophe phrase (as the old test did) missed the #945 reintroduction. This +# guards the *shape* directly against the whole file, so any future DOD or +# section builder that reintroduces the class fails here regardless of prose. +test_no_heredoc_in_command_substitution() { + local unsafe safe + unsafe="$TMP_ROOT/heredoc-in-substitution.sh" + safe="$TMP_ROOT/plain-heredoc.sh" + # shellcheck disable=SC2016 # Literal shell fixtures must remain unexpanded. + printf '%s\n' 'value=$(' ' cat <<EOF' 'body' 'EOF' ')' > "$unsafe" + # shellcheck disable=SC2016 # Literal shell fixtures must remain unexpanded. + printf '%s\n' 'cat <<EOF' '$(' ' cat <<INNER' 'INNER' ')' 'EOF' > "$safe" + if no_heredoc_in_command_substitution "$unsafe"; then + fail "structural guard accepted a multiline heredoc nested in a command substitution" + fi + no_heredoc_in_command_substitution "$safe" \ + || fail "structural guard treated heredoc body prose as shell structure" + no_heredoc_in_command_substitution "$ROOT/bin/fm-brief.sh" \ + || fail "fm-brief.sh wraps a heredoc in a command substitution (breaks Bash 3.2 parsing)" + pass "fm-brief.sh: no heredoc is nested inside a command substitution (Bash 3.2 parse-safe)" +} + +no_heredoc_in_command_substitution() { + perl - "$1" <<'PERL' +use strict; +use warnings; + +my $path = shift; +open my $source, '<', $path or die "$path: $!\n"; +my @frames; +my @heredocs; +my $quote = ''; +my $line_number = 0; + +while (my $line = <$source>) { + $line_number++; + if (@heredocs) { + my $candidate = $line; + $candidate =~ s/\r?\n\z//; + $candidate =~ s/^\t+// if $heredocs[0]{strip_tabs}; + shift @heredocs if $candidate eq $heredocs[0]{delimiter}; + next; + } + + my $length = length $line; + for (my $i = 0; $i < $length; $i++) { + my $char = substr($line, $i, 1); + if ($quote eq "'") { + $quote = '' if $char eq "'"; + next; + } + if ($char eq '\\') { + $i++; + next; + } + if ($quote eq '"' && $char eq '"') { + $quote = ''; + next; + } + if ($char eq "'" && $quote eq '') { + $quote = "'"; + next; + } + if ($char eq '"' && $quote eq '') { + $quote = '"'; + next; + } + if ($char eq '#' && $quote eq '' && ($i == 0 || substr($line, $i - 1, 1) =~ /[\s;|&()]/)) { + last; + } + if ($char eq '$' && substr($line, $i + 1, 1) eq '(') { + push @frames, { depth => 1, quote => $quote }; + $quote = ''; + $i++; + next; + } + if (@frames && $quote eq '' && $char eq '(') { + $frames[-1]{depth}++; + next; + } + if (@frames && $quote eq '' && $char eq ')') { + $frames[-1]{depth}--; + if ($frames[-1]{depth} == 0) { + my $frame = pop @frames; + $quote = $frame->{quote}; + } + next; + } + next unless $quote eq '' && $char eq '<' && substr($line, $i + 1, 1) eq '<'; + if (@frames) { + print STDERR "$path:$line_number\n"; + exit 1; + } + + my $j = $i + 2; + my $strip_tabs = substr($line, $j, 1) eq '-'; + $j++ if $strip_tabs; + $j++ while substr($line, $j, 1) =~ /[ \t]/; + my $delimiter = ''; + my $delimiter_quote = ''; + for (; $j < $length; $j++) { + my $token = substr($line, $j, 1); + if ($delimiter_quote) { + if ($token eq $delimiter_quote) { + $delimiter_quote = ''; + } elsif ($token eq '\\' && $delimiter_quote eq '"') { + $j++; + $delimiter .= substr($line, $j, 1); + } else { + $delimiter .= $token; + } + next; + } + if ($token eq "'" || $token eq '"') { + $delimiter_quote = $token; + next; + } + if ($token eq '\\') { + $j++; + $delimiter .= substr($line, $j, 1); + next; + } + last if $token =~ /[\s;|&()<>]/; + $delimiter .= $token; + } + push @heredocs, { delimiter => $delimiter, strip_tabs => $strip_tabs }; + $i = $j - 1; + } +} + +exit 0; +PERL +} + test_help_includes_entire_header() { local help help=$("$ROOT/bin/fm-brief.sh" --help) @@ -114,9 +255,13 @@ test_no_mistakes_dod_wording() { # shellcheck disable=SC2016 # single quotes are deliberate: the backticks must stay literal assert_grep '`help`' "$brief" \ "no-mistakes DOD must render literal backticks around help" - assert_no_grep "no-mistakes' own guidance" "$brief" \ - "no-mistakes DOD regressed to the apostrophe form that breaks bash -n" - pass "fm-brief.sh: no-mistakes DOD wording avoids the apostrophe regression" + # The apostrophe in "firstmate's authority check" is now structurally safe + # (no `$(...)` wrapper around the heredoc), so it renders verbatim instead of + # being reworded or escaped away. test_no_heredoc_in_command_substitution + # guards the structure that makes it safe. + assert_grep "firstmate's authority check" "$brief" \ + "no-mistakes DOD lost the apostrophe prose that the structural fix makes parse-safe" + pass "fm-brief.sh: no-mistakes DOD keeps its apostrophe prose, now parse-safe" } test_ship_project_memory_wording() { @@ -291,6 +436,97 @@ test_secondmate_marked_request_reporting_contract() { pass "fm-brief.sh: marked requests avoid generic acknowledgements and preserve material reporting" } +test_secondmate_directory_paths_are_absolute_and_output_is_stable() { + local root home data_override state_override brief baseline err status + root="$TMP_ROOT/relative-directory-inputs" + mkdir -p "$root" + root=$(cd "$root" && pwd -P) + home="$root/home" + data_override="$root/data-override" + state_override="$root/state-override" + mkdir -p "$home/data" "$home/state" "$data_override" "$state_override" \ + "$root/cdpath/home/data" "$root/cdpath/home/state" \ + "$root/cdpath/data-override" "$root/cdpath/state-override" + + brief="$home/data/relative-home/brief.md" + FM_HOME="$home" FM_SECONDMATE_CHARTER=x \ + "$ROOT/bin/fm-brief.sh" relative-home --secondmate --no-projects >/dev/null 2>&1 + baseline="$root/absolute-home-charter" + cp "$brief" "$baseline" + rm -f "$brief" + ( + cd "$root" || exit 1 + CDPATH="$root/cdpath" FM_HOME=home FM_SECONDMATE_CHARTER=x \ + "$ROOT/bin/fm-brief.sh" relative-home --secondmate --no-projects >/dev/null 2>&1 + ) + cmp -s "$baseline" "$brief" \ + || fail "relative FM_HOME changed charter bytes compared with the same absolute home" + assert_grep ">> '$home/state/relative-home.status'" "$brief" \ + "relative FM_HOME did not render an absolute secondmate status path" + + brief="$home/data/relative-state/brief.md" + FM_HOME="$home" FM_STATE_OVERRIDE="$state_override" FM_SECONDMATE_CHARTER=x \ + "$ROOT/bin/fm-brief.sh" relative-state --secondmate --no-projects >/dev/null 2>&1 + baseline="$root/absolute-state-charter" + cp "$brief" "$baseline" + rm -f "$brief" + ( + cd "$root" || exit 1 + CDPATH="$root/cdpath" FM_HOME="$home" FM_STATE_OVERRIDE=state-override FM_SECONDMATE_CHARTER=x \ + "$ROOT/bin/fm-brief.sh" relative-state --secondmate --no-projects >/dev/null 2>&1 + ) + cmp -s "$baseline" "$brief" \ + || fail "relative FM_STATE_OVERRIDE changed charter bytes compared with the same absolute state directory" + assert_grep ">> '$state_override/relative-state.status'" "$brief" \ + "relative FM_STATE_OVERRIDE did not render an absolute secondmate status path" + + brief="$data_override/relative-data/brief.md" + FM_HOME="$home" FM_DATA_OVERRIDE="$data_override" FM_SECONDMATE_CHARTER=x \ + "$ROOT/bin/fm-brief.sh" relative-data --secondmate --no-projects >/dev/null 2>&1 + baseline="$root/absolute-data-charter" + cp "$brief" "$baseline" + rm -f "$brief" + ( + cd "$root" || exit 1 + CDPATH="$root/cdpath" FM_HOME="$home" FM_DATA_OVERRIDE=data-override FM_SECONDMATE_CHARTER=x \ + "$ROOT/bin/fm-brief.sh" relative-data --secondmate --no-projects >/dev/null 2>&1 + ) + cmp -s "$baseline" "$brief" \ + || fail "relative FM_DATA_OVERRIDE changed charter bytes compared with the same absolute data directory" + assert_grep ">> '$home/state/relative-data.status'" "$brief" \ + "relative FM_DATA_OVERRIDE changed the absolute default status path" + + err="$root/unresolved.err" + ( + cd "$root" || exit 1 + FM_HOME=missing-home FM_SECONDMATE_CHARTER=x \ + "$ROOT/bin/fm-brief.sh" unresolved-home --secondmate --no-projects >/dev/null 2>"$err" + ); status=$? + expect_code 1 "$status" "an unresolved relative FM_HOME must fail" + assert_grep "FM_HOME directory cannot be resolved: missing-home" "$err" \ + "unresolved relative FM_HOME did not fail loudly" + + ( + cd "$root" || exit 1 + FM_HOME="$home" FM_STATE_OVERRIDE=missing-state FM_SECONDMATE_CHARTER=x \ + "$ROOT/bin/fm-brief.sh" unresolved-state --secondmate --no-projects >/dev/null 2>"$err" + ); status=$? + expect_code 1 "$status" "an unresolved relative FM_STATE_OVERRIDE must fail" + assert_grep "FM_STATE_OVERRIDE directory cannot be resolved: missing-state" "$err" \ + "unresolved relative FM_STATE_OVERRIDE did not fail loudly" + + ( + cd "$root" || exit 1 + FM_HOME="$home" FM_DATA_OVERRIDE=missing-data FM_SECONDMATE_CHARTER=x \ + "$ROOT/bin/fm-brief.sh" unresolved-data --secondmate --no-projects >/dev/null 2>"$err" + ); status=$? + expect_code 1 "$status" "an unresolved relative FM_DATA_OVERRIDE must fail" + assert_grep "FM_DATA_OVERRIDE directory cannot be resolved: missing-data" "$err" \ + "unresolved relative FM_DATA_OVERRIDE did not fail loudly" + + pass "fm-brief.sh: relative directory inputs ignore CDPATH, render stable absolute charter paths, or fail loudly" +} + test_herdr_lab_contract_applies_to_scouts_but_not_secondmates() { local home brief status=0 home="$TMP_ROOT/herdr-kind-home" @@ -383,6 +619,7 @@ test_scout_and_secondmate_scaffold() { } test_script_parses +test_no_heredoc_in_command_substitution test_help_includes_entire_header test_ship_modes_generate_clean_briefs test_faster_paths_use_configured_authority_without_stacked_review @@ -394,6 +631,7 @@ test_herdr_lab_omission_is_loud_for_ship_and_scout test_herdr_lab_contract_applies_to_scouts_but_not_secondmates test_secondmate_no_projects_charter test_secondmate_marked_request_reporting_contract +test_secondmate_directory_paths_are_absolute_and_output_is_stable test_pause_verb_override_renders_all_brief_scaffolds test_scout_and_secondmate_load_decision_hold_policy test_scout_and_secondmate_scaffold diff --git a/tests/fm-calm-pi-extension.test.sh b/tests/fm-calm-pi-extension.test.sh index c0455a5de42..a8e09be1456 100755 --- a/tests/fm-calm-pi-extension.test.sh +++ b/tests/fm-calm-pi-extension.test.sh @@ -16,6 +16,16 @@ PI_OPERATIONAL_INPUT="$ROOT/.pi/extensions/lib/fm-operational-input.ts" PI_PACKAGE_DIR=${FM_PI_PACKAGE_DIR:-"$(npm root -g 2>/dev/null)/@earendil-works/pi-coding-agent"} TMUX_SOCKET="fm-calm-$$" TMUX_SESSION="fm-calm-e2e" +# Verified against Pi 0.81.1 and 0.82.0 (docs/calm-mode-feasibility.md). This is +# known-good evidence, not a support ceiling: the fixtures below run against whatever +# Pi is actually installed, and record_pi_version_evidence never rejects a newer +# version. The tracked presentation adapters probe the exact API they patch (see +# .pi/extensions/fm-calm.ts) instead of relying on version inference, so a version +# string is evidence for the record, not a gate. +record_pi_version_evidence() { + local version=$1 context=$2 + [ -n "$version" ] || fail "$context could not determine the installed Pi version" +} cleanup() { if command -v tmux >/dev/null 2>&1; then @@ -57,64 +67,6 @@ find_chrome() { return 1 } -test_static_contract() { - local text assistant_layout operational_user_layout visibility watch operational - assert_present "$EXT" "tracked Pi calm extension is missing" - assert_present "$ASSISTANT_LAYOUT" "tracked Pi Calm assistant-layout adapter is missing" - assert_present "$OPERATIONAL_USER_LAYOUT" "tracked Pi Calm operational-user layout adapter is missing" - assert_present "$VISIBILITY" "tracked Pi calm visibility policy is missing" - text=$(cat "$EXT") - assistant_layout=$(cat "$ASSISTANT_LAYOUT") - operational_user_layout=$(cat "$OPERATIONAL_USER_LAYOUT") - visibility=$(cat "$VISIBILITY") - watch=$(cat "$WATCH_EXT") - operational=$(cat "$PI_OPERATIONAL_INPUT") - assert_contains "$text" 'pi.registerCommand("calm"' "Pi calm extension does not register /calm" - assert_contains "$text" 'pi.on("session_start"' "Pi calm extension does not restore presentation on every session start" - assert_contains "$text" 'loadCalmPreference()' "Pi calm extension does not restore the home-persistent toggle choice" - assert_contains "$text" 'persistCalmPreference(active)' "Pi calm extension does not persist the captain's toggle choice" - assert_not_contains "$text" 'setCalmPresentation(false)' "Pi calm extension still resets the toggle on session start" - assert_contains "$text" 'ctx.ui.setToolsExpanded(!expanded)' "Pi calm extension does not redraw existing custom entries" - assert_contains "$text" 'ctx.ui.setToolsExpanded(expanded)' "Pi calm extension does not restore Ctrl+O state after redraw" - assert_not_contains "$text" 'ctx.navigateTree' "Pi calm extension reconstructs the transcript and drops transient diagnostics" - assert_not_contains "$visibility" 'deliverFirstmateSyntheticInput' "Pi calm visibility policy can still replace operational input semantics" - assert_not_contains "$visibility" 'classifyFirstmateSyntheticInput' "Pi calm visibility policy still classifies operational input for interception" - assert_contains "$text" 'ctx.ui.setWorkingVisible(true)' "Pi calm extension does not preserve Pi's live working row" - assert_not_contains "$text" 'ctx.ui.setWorkingVisible(!active)' "Pi calm extension still hides Pi's live working row" - assert_contains "$text" 'ctx.ui.setHiddenThinkingLabel(active ? "" : undefined)' "Pi calm extension does not hide collapsed thinking labels" - assert_contains "$text" 'installCalmAssistantLayout()' "Pi Calm extension does not install its zero-height assistant layout" - assert_contains "$text" 'installCalmOperationalUserLayout()' "Pi Calm extension does not install its operational-user layout" - assert_contains "$assistant_layout" 'AssistantMessageComponent.prototype.updateContent' "Pi Calm assistant layout does not control the exported component presentation path" - assert_contains "$assistant_layout" 'block.type !== "thinking"' "Pi Calm assistant layout does not remove thinking from its presentation copy" - assert_contains "$operational_user_layout" 'InteractiveMode.prototype' "Pi Calm operational-user layout does not control the transcript owner" - assert_contains "$operational_user_layout" 'classifyFirstmateCurrentOperationalText(text)' "Pi Calm operational-user layout bypasses canonical current classification" - assert_contains "$operational_user_layout" 'text.includes("\u2063")' "Pi Calm operational-user layout spawns its classifier for ordinary captain rows" - assert_contains "$operational_user_layout" '"\u2063Supervisor escalate ("' "Pi Calm operational-user layout lost the narrow legacy marker" - assert_contains "$operational_user_layout" 'hidesOperationalInput()' "Pi Calm operational-user row does not use presentation-only hiding" - assert_not_contains "$operational_user_layout" 'FIRSTMATE_OP: ' "Pi Calm operational-user layout duplicates the canonical marker grammar" - assert_not_contains "$text" 'calm transcript' "Pi calm extension still adds a persistent Calm status row" - assert_not_contains "$text" 'pi.on("input"' "Pi calm extension still intercepts semantic input" - assert_not_contains "$text" 'sendMessage' "Pi calm extension still replaces user-role input with custom context" - assert_contains "$text" 'ctx.ui.onTerminalInput' "Pi calm extension does not scope export rendering to terminal submissions" - assert_contains "$text" 'getKeybindings().matches(data, "tui.input.submit")' "Pi calm export boundary ignores the active submit keybinding" - assert_contains "$text" 'input !== "/share"' "Pi calm export boundary does not cover /share" - assert_not_contains "$text" 'FIRSTMATE_PI_LAUNCH_BRIEF_ENV' "Pi calm presentation still depends on launch-input provenance" - assert_contains "$text" 'renderShell: "self"' "Pi calm extension cannot remove complete built-in tool shells" - assert_contains "$visibility" 'CALM_VISIBLE_CLASSES' "Pi calm policy does not centralize its visibility allowlist" - assert_contains "$operational" 'fm-operational-input.sh' "Pi adapter does not delegate to the canonical cross-language owner" - assert_not_contains "$visibility" 'FIRSTMATE WATCHER WAKE:' "current Calm classification still matches watcher payload prose" - assert_not_contains "$visibility" 'TURN WOULD END BLIND' "current Calm classification still matches turn-end payload prose" - # shellcheck disable=SC2016 # Backticks are literal prompt markup. - assert_not_contains "$visibility" 'Run `bin/fm-session-start.sh`' "current Calm classification still matches session-start payload prose" - assert_not_contains "$visibility" 'FIRSTMATE_OP: ' "current Calm classification duplicates the canonical marker grammar" - assert_contains "$watch" 'calmHides("assistant-tool-call")' "Firstmate watcher tool does not participate in Calm presentation" - assert_contains "$watch" 'renderShell: "self"' "Firstmate watcher tool cannot remove its complete shell" - for name in Read Bash Edit Write Grep Find Ls; do - assert_contains "$text" "create${name}ToolDefinition" "Pi calm extension does not wrap the $name built-in" - done - pass "Pi calm extension is presentation-only with one persisted visibility choice, no Calm status row, native working visibility, supported redraw controls, and the Firstmate watcher-tool integration" -} - test_home_resolution() { local fixture out status version if ! command -v node >/dev/null 2>&1 || ! command -v npm >/dev/null 2>&1; then @@ -126,7 +78,7 @@ test_home_resolution() { return 0 fi version=$(node -p "require('$PI_PACKAGE_DIR/package.json').version") - [ "$version" = "0.81.1" ] || fail "Pi calm compatibility assumptions require Pi 0.81.1, found $version" + record_pi_version_evidence "$version" "Pi calm compatibility assumptions" fixture="$TMP_ROOT/home-resolution" mkdir -p \ @@ -224,6 +176,173 @@ JS pass "Pi calm resolves its persistent home independently of Pi's launch directory" } +test_pi_compat_no_upper_bound() { + local version + for version in 0.83.0 0.90.0 1.0.0 2.3.4 0.82.1 10.20.30; do + record_pi_version_evidence "$version" "synthetic newer Pi" \ + || fail "record_pi_version_evidence rejected Pi $version solely for being newer than 0.82.0" + done + if (record_pi_version_evidence "" "malformed Pi version probe") 2>/dev/null; then + fail "record_pi_version_evidence accepted a missing/malformed Pi version" + fi + pass "Pi calm compatibility evidence never rejects a Pi version for being newer than 0.82.0, and still fails closed on a missing or malformed version" +} + +test_pi_compat_degraded_adapter() { + local fixture out status + if ! command -v node >/dev/null 2>&1 || ! command -v npm >/dev/null 2>&1; then + echo "skip: node or npm not found for Pi calm degraded-adapter test" + return 0 + fi + if [ ! -f "$PI_PACKAGE_DIR/package.json" ]; then + echo "skip: installed @earendil-works/pi-coding-agent package not found" + return 0 + fi + + fixture="$TMP_ROOT/degraded-adapter" + mkdir -p \ + "$fixture/project/.pi/extensions/lib" \ + "$fixture/project/node_modules/@earendil-works" + cp "$EXT" "$fixture/project/.pi/extensions/fm-calm.ts" + cp "$ASSISTANT_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-assistant-layout.ts" + cp "$OPERATIONAL_USER_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" + cp "$VISIBILITY" "$fixture/project/.pi/extensions/lib/fm-calm-visibility.ts" + cp "$PI_OPERATIONAL_INPUT" "$fixture/project/.pi/extensions/lib/fm-operational-input.ts" + ln -s "$PI_PACKAGE_DIR" "$fixture/project/node_modules/@earendil-works/pi-coding-agent" + ln -s "$PI_PACKAGE_DIR/node_modules/@earendil-works/pi-tui" "$fixture/project/node_modules/@earendil-works/pi-tui" + ln -s "$PI_PACKAGE_DIR/node_modules/typebox" "$fixture/project/node_modules/typebox" + printf '%s\n' '{"type":"module"}' >"$fixture/project/package.json" + + out=$(cd "$fixture/project" && \ + EXT="$fixture/project/.pi/extensions/fm-calm.ts" \ + PI_PACKAGE_DIR="$PI_PACKAGE_DIR" \ + node --input-type=module 2>&1 <<'JS' +import { pathToFileURL } from "node:url"; + +const packageRoot = process.env.PI_PACKAGE_DIR; +const { AssistantMessageComponent } = await import( + pathToFileURL(`${packageRoot}/dist/modes/interactive/components/assistant-message.js`).href +); +const originalUpdateContent = AssistantMessageComponent.prototype.updateContent; +if (typeof originalUpdateContent !== "function") { + throw new Error( + "fixture precondition failed: installed Pi lacks AssistantMessageComponent.prototype.updateContent", + ); +} +delete AssistantMessageComponent.prototype.updateContent; + +const diagnostics = []; +const originalConsoleError = console.error; +console.error = (...args) => diagnostics.push(args.join(" ")); + +let calmCommand; +const handlers = new Map(); +const pi = { + events: { emit() {}, on() {} }, + on(event, handler) { + handlers.set(event, handler); + }, + registerCommand(name, command) { + if (name === "calm") calmCommand = command; + }, + registerEntryRenderer() {}, + registerTool() {}, +}; + +let threw = false; +try { + const extension = await import(`${pathToFileURL(process.env.EXT).href}?degraded=${Date.now()}`); + extension.default(pi); +} catch { + threw = true; +} +console.error = originalConsoleError; + +if (threw) { + throw new Error( + "a missing presentation API crashed the whole Calm extension instead of degrading just that adapter", + ); +} +if (!calmCommand || !handlers.has("session_start")) { + throw new Error( + "Calm command/session lifecycle did not register when only one presentation adapter was unavailable", + ); +} +if (typeof AssistantMessageComponent.prototype.updateContent !== "undefined") { + throw new Error( + "the degraded adapter path patched updateContent anyway despite the missing API, which would claim false success", + ); +} +const sawClearSkipReason = diagnostics.some( + (line) => line.includes("collapsed-thinking") && /unavailable|skip/i.test(line), +); +if (!sawClearSkipReason) { + throw new Error( + `missing a clear skip reason for the degraded collapsed-thinking adapter; saw: ${JSON.stringify(diagnostics)}`, + ); +} + +AssistantMessageComponent.prototype.updateContent = originalUpdateContent; +JS +) + status=$? + [ "$status" -eq 0 ] || fail "Pi calm degraded-adapter path failed: $out" + [ -z "$out" ] || fail "Pi calm degraded-adapter test printed output: $out" + pass "a missing collapsed-thinking presentation API degrades only that Calm adapter with a clear skip reason, while the rest of Calm still registers" +} + +test_pi_compat_missing_adapter_exports() { + local fixture out status + if ! command -v node >/dev/null 2>&1; then + echo "skip: node not found for Pi calm missing-adapter-export test" + return 0 + fi + + fixture="$TMP_ROOT/missing-adapter-exports" + mkdir -p \ + "$fixture/project/.pi/extensions/lib" \ + "$fixture/project/node_modules/@earendil-works/pi-coding-agent" + cp "$ASSISTANT_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-assistant-layout.ts" + cp "$OPERATIONAL_USER_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" + cp "$VISIBILITY" "$fixture/project/.pi/extensions/lib/fm-calm-visibility.ts" + cp "$PI_OPERATIONAL_INPUT" "$fixture/project/.pi/extensions/lib/fm-operational-input.ts" + printf '%s\n' '{"type":"module"}' >"$fixture/project/package.json" + printf '%s\n' \ + '{"name":"@earendil-works/pi-coding-agent","type":"module","exports":"./index.js"}' \ + >"$fixture/project/node_modules/@earendil-works/pi-coding-agent/package.json" + printf '%s\n' \ + 'export function getMarkdownTheme() { return {}; }' \ + 'export class UserMessageComponent {}' \ + >"$fixture/project/node_modules/@earendil-works/pi-coding-agent/index.js" + + out=$(cd "$fixture/project" && node --input-type=module 2>&1 <<'JS' +const assistant = await import("./.pi/extensions/lib/fm-calm-assistant-layout.ts"); +const operational = await import("./.pi/extensions/lib/fm-calm-operational-user-layout.ts"); + +for (const [name, install, expected] of [ + ["collapsed-thinking", assistant.installCalmAssistantLayout, "AssistantMessageComponent"], + ["operational-user-row", operational.installCalmOperationalUserLayout, "InteractiveMode"], +]) { + let reason; + try { + install(); + } catch (error) { + reason = error instanceof Error ? error.message : String(error); + } + if (!reason?.includes(expected)) { + throw new Error( + `${name} adapter did not load and report its missing runtime export: ${String(reason)}`, + ); + } +} +JS +) + status=$? + [ "$status" -eq 0 ] || fail "Pi calm missing-adapter-export path failed: $out" + [ -z "$out" ] || fail "Pi calm missing-adapter-export test printed output: $out" + pass "missing Pi presentation class exports reach the independent adapter degradation path" +} + test_rendering_and_session_lifecycle() { local fixture out status version if ! command -v node >/dev/null 2>&1 || ! command -v npm >/dev/null 2>&1; then @@ -235,7 +354,7 @@ test_rendering_and_session_lifecycle() { return 0 fi version=$(node -p "require('$PI_PACKAGE_DIR/package.json').version") - [ "$version" = "0.81.1" ] || fail "Pi calm compatibility assumptions require Pi 0.81.1, found $version" + record_pi_version_evidence "$version" "Pi calm compatibility assumptions" fixture="$TMP_ROOT/renderer" mkdir -p "$fixture/home" "$fixture/lib" "$fixture/node_modules/@earendil-works" @@ -886,7 +1005,7 @@ test_operational_followup_turn_e2e() { return 0 fi version=$(pi --version 2>/dev/null || true) - [ "$version" = "0.81.1" ] || fail "Pi operational follow-up E2E requires Pi 0.81.1, found $version" + record_pi_version_evidence "$version" "Pi operational follow-up E2E" project="$TMP_ROOT/followup-project" home="$TMP_ROOT/followup-home" @@ -915,7 +1034,7 @@ let adjacent = false; let latestInputRole: "user" | "custom" | undefined; const EXACT_WATCHER_INPUT = - "\u2063FIRSTMATE_OP: v1 watcher: FIRSTMATE WATCHER WAKE: signal: /Users/kunchen/github/kunchenguid/firstmate/state/oss-triage-t4.status\n\n" + + "\u2063FIRSTMATE_OP: v1 watcher: FIRSTMATE WATCHER WAKE: signal: /home/fixture/github/kunchenguid/firstmate/state/oss-triage-t4.status\n\n" + "Run bin/fm-wake-drain.sh first and handle the queued wake. Watcher continuity is extension-owned."; function monitorInput(suffix: "ONE" | "TWO"): string { @@ -1086,7 +1205,7 @@ TS if [ "$calm_state" = on ]; then assert_not_contains "$pane" "MONITOR_${label}_ONE" "Pi follow-up $label case rendered a Calm-hidden operational user row" if [ "$label" = exact_watcher ]; then - assert_not_contains "$pane" "FIRSTMATE WATCHER WAKE: signal: /Users/kunchen/github/kunchenguid/firstmate/state/oss-triage-t4.status" \ + assert_not_contains "$pane" "FIRSTMATE WATCHER WAKE: signal: /home/fixture/github/kunchenguid/firstmate/state/oss-triage-t4.status" \ "Pi exact watcher case rendered the Calm-hidden authoritative payload" assert_not_contains "$pane" "Run bin/fm-wake-drain.sh first and handle the queued wake." \ "Pi exact watcher case rendered the Calm-hidden drain instruction" @@ -1126,7 +1245,7 @@ const handled = expected === 2 const expectedOperationalTexts = Array.from({ length: expected }, (_, index) => { const suffix = index === 0 ? "ONE" : "TWO"; return label === "exact_watcher" && suffix === "ONE" - ? "\u2063FIRSTMATE_OP: v1 watcher: FIRSTMATE WATCHER WAKE: signal: /Users/kunchen/github/kunchenguid/firstmate/state/oss-triage-t4.status\n\nRun bin/fm-wake-drain.sh first and handle the queued wake. Watcher continuity is extension-owned." + ? "\u2063FIRSTMATE_OP: v1 watcher: FIRSTMATE WATCHER WAKE: signal: /home/fixture/github/kunchenguid/firstmate/state/oss-triage-t4.status\n\nRun bin/fm-wake-drain.sh first and handle the queued wake. Watcher continuity is extension-owned." : label === "legacy_away" && suffix === "ONE" ? "\u2063Supervisor escalate (LEGACY_AWAY_E2E)" : `\u2063FIRSTMATE_OP: v1 watcher: MONITOR_${label}_${suffix}`; @@ -1190,7 +1309,7 @@ JS done assert_contains "$pane" "CAPTAIN_PROMPT_exact_watcher" "Pi restart lost the genuine captain prompt" assert_contains "$pane" "MONITOR_HANDLED_exact_watcher_ONE" "Pi restart lost the operational processing response" - assert_not_contains "$pane" "FIRSTMATE WATCHER WAKE: signal: /Users/kunchen/github/kunchenguid/firstmate/state/oss-triage-t4.status" \ + assert_not_contains "$pane" "FIRSTMATE WATCHER WAKE: signal: /home/fixture/github/kunchenguid/firstmate/state/oss-triage-t4.status" \ "Pi restart replayed the Calm-hidden exact watcher row" captain_line=$(printf '%s\n' "$pane" | grep -Fn 'CAPTAIN_ANSWER_exact_watcher' | tail -1 | cut -d: -f1) handled_line=$(printf '%s\n' "$pane" | grep -Fn 'MONITOR_HANDLED_exact_watcher_ONE' | tail -1 | cut -d: -f1) @@ -1203,7 +1322,7 @@ const entries = fs.readFileSync(process.argv[2], "utf8").trim().split("\n").map( const text = (content) => typeof content === "string" ? content : (content ?? []).filter((item) => item.type === "text").map((item) => item.text).join(""); -const exact = "\u2063FIRSTMATE_OP: v1 watcher: FIRSTMATE WATCHER WAKE: signal: /Users/kunchen/github/kunchenguid/firstmate/state/oss-triage-t4.status\n\nRun bin/fm-wake-drain.sh first and handle the queued wake. Watcher continuity is extension-owned."; +const exact = "\u2063FIRSTMATE_OP: v1 watcher: FIRSTMATE WATCHER WAKE: signal: /home/fixture/github/kunchenguid/firstmate/state/oss-triage-t4.status\n\nRun bin/fm-wake-drain.sh first and handle the queued wake. Watcher continuity is extension-owned."; const users = entries.filter((entry) => entry.type === "message" && entry.message.role === "user" && text(entry.message.content) === exact); const responses = entries.filter((entry) => entry.type === "message" && entry.message.role === "assistant" && text(entry.message.content) === "MONITOR_HANDLED_exact_watcher_ONE"); if (users.length !== 1 || responses.length !== 1) { @@ -1239,7 +1358,7 @@ test_hidden_block_geometry_e2e() { return 0 fi version=$(pi --version 2>/dev/null || true) - [ "$version" = "0.81.1" ] || fail "Pi Calm hidden-block geometry E2E requires Pi 0.81.1, found $version" + record_pi_version_evidence "$version" "Pi Calm hidden-block geometry E2E" project="$TMP_ROOT/geometry-project" home="$TMP_ROOT/geometry-home" @@ -1479,7 +1598,7 @@ test_interactive_terminal_e2e() { return 0 fi version=$(pi --version 2>/dev/null || true) - [ "$version" = "0.81.1" ] || fail "Pi calm interactive E2E requires Pi 0.81.1, found $version" + record_pi_version_evidence "$version" "Pi calm interactive E2E" project="$TMP_ROOT/e2e-project" config="$TMP_ROOT/e2e-config" @@ -1966,8 +2085,10 @@ JS pass "Pi calm native E2E keeps Working and captain turns visible, hides exact operational user rows without changing persistence, restores them Calm-off, survives restart, and preserves export plus Ctrl+O behavior" } -test_static_contract test_home_resolution +test_pi_compat_no_upper_bound +test_pi_compat_degraded_adapter +test_pi_compat_missing_adapter_exports test_rendering_and_session_lifecycle test_operational_followup_turn_e2e test_hidden_block_geometry_e2e diff --git a/tests/fm-captain-translation-contract.test.sh b/tests/fm-captain-translation-contract.test.sh index a5b9bcd8156..142b160dc29 100755 --- a/tests/fm-captain-translation-contract.test.sh +++ b/tests/fm-captain-translation-contract.test.sh @@ -1,350 +1,27 @@ #!/usr/bin/env bash -# Static regression tests for the captain-facing plain-English translation -# contract owned by AGENTS.md section 9. -# shellcheck disable=SC2016 set -u # shellcheck source=tests/lib.sh . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" -AGENTS="$ROOT/AGENTS.md" -BOOTSTRAP="$ROOT/.agents/skills/bootstrap-diagnostics/SKILL.md" -AFK="$ROOT/.agents/skills/afk/SKILL.md" -DECISION="$ROOT/.agents/skills/decision-hold-lifecycle/SKILL.md" -RECOVERY="$ROOT/.agents/skills/stuck-crewmate-recovery/SKILL.md" -HARNESS="$ROOT/.agents/skills/harness-adapters/SKILL.md" -CODEXAPP="$ROOT/.agents/skills/firstmate-codexapp/SKILL.md" -FMX="$ROOT/.agents/skills/fmx-respond/SKILL.md" -UPDATE="$ROOT/.agents/skills/updatefirstmate/SKILL.md" -AHOY="$ROOT/.agents/skills/ahoy/SKILL.md" -ADHD="$ROOT/.agents/skills/i-have-adhd/SKILL.md" -README="$ROOT/README.md" +BRIEF="$ROOT/bin/fm-brief.sh" +TMP_ROOT=$(fm_test_tmproot fm-captain-translation) -section_9() { - awk ' - /^## 9\. Escalation and captain etiquette$/ { found = 1 } - found && /^## 10\. / { exit } - found { print } - ' "$AGENTS" -} - -test_section_9_owns_positive_translation_contract() { - local contract - contract=$(section_9) - assert_contains "$contract" "Every captain-facing message must translate internal state into the project outcome, consequence, and next decision." \ - "section 9 does not own the positive captain-facing translation contract" - assert_contains "$contract" "Use the captain's nouns:" \ - "section 9 does not require captain-owned nouns" - assert_contains "$contract" "When evidence uses an internal label, rewrite it before sending:" \ - "section 9 does not own the rewrite mapping list" - pass "section 9 owns the positive captain-facing translation contract" -} - -test_scout_remains_allowed_house_vocabulary() { - local contract - contract=$(section_9) - assert_contains "$contract" "Scout and second mate are accepted Firstmate nautical house vocabulary and do not need translation" \ - "section 9 does not preserve scout as allowed Firstmate vocabulary" - assert_not_contains "$contract" "scout -> investigation" \ - "section 9 must not map scout to investigation" - assert_not_contains "$contract" "scout, ship" \ - "section 9 must not add scout to the internal-vocabulary ban" - assert_not_contains "$contract" "secondmate -> domain supervisor" \ - "section 9 must not map secondmate to domain supervisor" - pass "scout remains allowed in private captain chat" -} - -test_compressed_safety_labels_have_plain_renderings() { - local contract - contract=$(section_9) - for phrase in \ - "fail-closed" \ - "fails closed" \ - "fail-open" \ - "fails open" \ - "fail loudly"; do - assert_contains "$contract" "$phrase" "section 9 does not cover compressed safety label '$phrase'" - done - assert_contains "$contract" "stops safely when something goes wrong" \ - "fail-closed behavior lacks a concrete plain rendering" - assert_contains "$contract" "refuses rather than proceeding" \ - "fail-closed behavior lacks refusal wording" - assert_contains "$contract" "steps aside and lets work continue when the check cannot complete" \ - "fail-open behavior lacks a concrete plain rendering" - pass "compressed safety labels require concrete plain renderings" -} - -test_mapping_list_covers_high_risk_internal_families() { - local contract - contract=$(section_9) - for phrase in \ - "worktree, checkout, primary checkout, or local-main -> local copy" \ - "teardown -> cleanup" \ - "wake, watcher, heartbeat, stale, signal, or check -> notification" \ - "hold, gate, ask-user, needs-decision, blocked, or paused -> the concrete decision" \ - "done, failed, fix-review, checks-passed, cancelled, validation step, or pipeline state -> the concrete result" \ - "brief -> instructions" \ - "crewmate -> worker" \ - "harness, backend, runtime, or adapter -> worker runtime or tool" \ - "status file, metadata, state, task id, or raw path -> durable record"; do - assert_contains "$contract" "$phrase" "section 9 mapping list is missing '$phrase'" - done - pass "section 9 maps high-risk internal vocabulary families" -} - -test_verbatim_internal_evidence_is_rejected_from_chat() { - local contract - contract=$(section_9) - assert_contains "$contract" "Never relay worker reports, status lines, tool output, validation-state labels, or decision records verbatim into captain chat." \ - "section 9 does not reject verbatim internal evidence in captain chat" - assert_contains "$contract" "Private evidence reports may retain exact identifiers, paths, status lines, validation labels, and internal terms" \ - "section 9 does not preserve private evidence precision" - assert_contains "$contract" "the captain-facing chat summary that points to the report still follows this translation rule" \ - "section 9 does not keep chat summaries plain English" - pass "captain chat rejects verbatim internal evidence while private reports stay precise" -} - -test_routine_no_action_response_is_event_scoped() { - local contract - contract=$(section_9) - assert_contains "$contract" 'report its concrete outcome and evidence without characterizing the visible session' \ - "section 9 does not require concrete event-scoped routine no-action evidence" - assert_contains "$contract" 'may add nautical flavor but must not stand alone or replace required content' \ - "section 9 does not prevent a canned phrase from replacing required content" - assert_not_contains "$contract" 'Captain, no decision is needed.' \ - "section 9 implies the visible session has no unrelated open decisions" - pass "routine no-action response is exact and scoped to its event" -} - -test_outward_facing_skill_points_reference_section_9_owner() { - assert_grep "using \`AGENTS.md\` section 9's captain-facing translation contract" "$BOOTSTRAP" \ - "bootstrap diagnostics do not reference section 9 at captain handoff" - assert_grep "Acknowledge** in \`AGENTS.md\` section 9 language" "$AFK" \ - "afk acknowledgement does not reference section 9" - assert_grep "Captain, away mode is active; I will batch routine updates" "$AFK" \ - "afk acknowledgement lacks a local plain-English example" - assert_grep "as decisions from Bearings' Captain's Call section under \`AGENTS.md\` section 9" "$DECISION" \ - "decision relay does not reference section 9" - assert_grep "using \`AGENTS.md\` section 9; do not mention metadata, harness, window, or worktree" "$RECOVERY" \ - "stuck-worker failure does not reference section 9" - assert_grep "under \`AGENTS.md\` section 9 that the requested worker runtime is not verified yet" "$HARNESS" \ - "runtime fallback does not reference section 9" - assert_grep "use firstmate's own verified runtime for current work" "$HARNESS" \ - "runtime fallback does not require the current-work fallback" - assert_grep "Do not pause current work for that future-verification choice, and never launch an unverified adapter." "$HARNESS" \ - "runtime fallback permits waiting on future verification or launching an unverified adapter" - assert_grep "translate status prefixes and return-channel evidence through \`AGENTS.md\` section 9" "$CODEXAPP" \ - "Codex Desktop result reporting does not reference section 9" - assert_grep "It supplements \`AGENTS.md\` section 9; apply both, and this public-channel rule wins wherever it is stricter." "$FMX" \ - "X reply safety does not state that it supplements section 9" - assert_grep "under \`AGENTS.md\` section 9 without firstmate's internal vocabulary" "$UPDATE" \ - "Firstmate update reporting does not reference section 9" - pass "outward-facing skill handoffs point to the section 9 owner" -} - -test_section_9_owner_is_not_duplicated_into_skills() { - local duplicate_count file - duplicate_count=0 - for file in "$BOOTSTRAP" "$AFK" "$DECISION" "$RECOVERY" "$HARNESS" "$CODEXAPP" "$UPDATE"; do - if grep -Fq "When evidence uses an internal label, rewrite it before sending:" "$file"; then - duplicate_count=$((duplicate_count + 1)) - fi - done - [ "$duplicate_count" -eq 0 ] || fail "skills duplicated section 9's mapping owner" - pass "skills cross-reference section 9 instead of duplicating the mapping list" -} - -test_adhd_presentation_skill_is_internal_and_always_loaded() { - assert_present "$ADHD" "ADHD presentation skill is missing" - assert_grep 'name: i-have-adhd' "$ADHD" "ADHD presentation skill has the wrong name" - assert_grep 'user-invocable: false' "$ADHD" "ADHD presentation skill must not be user-invocable" - assert_grep ' internal: true' "$ADHD" "ADHD presentation skill is not internal" - assert_grep 'Load `i-have-adhd` before every captain-facing response' "$AGENTS" \ - "section 9 does not always load the ADHD presentation skill" - pass "ADHD presentation skill is internal and always loaded for captain chat" -} - -test_adhd_presentation_contract_has_one_owner() { - local skill contract - skill=$(cat "$ADHD") - contract=$(section_9) - for phrase in \ - "Lead with the outcome, decision, blocker, or next action" \ - "Number multi-step work" \ - "End with one concrete captain action" \ - "Suppress tangents" \ - "Restate the current step or state every turn" \ - "Give concrete time estimates when they are decision-useful" \ - "Make completed work visible" \ - "State errors matter-of-factly" \ - "Cap a list at five items" \ - "Remove preambles, recap-as-filler, filler closers, and figurative language"; do - assert_contains "$skill" "$phrase" "ADHD presentation skill is missing '$phrase'" - assert_not_contains "$contract" "$phrase" "section 9 duplicated ADHD presentation rule '$phrase'" - done - pass "ADHD presentation contract has one owner" -} - -test_adhd_skill_preserves_firstmate_boundaries() { - assert_grep "messages addressed to the captain, not worker instructions, commits, PRs, or evidence reports" "$ADHD" \ - "ADHD skill scope is not limited to captain-facing messages" - assert_grep "first line is the result or active next action" "$ADHD" \ - "ADHD skill does not define autonomous first-line behavior" - assert_grep "finish on concrete completion evidence" "$ADHD" \ - "ADHD skill invents a next action after completion" - assert_grep "completion-evidence ending takes precedence over any exact canned phrase" "$ADHD" \ - "ADHD skill does not define precedence over canned routine responses" - assert_grep "may open or season the response but never stand alone or replace that required content" "$ADHD" \ - "ADHD skill permits canned routine responses to replace required content" - assert_grep "safety, accuracy, task completeness, and required tool-call commentary outrank brevity" "$ADHD" \ - "ADHD skill does not preserve safety and completeness precedence" - assert_grep "AGENTS.md section 9 applies underneath this presentation layer and continues to own outcome translation, internal-vocabulary rewriting, standalone escalations, and concrete evidence-first context" "$ADHD" \ - "ADHD skill does not preserve section 9 ownership" - assert_grep "explicit captain request for normal mode may suspend" "$ADHD" \ - "ADHD skill does not support an explicit normal-mode override" - pass "ADHD presentation skill preserves Firstmate ownership and safety boundaries" -} - -test_ahoy_is_an_internal_user_invocable_skill() { - assert_present "$AHOY" "ahoy skill is missing" - assert_grep 'name: ahoy' "$AHOY" "ahoy skill metadata has the wrong name" - assert_grep 'user-invocable: true' "$AHOY" "ahoy skill is not user-invocable" - assert_grep ' internal: true' "$AHOY" "ahoy skill is not internal" - [ ! -e "$ROOT/skills/ahoy" ] || fail "ahoy must not exist in the public installer-facing skills directory" - pass "ahoy is internal, user-invocable, and absent from public skills" -} - -test_ahoy_readme_uses_cross_harness_convention() { - assert_grep 'Claude and grok use the slash form shown here; codex uses the same names with `$`' "$README" \ - "README lost the cross-harness slash and dollar convention" - assert_grep '| `/ahoy`' "$README" "README built-in skills table does not list /ahoy" - pass "README lists ahoy under the shared cross-harness invocation convention" -} - -test_ahoy_owns_only_the_visible_session_recap() { - assert_grep '[`../bearings/SKILL.md`](../bearings/SKILL.md)' "$AHOY" \ - "first-message fallback does not delegate to Bearings by relative pointer" - assert_grep 'If no prior real captain message exists' "$AHOY" \ - "ahoy does not limit Bearings fallback to the first real captain message" - assert_grep 'Bearings alone owns its gathering, artifact, and response contract.' "$AHOY" \ - "ahoy first-message fallback does not delegate to Bearings alone" - assert_grep 'A captain boundary is an ordinary user-role message unless it matches one of the narrow operational exclusions below.' "$AHOY" \ - "ahoy lacks an explicit captain-authored boundary rule" - assert_grep 'Exclude messages that begin with the current U+2063 `FIRSTMATE_OP:` injection prefix.' "$AHOY" \ - "ahoy does not exclude current marked operational injections" - assert_grep 'Exclude legacy bare-marker away-mode injections only when U+2063 is immediately followed by `Supervisor escalate (`.' "$AHOY" \ - "ahoy does not narrowly exclude the legacy away-mode injection shape" - assert_grep 'Exclude the exact legacy unmarked session-start payload ``Run `bin/fm-session-start.sh` now, exactly once, before executing any other instructions.``' "$AHOY" \ - "ahoy does not exclude the legacy unmarked session-start payload" - assert_grep 'quotes or embeds a current operational message after ordinary captain text' "$AHOY" \ - "ahoy lacks quoted-current near-miss protection" - assert_grep 'Apply the current exclusion only when U+2063 `FIRSTMATE_OP:` begins at the first character of the whole message' "$AHOY" \ - "ahoy does not pin the current-prefix whole-message boundary" - assert_grep 'contains ASCII `FIRSTMATE_OP:` without a leading U+2063' "$AHOY" \ - "ahoy lacks ASCII-only near-miss protection" - assert_grep 'Apply the legacy startup exclusion as a literal whole-message match: ``Captain quote: Run `bin/fm-session-start.sh` now, exactly once, before executing any other instructions.`` is a captain boundary.' "$AHOY" \ - "ahoy does not pin the altered-startup behavioral near miss" - assert_grep 'System, developer, tool, watcher, guard, away-mode, and other injected operational messages are not captain messages.' "$AHOY" \ - "ahoy incorrectly treats synthetic operational messages as captain messages" - assert_grep 'The normal recap branch is session-history-only.' "$AHOY" \ - "later ahoy invocation is not explicitly session-history-only" - assert_grep 'Do not call Bearings, shell commands, fleet snapshots, status readers, GitHub or browser APIs, tools, or file reads or writes.' "$AHOY" \ - "normal recap does not prohibit fresh fleet, file, and tool reads" - assert_grep 'Create no report, persist nothing' "$AHOY" \ - "normal recap does not prohibit artifacts and storage" - assert_grep 'do not guess current live state beyond the last visible event' "$AHOY" \ - "normal recap may falsely claim a live snapshot" - assert_grep 'The current `/ahoy` message is outside the recap interval.' "$AHOY" \ - "current ahoy invocation is not excluded from the recap interval" - assert_grep 'If context compaction makes the prior boundary unavailable' "$AHOY" \ - "ahoy does not disclose an unavailable compacted boundary" - assert_grep 'summarize only visibly supported events' "$AHOY" \ - "compacted fallback may invent unsupported events" - assert_no_grep 'fm-bearings-snapshot.sh' "$AHOY" \ - "ahoy copied Bearings gathering mechanics instead of referencing its owner" - assert_no_grep "Captain's Call" "$AHOY" \ - "ahoy copied Bearings response contract instead of referencing its owner" - pass "ahoy delegates first-message fallback and keeps later recaps visible-session-only" -} - -test_ahoy_scans_visible_history_for_open_decisions() { - assert_grep 'preserve the ordinary recap interval: recap what happened after that message and before the current invocation.' "$AHOY" \ - "ahoy no longer preserves its ordinary recap interval" - assert_grep 'inspect the entire session history visible to the current first mate before the current invocation for every explicit captain decision that remains unanswered' "$AHOY" \ - "ahoy does not scan globally visible session history for open decisions" - assert_grep 'including decisions raised before the ordinary recap boundary.' "$AHOY" \ - "ahoy does not include open decisions from before the recap boundary" - assert_grep 'A later unrelated captain message establishes a recap boundary but does not close an earlier decision.' "$AHOY" \ - "ahoy lets unrelated captain messages close earlier decisions" - assert_grep 'Treat a decision as closed only when a later visible response substantively resolves it, chooses an option, declines it, grants or denies the requested approval, or otherwise directly addresses that decision.' "$AHOY" \ - "ahoy lacks substantive-answer closure semantics" - assert_grep 'Include every visibly supported open decision once, and deduplicate by the decision' "$AHOY" \ - "ahoy does not include and deduplicate visibly open decisions" - assert_grep "substance when the ordinary interval recap already represents it or its wording differs." "$AHOY" \ - "ahoy deduplicates decisions by wording instead of substance" - assert_grep 'If no ordinary events occurred after the previous captain message but an older visibly open decision exists, report that decision instead of claiming nothing happened.' "$AHOY" \ - "ahoy can incorrectly claim nothing happened while an older decision is open" - assert_grep 'Compacted history supports an open decision only when both its request and its still-unanswered status are visible' "$AHOY" \ - "ahoy does not limit compacted decision reporting to visible support" - assert_grep 'report uncertainty instead of reconstructing hidden requests or answers.' "$AHOY" \ - "ahoy may reconstruct hidden decision history after compaction" - pass "ahoy adds visibly open decisions without changing the ordinary recap boundary" -} - -test_ahoy_user_role_injections_share_one_marker() { - local daemon grok_guard opencode_guard opencode_watch pi_guard pi_watch owner sessionstart spawn - daemon=$(cat "$ROOT/bin/fm-supervise-daemon.sh") - grok_guard=$(cat "$ROOT/bin/fm-turnend-guard-grok.sh") - opencode_guard=$(cat "$ROOT/.opencode/plugins/fm-primary-turnend-guard.js") - opencode_watch=$(cat "$ROOT/.opencode/plugins/fm-primary-watch-arm.js") - pi_guard=$(cat "$ROOT/.pi/extensions/fm-primary-turnend-guard.ts") - pi_watch=$(cat "$ROOT/.pi/extensions/fm-primary-pi-watch.ts") - owner=$(cat "$ROOT/bin/fm-operational-input.sh") - sessionstart=$(cat "$ROOT/bin/fm-sessionstart-nudge.sh") - spawn=$(cat "$ROOT/bin/fm-spawn.sh") +test_captain_facing_scout_path_preserves_evidence_and_action() { + local home id report + home="$TMP_ROOT/home" + id="captain-translation" - assert_contains "$owner" 'FM_OPERATIONAL_PREFIX="${FM_OPERATIONAL_MARK}FIRSTMATE_OP: "' \ - "canonical owner lost the landed Ahoy prefix" - assert_contains "$sessionstart" 'fm_operational_input_encode session-start' \ - "session-start does not use the canonical typed constructor" - assert_contains "$daemon" 'fm_operational_input_encode away-supervisor' \ - "away-mode does not use the canonical typed constructor" - assert_contains "$grok_guard" 'fm_operational_input_encode turn-end-guard' \ - "Grok guard does not use the canonical typed constructor" - assert_contains "$opencode_guard" 'encodeFirstmateOperationalInput(' \ - "OpenCode guard does not use the cross-language constructor" - assert_contains "$opencode_guard" '"turn-end-guard"' \ - "OpenCode guard does not retain its exact current kind" - assert_contains "$opencode_watch" 'encodeFirstmateOperationalInput(paths.root, "watcher"' \ - "OpenCode watcher does not retain its exact current kind" - assert_contains "$pi_guard" 'encodeFirstmateOperationalInput(' \ - "Pi guard does not use the cross-language constructor" - assert_contains "$pi_guard" '"turn-end-guard"' \ - "Pi guard does not retain its exact current kind" - assert_contains "$pi_watch" '"watcher"' \ - "Pi watcher does not retain its exact current kind" - assert_contains "$spawn" 'encode launch-brief' \ - "cross-harness launches do not use the canonical launch-instruction kind" - for producer in "$daemon" "$grok_guard" "$opencode_guard" "$opencode_watch" "$pi_guard" "$pi_watch" "$sessionstart" "$spawn"; do - assert_not_contains "$producer" 'FIRSTMATE_OP: ' \ - "a current producer copied the canonical marker grammar" - done - pass "ahoy: one canonical owner constructs typed operational input for every Firstmate-controlled user-role producer" + FM_HOME="$home" "$BRIEF" "$id" firstmate --scout >/dev/null 2>&1 \ + || fail "scout brief generation failed" + report="$home/data/$id/brief.md" + assert_present "$report" "scout brief was not generated" + assert_grep '# Definition of done' "$report" "scout brief lacks a completion contract" + assert_grep 'what you did, what you found, the evidence' "$report" \ + "scout brief does not require concrete evidence in its report" + assert_grep 'what you recommend' "$report" \ + "scout brief does not require a concrete recommendation" + pass "captain-facing scout work preserves evidence and recommendation handoff" } -test_section_9_owns_positive_translation_contract -test_scout_remains_allowed_house_vocabulary -test_compressed_safety_labels_have_plain_renderings -test_mapping_list_covers_high_risk_internal_families -test_verbatim_internal_evidence_is_rejected_from_chat -test_routine_no_action_response_is_event_scoped -test_outward_facing_skill_points_reference_section_9_owner -test_section_9_owner_is_not_duplicated_into_skills -test_adhd_presentation_skill_is_internal_and_always_loaded -test_adhd_presentation_contract_has_one_owner -test_adhd_skill_preserves_firstmate_boundaries -test_ahoy_is_an_internal_user_invocable_skill -test_ahoy_readme_uses_cross_harness_convention -test_ahoy_owns_only_the_visible_session_recap -test_ahoy_scans_visible_history_for_open_decisions -test_ahoy_user_role_injections_share_one_marker +test_captain_facing_scout_path_preserves_evidence_and_action diff --git a/tests/fm-cd-pretool-check.test.sh b/tests/fm-cd-pretool-check.test.sh index f623430b137..80f8c03fc90 100755 --- a/tests/fm-cd-pretool-check.test.sh +++ b/tests/fm-cd-pretool-check.test.sh @@ -372,68 +372,6 @@ test_policy_cli_direct() { # --- per-harness wiring ----------------------------------------------------- -test_claude_wiring() { - local settings n - settings="$ROOT/.claude/settings.json" - [ -f "$settings" ] || fail "tracked .claude/settings.json is missing" - n=$(jq -r '[.hooks.PreToolUse[0].hooks[].command | select(contains("fm-cd-pretool-check.sh"))] | length' "$settings") - [ "$n" = 1 ] || fail "claude PreToolUse must invoke fm-cd-pretool-check.sh exactly once" - jq -e '[.hooks.PreToolUse[0].hooks[].command | select(contains("fm-cd-pretool-check.sh") and contains("--claude") and contains("CLAUDE_PROJECT_DIR"))] | length == 1' "$settings" >/dev/null \ - || fail "claude cd hook must use CLAUDE_PROJECT_DIR and --claude" - jq -e '[.hooks.PreToolUse[0].hooks[].command | select(contains("fm-arm-pretool-check.sh"))] | length == 1' "$settings" >/dev/null \ - || fail "claude cd hook must not displace the watcher-arm hook" - pass ".claude/settings.json: PreToolUse invokes the cd-guard alongside the arm guard" -} - -test_codex_wiring() { - local settings command - settings="$ROOT/.codex/hooks.json" - [ -f "$settings" ] || fail "tracked .codex/hooks.json is missing" - command=$(jq -r '[.hooks.PreToolUse[0].hooks[].command | select(contains("fm-cd-pretool-check.sh"))][0] // empty' "$settings") - [ -n "$command" ] || fail "codex PreToolUse must invoke fm-cd-pretool-check.sh" - assert_contains "$command" 'pwd -P' "codex cd hook must anchor from the hook process working directory" - assert_contains "$command" 'fm-cd-pretool-check.sh' "codex cd hook must invoke the cd-guard" - jq -e '[.hooks.PreToolUse[0].hooks[].command | select(contains("fm-arm-pretool-check.sh"))] | length == 1' "$settings" >/dev/null \ - || fail "codex cd hook must not displace the watcher-arm hook" - pass ".codex/hooks.json: PreToolUse invokes the cd-guard alongside the arm guard" -} - -test_grok_wiring() { - local settings command - settings="$ROOT/.grok/hooks/fm-primary-cd-check.json" - [ -f "$settings" ] || fail "tracked grok cd hook config is missing" - command=$(jq -r '.hooks.PreToolUse[0].hooks[0].command // empty' "$settings") - [ -n "$command" ] || fail "grok cd hook command is missing" - assert_contains "$command" 'GROK_WORKSPACE_ROOT' "grok cd hook must anchor from GROK_WORKSPACE_ROOT" - assert_contains "$command" 'fm-cd-pretool-check.sh' "grok cd hook must invoke the cd-guard" - assert_contains "$command" '${GROK_WORKSPACE_ROOT:-}' "grok cd hook must default-guard the workspace var" - pass ".grok primary cd hook: PreToolUse invokes the cd-guard" -} - -test_opencode_wiring() { - local plugin content - plugin="$ROOT/.opencode/plugins/fm-primary-cd-check.js" - [ -f "$plugin" ] || fail "tracked OpenCode cd plugin is missing" - content=$(cat "$plugin") - assert_contains "$content" 'tool.execute.before' "OpenCode cd plugin must run before tool execution" - assert_contains "$content" 'fm-cd-pretool-check.sh' "OpenCode cd plugin must invoke the cd-guard" - assert_contains "$content" 'throw new Error' "OpenCode cd plugin must block by throwing" - assert_contains "$content" 'worktree' "OpenCode cd plugin must anchor from the git worktree path" - pass ".opencode cd plugin: tool.execute.before invokes the cd-guard and blocks by throwing" -} - -test_pi_wiring() { - local ext content - ext="$ROOT/.pi/extensions/fm-primary-turnend-guard.ts" - [ -f "$ext" ] || fail "tracked pi primary extension is missing" - content=$(cat "$ext") - assert_contains "$content" 'runCdCheck(command)' "pi extension must run the cd check in tool_call" - assert_contains "$content" 'fm-cd-pretool-check.sh' "pi extension must invoke the cd-guard owner" - assert_contains "$content" 'runPretoolCheck(command)' "pi extension must keep running the watcher-arm check" - assert_contains "$content" 'return { block: true, reason:' "pi extension must block on a checker exit 2" - pass ".pi primary extension: tool_call runs the cd-guard alongside the watcher-arm check" -} - test_scripts_are_shellcheck_clean() { command -v shellcheck >/dev/null 2>&1 || { pass "shellcheck not installed, skipping"; return; } shellcheck "$ROOT/bin/fm-cd-pretool-check.sh" >/dev/null 2>&1 \ @@ -453,9 +391,4 @@ test_fail_open_missing_node test_fail_open_missing_jq_on_stdin test_prefilter_skips_node_without_cd_substring test_policy_cli_direct -test_claude_wiring -test_codex_wiring -test_grok_wiring -test_opencode_wiring -test_pi_wiring test_scripts_are_shellcheck_clean diff --git a/tests/fm-claude-continuity-live-e2e.test.sh b/tests/fm-claude-continuity-live-e2e.test.sh deleted file mode 100755 index e6c39cda00d..00000000000 --- a/tests/fm-claude-continuity-live-e2e.test.sh +++ /dev/null @@ -1,74 +0,0 @@ -#!/usr/bin/env bash -# Opt-in credentialed Claude regression for the post-background-completion -# continuity gate. The project and FM_HOME are isolated; Claude keeps using its -# existing managed authentication. -set -u - -if [ "${FM_CLAUDE_LIVE_E2E:-0}" != 1 ]; then - echo "skip: set FM_CLAUDE_LIVE_E2E=1 to run the Claude continuity regression" - exit 0 -fi - -ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" - -fail() { - printf 'not ok - %s\n' "$1" >&2 - exit 1 -} - -command -v claude >/dev/null 2>&1 || fail "claude not found" - -LAB="$ROOT/.claude-live-e2e.$$" -PROJECT="$LAB/project" -HOME_DIR="$LAB/fmhome" -TRANSCRIPT="$LAB/claude.jsonl" -CLAUDE_VERSION=$(claude --version) - -cleanup() { - rm -rf "$LAB" -} -trap cleanup EXIT - -mkdir -p "$LAB" -git clone -q "$ROOT" "$PROJECT" -cp "$ROOT/.claude/settings.json" "$PROJECT/.claude/settings.json" -cp "$ROOT/bin/fm-continuity-pretool-check.sh" "$PROJECT/bin/fm-continuity-pretool-check.sh" -cp "$ROOT/bin/fm-continuity-command-policy.mjs" "$PROJECT/bin/fm-continuity-command-policy.mjs" -mkdir -p "$HOME_DIR/state" "$HOME_DIR/config" -printf 'project=fixture\n' > "$HOME_DIR/state/claude-e2e.meta" - -cat > "$PROJECT/bin/fm-watch-arm.sh" <<'SH' -#!/usr/bin/env bash -printf 'started\n' > "$FM_HOME/state/claude-arm-ran" -printf 'watcher: started pid=%s (fixture)\n' "$$" -sleep 0.2 -printf 'signal: fixture background completion\n' -SH -cat > "$PROJECT/bin/fm-wake-drain.sh" <<'SH' -#!/usr/bin/env bash -printf 'drained\n' > "$FM_HOME/state/claude-drain-ran" -printf 'signal: fixture background completion\n' -SH -cat > "$PROJECT/bin/fm-crew-state.sh" <<'SH' -#!/usr/bin/env bash -printf 'forbidden\n' > "$FM_HOME/state/claude-forbidden-ran" -printf 'crew state should not run\n' -SH -chmod +x "$PROJECT/bin/fm-watch-arm.sh" "$PROJECT/bin/fm-wake-drain.sh" "$PROJECT/bin/fm-crew-state.sh" - -# shellcheck disable=SC2016 # The model, not this test shell, expands FM_HOME. -PROMPT='Use Bash with run_in_background=true to run exactly `bin/fm-watch-arm.sh`. Wait for its background-task completion. Then run exactly `bin/fm-wake-drain.sh`. Without re-arming, next attempt exactly `bin/fm-crew-state.sh claude-e2e`. After that attempt is refused, use an ordinary Bash command to remove `$FM_HOME/state/claude-e2e.meta`, then reply briefly. Do not retry the refused fleet command and do not re-arm.' - -( - cd "$PROJECT" || exit 1 - FM_HOME="$HOME_DIR" FM_ROOT_OVERRIDE="$PROJECT" CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false \ - claude -p "$PROMPT" --dangerously-skip-permissions --effort low --output-format stream-json --verbose -) > "$TRANSCRIPT" 2>&1 || fail "Claude credentialed continuity turn failed: $(tail -20 "$TRANSCRIPT")" - -[ -f "$HOME_DIR/state/claude-arm-ran" ] || fail "Claude did not run the tracked background arm fixture" -[ -f "$HOME_DIR/state/claude-drain-ran" ] || fail "Claude continuity gate blocked the allowed wake drain" -[ ! -f "$HOME_DIR/state/claude-forbidden-ran" ] || fail "Claude continuity gate allowed an unrelated fleet command" -GUIDANCE='[watcher-continuity] tasks are in flight and no live watcher holds this home lock; drain wakes with bin/fm-wake-drain.sh, use fail-closed bin/fm-teardown.sh for completed tasks when needed, then re-arm with bin/fm-watch-arm.sh as a tracked Claude background task before running other fleet commands (blocked: fm-crew-state.sh)' -grep -F "$GUIDANCE" "$TRANSCRIPT" >/dev/null || fail "Claude transcript omitted the exact continuity recovery guidance" - -printf 'ok - Claude %s live E2E refused only the post-completion fleet command with exact re-arm guidance\n' "$CLAUDE_VERSION" diff --git a/tests/fm-claude-stop-autoarm-live-e2e.test.sh b/tests/fm-claude-stop-autoarm-live-e2e.test.sh new file mode 100755 index 00000000000..c7e2cab880b --- /dev/null +++ b/tests/fm-claude-stop-autoarm-live-e2e.test.sh @@ -0,0 +1,164 @@ +#!/usr/bin/env bash +# Opt-in credentialed Claude live regression for the Stop-owned auto-arm +# (bin/fm-claude-stop-autoarm.sh + bin/fm-turnend-guard.sh --claude). +# Proves, against the real installed Claude Code and the real tracked hook +# registration: a fresh session with in-flight work, no watcher, and a stale +# session lock can run fm-session-start.sh first; session start reclaims the +# dead owner; at least two tokenless auto-arm and rewake cycles then complete +# with zero model-issued arm commands; and the cooperative guard consumes no +# forced continuation while the hook's launch is healthy. +# The project and FM_HOME are isolated; Claude keeps using its existing managed +# authentication. No live fleet home, worktree, or session is touched. +# shellcheck disable=SC2016 # the model, not this test shell, reads the prompt text +set -u + +if [ "${FM_CLAUDE_LIVE_E2E:-0}" != 1 ]; then + echo "skip: set FM_CLAUDE_LIVE_E2E=1 to run the Claude Stop auto-arm regression" + exit 0 +fi + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" + +fail() { + printf 'not ok - %s\n' "$1" >&2 + exit 1 +} + +command -v claude >/dev/null 2>&1 || fail "claude not found" + +LAB="$ROOT/.claude-autoarm-live-e2e.$$" +PROJECT="$LAB/project" +HOME_DIR="$LAB/fmhome" +LIVE_OWNER_HOME="$LAB/live-owner-home" +TRANSCRIPT="$LAB/claude.jsonl" +CLAUDE_VERSION=$(claude --version) + +cleanup() { + rm -rf "$LAB" +} +trap cleanup EXIT + +mkdir -p "$LAB" +# git clone of this worktree carries only committed state, so copy the +# working-tree surfaces under test (same pattern as the continuity live E2E). +git clone -q "$ROOT" "$PROJECT" +cp -R "$ROOT/bin/." "$PROJECT/bin/" +cp "$ROOT/.claude/settings.json" "$PROJECT/.claude/settings.json" +# The lab keeps the real tracked .claude/settings.json SessionStart nudge, +# Stop guard, and asyncRewake auto-arm registration. +# The only local hook records model-issued Bash calls without acquiring the +# session lock or otherwise changing lifecycle behavior. +cat > "$PROJECT/.claude/settings.local.json" <<'JSON' +{ + "hooks": { + "PreToolUse": [ + { + "matcher": "Bash", + "hooks": [ + { "type": "command", "command": "\"$CLAUDE_PROJECT_DIR\"/bin/tool-logger.sh" } + ] + } + ] + } +} +JSON + +cat > "$PROJECT/bin/tool-logger.sh" <<'SH' +#!/usr/bin/env bash +P=$(cat 2>/dev/null || true) +printf '%s\n' "$P" | jq -r '.tool_input.command // "unknown"' >> "$FM_HOME/state/tool-calls.log" 2>/dev/null +exit 0 +SH +chmod +x "$PROJECT/bin/tool-logger.sh" + +mkdir -p "$HOME_DIR/state" "$HOME_DIR/config" "$HOME_DIR/data" +printf 'project=fixture\nwindow=fixture\nbackend=tmux\n' > "$HOME_DIR/state/task.meta" +# A numeric pid above the supported OS pid range is a demonstrably dead prior +# harness owner under fm_harness_pid_alive, matching the reproduced incident. +printf '9999999\n' > "$HOME_DIR/state/.lock" + +# Rapid-death arm fixture: started plus an immediate actionable reason, the +# exact spent-Stop edge shape. Runs 1-2 close actionable; run 3 closes clean so +# a misbehaving session can never loop forever. +cat > "$PROJECT/bin/fm-watch-arm.sh" <<'SH' +#!/usr/bin/env bash +N=$(cat "$FM_HOME/state/arm-count" 2>/dev/null || echo 0); N=$((N+1)); echo "$N" > "$FM_HOME/state/arm-count" +echo "arm-run=$N pid=$$" >> "$FM_HOME/state/arm-ran" +if [ "$N" -ge 3 ]; then + rm -f "$FM_HOME/state/task.meta" + printf 'watcher: attached pid=%s (beacon 2s)\n' "$$" + exit 0 +fi +printf 'watcher: started pid=%s (beacon fresh)\n' "$$" +printf 'stale: fixture-rapid-%s\n' "$N" +exit 0 +SH +# Drain fixture: session start invokes it once, then the model invokes it once +# per rewake. The third total drain ends the in-flight need after two complete +# Stop-owned cycles. +cat > "$PROJECT/bin/fm-wake-drain.sh" <<'SH' +#!/usr/bin/env bash +N=$(cat "$FM_HOME/state/drain-count" 2>/dev/null || echo 0); N=$((N+1)); echo "$N" > "$FM_HOME/state/drain-count" +echo "drain-run=$N" >> "$FM_HOME/state/drain-ran" +if [ "$N" -ge 3 ]; then + rm -f "$FM_HOME/state/task.meta" +fi +printf 'stale: fixture-rapid drained\n' +SH +chmod +x "$PROJECT/bin/fm-watch-arm.sh" "$PROJECT/bin/fm-wake-drain.sh" + +PROMPT='Run exactly `bin/fm-session-start.sh` with Bash as your first tool call. After reading its complete digest, reply with exactly CYCLE0 and stop. Whenever a Stop hook feedback message wakes you, run exactly `bin/fm-wake-drain.sh` once with Bash, then reply with exactly ACK and stop. Never run bin/fm-watch-arm.sh or any other arm command, and never use any other tool.' + +( + cd "$PROJECT" || exit 1 + FM_HOME="$HOME_DIR" CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false \ + claude -p "$PROMPT" --dangerously-skip-permissions --effort low --output-format stream-json --verbose +) > "$TRANSCRIPT" 2>&1 || fail "Claude credentialed auto-arm session failed: $(tail -20 "$TRANSCRIPT")" + +ARM_RUNS=$(wc -l < "$HOME_DIR/state/arm-ran" 2>/dev/null | tr -d ' ') +[ "$ARM_RUNS" = 2 ] || fail "expected exactly 2 hook-owned arm cycles, got $ARM_RUNS: $(cat "$HOME_DIR/state/arm-ran" 2>/dev/null)" +DRAIN_RUNS=$(wc -l < "$HOME_DIR/state/drain-ran" 2>/dev/null | tr -d ' ') +[ "$DRAIN_RUNS" = 3 ] || fail "expected one session-start drain plus two model wake drains, got $DRAIN_RUNS drains" +REWAKES=$(grep -c 'Stop hook feedback' "$TRANSCRIPT" 2>/dev/null || true) +[ "$REWAKES" -ge 2 ] || fail "expected at least 2 exit-2 rewake deliveries, got $REWAKES" +grep -q 'stale: fixture-rapid-1' "$TRANSCRIPT" || fail "first rapid rewake reason missing from the transcript" +grep -q 'stale: fixture-rapid-2' "$TRANSCRIPT" || fail "second rapid rewake reason missing from the transcript" +[ "$(sed -n '1p' "$HOME_DIR/state/tool-calls.log" 2>/dev/null)" = 'bin/fm-session-start.sh' ] \ + || fail "fresh Claude session did not run session start first: $(cat "$HOME_DIR/state/tool-calls.log" 2>/dev/null)" +[ "$(cat "$HOME_DIR/state/.lock" 2>/dev/null)" != 9999999 ] \ + || fail "session start did not reclaim the stale dead-owner lock" +if [ -f "$HOME_DIR/state/tool-calls.log" ]; then + ! grep -q 'fm-watch-arm.sh' "$HOME_DIR/state/tool-calls.log" \ + || fail "model issued an arm command despite Stop-owned continuity: $(cat "$HOME_DIR/state/tool-calls.log")" + ! grep -q '&' "$HOME_DIR/state/tool-calls.log" \ + || fail "model used a shell ampersand: $(cat "$HOME_DIR/state/tool-calls.log")" +fi +! grep -q 'TURN WOULD END BLIND' "$TRANSCRIPT" \ + || fail "cooperative guard consumed a forced continuation while the auto-arm launch was healthy" +[ "$(sed -n 's/^.*outcome=\([a-z][a-z]*\) .*$/\1/p' "$HOME_DIR/state/.claude-autoarm-epoch" 2>/dev/null)" = rewake ] \ + || fail "auto-arm epoch ledger must record the rewake outcome" +[ ! -e "$HOME_DIR/state/.claude-autoarm.lock" ] || fail "auto-arm owner lock was left behind" + +# Live-owner negative control: a separate supported-harness process owns a +# second isolated home while another Stop hook fires from the same primary +# project. The competing hook must not replace the session lock, arm, write an +# epoch, or rewake. +FAKE_CLAUDE="$LAB/claude" +ln -s /bin/bash "$FAKE_CLAUDE" +mkdir -p "$LIVE_OWNER_HOME/state" "$LIVE_OWNER_HOME/config" +printf 'project=fixture\n' > "$LIVE_OWNER_HOME/state/task.meta" +"$FAKE_CLAUDE" -c 'sleep 3; :' & +LIVE_OWNER_PID=$! +printf '%s\n' "$LIVE_OWNER_PID" > "$LIVE_OWNER_HOME/state/.lock" +LIVE_OWNER_RC=0 +printf '%s\n' '{"session_id":"live-owner-control"}' \ + | FM_HOME="$LIVE_OWNER_HOME" FM_ROOT_OVERRIDE="$PROJECT" "$FAKE_CLAUDE" -c '"$FM_ROOT_OVERRIDE/bin/fm-claude-stop-autoarm.sh"' \ + >"$LAB/live-owner.out" 2>"$LAB/live-owner.err" || LIVE_OWNER_RC=$? +[ "$LIVE_OWNER_RC" -eq 0 ] || fail "competing Stop hook returned $LIVE_OWNER_RC while another live session owned the home" +[ "$(cat "$LIVE_OWNER_HOME/state/.lock")" = "$LIVE_OWNER_PID" ] || fail "competing Stop hook replaced the live session owner" +[ ! -e "$LIVE_OWNER_HOME/state/arm-ran" ] || fail "competing Stop hook armed while another live session owned the home" +[ ! -e "$LIVE_OWNER_HOME/state/.claude-autoarm-epoch" ] || fail "competing Stop hook wrote an epoch while another live session owned the home" +[ ! -s "$LAB/live-owner.out" ] && [ ! -s "$LAB/live-owner.err" ] || fail "competing Stop hook produced a rewake while another live session owned the home" +wait "$LIVE_OWNER_PID" + +printf 'ok - Claude %s live E2E reclaimed a stale session lock through session start, completed two tokenless Stop-owned rewake cycles, and preserved the competing-live-owner boundary\n' "$CLAUDE_VERSION" diff --git a/tests/fm-claude-stop-autoarm.test.sh b/tests/fm-claude-stop-autoarm.test.sh new file mode 100755 index 00000000000..6be8bc15333 --- /dev/null +++ b/tests/fm-claude-stop-autoarm.test.sh @@ -0,0 +1,435 @@ +#!/usr/bin/env bash +# Behavior tests for the Claude Stop-owned watcher auto-arm +# (bin/fm-claude-stop-autoarm.sh, docs/watcher-continuity.md). +# +# The hook fires as a Claude asyncRewake Stop hook. These tests run it hermetically +# as a child of a fake harness (a bash symlink named "claude") whose pid is +# written into the fixture home's state/.lock for ordinary owned-lock cases. +# Stale-owner cases instead leave a dead recorded pid for the hook to reclaim +# through the real fm-lock.sh path. The arm wrapper is a per-test fixture, so no +# real watcher, model, or fleet state is touched. +# shellcheck disable=SC2016 # single quotes are deliberate: $FM_HOME expands inside the fake harness child, and grep needles are literal strings +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +TMP_ROOT=$(fm_test_tmproot fm-claude-stop-autoarm) +fm_git_identity fmtest fmtest@example.invalid + +FAKEBIN=$(fm_fakebin "$TMP_ROOT/fakebin") +ln -s /bin/bash "$FAKEBIN/claude" +FAKE_CLAUDE="$FAKEBIN/claude" +export FAKE_CLAUDE + +# Copy the hook and its sourced dependencies into a fixture checkout. +install_autoarm_scripts() { + local dir=$1 + mkdir -p "$dir/bin" + cp "$ROOT/bin/fm-claude-stop-autoarm.sh" "$dir/bin/fm-claude-stop-autoarm.sh" + cp "$ROOT/bin/fm-primary-scope-lib.sh" "$dir/bin/fm-primary-scope-lib.sh" + cp "$ROOT/bin/fm-supervision-lib.sh" "$dir/bin/fm-supervision-lib.sh" + cp "$ROOT/bin/fm-wake-lib.sh" "$dir/bin/fm-wake-lib.sh" + cp "$ROOT/bin/fm-session-lock-lib.sh" "$dir/bin/fm-session-lock-lib.sh" + cp "$ROOT/bin/fm-lock.sh" "$dir/bin/fm-lock.sh" + chmod +x "$dir/bin/fm-claude-stop-autoarm.sh" "$dir/bin/fm-lock.sh" +} + +make_primary_dir() { + local dir=$1 + mkdir -p "$dir/state" + git init -q "$dir" + git -C "$dir" commit -q --allow-empty -m init + : > "$dir/AGENTS.md" + install_autoarm_scripts "$dir" + printf '%s\n' "$dir" +} + +make_secondmate_dir() { + local dir=$1 + make_primary_dir "$dir" >/dev/null + printf 'sm-autoarm-1\n' > "$dir/.fm-secondmate-home" + printf '%s\n' "$dir" +} + +# A genuine linked git worktree: the shape every crewmate/scout task worktree +# has (git-dir != git-common-dir), which must keep the hook inert. +make_crewmate_worktree_dir() { + local base=$1 dir=$2 + fm_git_worktree "$base" "$dir" fm/autoarm-test-branch + mkdir -p "$dir/state" + : > "$dir/AGENTS.md" + install_autoarm_scripts "$dir" + printf '%s\n' "$dir" +} + +# Run the hook as a child of the fake harness holding the fixture home's +# session lock. $1 = fixture dir. Any extra env assignments must be exported +# before invocation. Captures stdout+stderr; exit code on stdout of the caller. +run_autoarm() { + local dir=$1 rc=0 + printf '%s\n' '{"session_id":"sess-autoarm","stop_hook_active":false}' \ + | FM_HOME="$dir" "$FAKE_CLAUDE" -c ' + printf "%s\n" "$$" > "$FM_HOME/state/.lock" + "$FM_HOME/bin/fm-claude-stop-autoarm.sh" + ' 2>&1 || rc=$? + printf 'RC=%s\n' "$rc" >&2 + return "$rc" +} + +# Arm fixture variants, installed per test as <dir>/bin/fm-watch-arm.sh. +write_arm_fixture() { + local dir=$1 kind=$2 + case "$kind" in + actionable) + cat > "$dir/bin/fm-watch-arm.sh" <<'SH' +#!/usr/bin/env bash +echo "$$" >> "$FM_HOME/state/arm-ran" +printf 'watcher: started pid=%s (beacon fresh)\n' "$$" +printf 'stale: fixture-win actionable\n' +exit 0 +SH + ;; + failed) + cat > "$dir/bin/fm-watch-arm.sh" <<'SH' +#!/usr/bin/env bash +echo "$$" >> "$FM_HOME/state/arm-ran" +printf 'watcher: FAILED - no live watcher with a fresh beacon\n' +exit 1 +SH + ;; + clean) + cat > "$dir/bin/fm-watch-arm.sh" <<'SH' +#!/usr/bin/env bash +echo "$$" >> "$FM_HOME/state/arm-ran" +printf 'watcher: attached pid=%s (beacon 2s)\n' "$$" +exit 0 +SH + ;; + slow-actionable) + cat > "$dir/bin/fm-watch-arm.sh" <<'SH' +#!/usr/bin/env bash +echo "$$" >> "$FM_HOME/state/arm-ran" +sleep 2 +printf 'watcher: started pid=%s (beacon fresh)\n' "$$" +printf 'signal: task.status done: slow fixture\n' +exit 0 +SH + ;; + meta-vanishes) + cat > "$dir/bin/fm-watch-arm.sh" <<'SH' +#!/usr/bin/env bash +echo "$$" >> "$FM_HOME/state/arm-ran" +rm -f "$FM_HOME/state/task.meta" +printf 'watcher: started pid=%s (beacon fresh)\n' "$$" +printf 'signal: task.status done: fixture\n' +exit 0 +SH + ;; + afk-appears) + cat > "$dir/bin/fm-watch-arm.sh" <<'SH' +#!/usr/bin/env bash +echo "$$" >> "$FM_HOME/state/arm-ran" +: > "$FM_HOME/state/.afk" +printf 'watcher: started pid=%s (beacon fresh)\n' "$$" +printf 'stale: fixture-win actionable\n' +exit 0 +SH + ;; + *) + echo "unknown arm fixture: $kind" >&2 + return 2 + ;; + esac + chmod +x "$dir/bin/fm-watch-arm.sh" +} + +epoch_outcome() { + sed -n 's/^.*outcome=\([a-z][a-z]*\) .*$/\1/p' "$1/state/.claude-autoarm-epoch" 2>/dev/null || true +} + +# --- registration contract ---------------------------------------------------- + +# --- scope and gates ---------------------------------------------------------- + +test_inert_in_child_worktree() { + local base dir out status + base="$TMP_ROOT/crew-base" + dir="$TMP_ROOT/crew-wt" + make_crewmate_worktree_dir "$base" "$dir" >/dev/null + : > "$dir/state/task.meta" + write_arm_fixture "$dir" actionable + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + expect_code 0 "$status" "hook must stay inert in a child task worktree" + [ ! -e "$dir/state/arm-ran" ] || fail "hook armed inside a child worktree" + [ ! -e "$dir/state/.claude-autoarm-epoch" ] || fail "hook wrote an epoch inside a child worktree" + pass "auto-arm: inert in a linked child worktree even when in-flight" +} + +test_inert_without_session_lock() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/no-lock") + : > "$dir/state/task.meta" + write_arm_fixture "$dir" actionable + # No state/.lock: run the hook directly (no fake harness, no lock file). + out=$(printf '%s\n' '{"session_id":"s"}' | FM_HOME="$dir" bash "$dir/bin/fm-claude-stop-autoarm.sh" 2>&1); status=$? + expect_code 0 "$status" "hook must stay inert when no session holds the home lock" + [ ! -e "$dir/state/arm-ran" ] || fail "hook armed without a session lock" + pass "auto-arm: inert with no session lock" +} + +test_reclaims_stale_session_lock_before_arming() { + local dir out status expected_owner actual_owner + dir=$(make_primary_dir "$TMP_ROOT/stale-lock") + : > "$dir/state/task.meta" + printf '9999999\n' > "$dir/state/.lock" + write_arm_fixture "$dir" actionable + out=$(printf '%s\n' '{"session_id":"stale"}' \ + | FM_HOME="$dir" "$FAKE_CLAUDE" -c ' + printf "%s\n" "$$" > "$FM_HOME/state/expected-owner" + "$FM_HOME/bin/fm-claude-stop-autoarm.sh" + ' 2>&1); status=$? + expect_code 2 "$status" "a dead recorded session owner must be reclaimed before the actionable rewake" + expected_owner=$(cat "$dir/state/expected-owner") + actual_owner=$(cat "$dir/state/.lock") + [ "$actual_owner" = "$expected_owner" ] || fail "stale session lock was not claimed by the current harness: expected $expected_owner, got $actual_owner" + [ -e "$dir/state/arm-ran" ] || fail "hook did not arm after reclaiming the stale session lock" + [ "$(epoch_outcome "$dir")" = rewake ] || fail "stale-lock recovery must record outcome=rewake" + pass "auto-arm: a demonstrably dead recorded session owner is reclaimed through fm-lock.sh before arming" +} + +test_inert_when_lock_held_by_other_harness() { + local dir other out status owner_after + dir=$(make_primary_dir "$TMP_ROOT/other-lock") + : > "$dir/state/task.meta" + write_arm_fixture "$dir" actionable + # The trailing no-op keeps the fake harness process alive instead of allowing + # bash to exec the final sleep into a non-harness process. + "$FAKE_CLAUDE" -c 'sleep 60; :' & + other=$! + printf '%s\n' "$other" > "$dir/state/.lock" + out=$(printf '%s\n' '{"session_id":"s"}' | FM_HOME="$dir" "$FAKE_CLAUDE" -c '"$FM_HOME/bin/fm-claude-stop-autoarm.sh"' 2>&1); status=$? + owner_after=$(cat "$dir/state/.lock") + kill "$other" 2>/dev/null || true + wait "$other" 2>/dev/null || true + expect_code 0 "$status" "hook must stay inert when another live harness holds the session lock" + [ "$owner_after" = "$other" ] || fail "hook replaced another live harness owner: expected $other, got $owner_after" + [ ! -e "$dir/state/arm-ran" ] || fail "hook armed while another session owned the lock" + [ ! -e "$dir/state/.claude-autoarm-epoch" ] || fail "hook wrote an epoch while another session owned the lock" + pass "auto-arm: inert without arm, rewake, or lock replacement when another live harness owns the home" +} + +test_inert_when_afk() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/afk") + : > "$dir/state/task.meta" + : > "$dir/state/.afk" + write_arm_fixture "$dir" actionable + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + expect_code 0 "$status" "hook must never arm or rewake while away mode owns triage" + [ ! -e "$dir/state/arm-ran" ] || fail "hook armed while state/.afk existed" + pass "auto-arm: inert while AFK owns supervision" +} + +test_stale_lock_recovery_preserves_afk_and_need_gates() { + local afk_dir idle_dir out status + afk_dir=$(make_primary_dir "$TMP_ROOT/stale-afk") + : > "$afk_dir/state/task.meta" + : > "$afk_dir/state/.afk" + printf '9999999\n' > "$afk_dir/state/.lock" + write_arm_fixture "$afk_dir" actionable + out=$(printf '%s\n' '{"session_id":"stale-afk"}' | FM_HOME="$afk_dir" "$FAKE_CLAUDE" -c '"$FM_HOME/bin/fm-claude-stop-autoarm.sh"' 2>&1); status=$? + expect_code 0 "$status" "a stale owner must not widen the AFK gate" + [ "$(cat "$afk_dir/state/.lock")" = 9999999 ] || fail "AFK stale lock was reclaimed despite away ownership" + [ ! -e "$afk_dir/state/arm-ran" ] || fail "stale AFK home armed" + + idle_dir=$(make_primary_dir "$TMP_ROOT/stale-idle") + printf '9999999\n' > "$idle_dir/state/.lock" + write_arm_fixture "$idle_dir" actionable + out=$(printf '%s\n' '{"session_id":"stale-idle"}' | FM_HOME="$idle_dir" "$FAKE_CLAUDE" -c '"$FM_HOME/bin/fm-claude-stop-autoarm.sh"' 2>&1); status=$? + expect_code 0 "$status" "a stale owner must not widen the supervision-need gate" + [ "$(cat "$idle_dir/state/.lock")" = 9999999 ] || fail "idle stale lock was reclaimed without supervision need" + [ ! -e "$idle_dir/state/arm-ran" ] || fail "stale idle home armed" + pass "auto-arm: stale-owner recovery leaves the AFK and supervision-need gates unchanged" +} + +test_resolves_outermost_claude_pid_in_nested_bgspare_chain() { + local dir out status inner_pid lock_pid + dir=$(make_primary_dir "$TMP_ROOT/nested-chain") + : > "$dir/state/task.meta" + write_arm_fixture "$dir" actionable + # A genuine multi-level contiguous claude-named ancestry: the hook fires + # inside an inner fake-claude process (its recorded pid is distinct from its + # own parent, a second, outer fake-claude process holding the session lock - + # the bg-spare shape). Only the outer pid may own the lock; a + # first-match-wins walk would resolve to the inner pid instead and leave the + # hook inert. The inner process records its own pid before running the hook + # so bash cannot tail-exec-collapse it into the outer pid, which would + # collapse the two-hop chain this test depends on down to one hop. + out=$(printf '%s\n' '{"session_id":"nested"}' \ + | FM_HOME="$dir" "$FAKE_CLAUDE" -c ' + printf "%s\n" "$$" > "$FM_HOME/state/.lock" + "$FAKE_CLAUDE" -c " + printf \"%s\n\" \"\$\$\" > \"\$FM_HOME/state/inner-pid\" + \"\$FM_HOME/bin/fm-claude-stop-autoarm.sh\" + " + ' 2>&1); status=$? + inner_pid=$(cat "$dir/state/inner-pid" 2>/dev/null || true) + lock_pid=$(cat "$dir/state/.lock" 2>/dev/null || true) + [ -n "$inner_pid" ] && [ "$inner_pid" != "$lock_pid" ] \ + || fail "test setup did not produce a genuine two-hop claude chain: inner=$inner_pid lock=$lock_pid" + expect_code 2 "$status" "a nested contiguous claude ancestry must resolve to the outer lock-owning pid and arm" + [ -e "$dir/state/arm-ran" ] || fail "hook did not resolve past the inner claude-named process to the outer lock owner" + [ "$(epoch_outcome "$dir")" = rewake ] || fail "nested-chain arm must record outcome=rewake" + pass "auto-arm: resolves the outermost pid of a nested contiguous claude ancestry (bg-spare chain)" +} + +test_inert_when_fleet_idle() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/idle") + write_arm_fixture "$dir" actionable + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + expect_code 0 "$status" "hook must exit 0 in an idle home with no X-mode poll" + [ ! -e "$dir/state/arm-ran" ] || fail "hook armed an idle home" + pass "auto-arm: inert with nothing in flight and no X-mode need" +} + +# --- the armed cycle ---------------------------------------------------------- + +test_actionable_close_rewakes_with_reason() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/actionable") + : > "$dir/state/task.meta" + write_arm_fixture "$dir" actionable + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + expect_code 2 "$status" "an actionable arm close must exit 2 so Claude rewakes" + assert_contains "$out" "firstmate watcher wake" "rewake must carry the wake banner" + assert_contains "$out" "stale: fixture-win actionable" "rewake must carry the arm's reason line" + assert_contains "$out" "bin/fm-wake-drain.sh" "rewake must direct the drain-first protocol" + assert_contains "$out" "do NOT run bin/fm-watch-arm.sh" "rewake must forbid a duplicate model re-arm" + [ "$(epoch_outcome "$dir")" = rewake ] || fail "epoch must record outcome=rewake, got: $(epoch_outcome "$dir")" + [ ! -e "$dir/state/.claude-autoarm.lock" ] || fail "owner lock must be released after the cycle" + [ -e "$dir/state/arm-ran" ] || fail "hook never foregrounded the arm wrapper" + pass "auto-arm: actionable close translates to exactly one exit-2 rewake with reason" +} + +test_failed_close_rewakes_with_failure_banner() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/failed") + : > "$dir/state/task.meta" + write_arm_fixture "$dir" failed + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + expect_code 2 "$status" "a typed watcher failure must rewake as an alarm" + assert_contains "$out" "watcher cycle FAILED" "failure rewake must carry the failure banner" + assert_contains "$out" "watcher: FAILED" "failure rewake must carry the arm's typed failure" + assert_contains "$out" "repair supervision" "failure rewake must direct the manual repair" + [ "$(epoch_outcome "$dir")" = rewake ] || fail "epoch must record outcome=rewake, got: $(epoch_outcome "$dir")" + pass "auto-arm: watcher: FAILED translates to an exit-2 alarm rewake" +} + +test_clean_close_exits_silently() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/clean") + : > "$dir/state/task.meta" + write_arm_fixture "$dir" clean + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + expect_code 0 "$status" "a clean arm close with no actionable reason must not rewake" + [ -z "$out" ] || fail "clean close produced output: $out" + [ "$(epoch_outcome "$dir")" = clean ] || fail "epoch must record outcome=clean, got: $(epoch_outcome "$dir")" + pass "auto-arm: clean close exits silently with a clean epoch" +} + +test_arms_for_x_mode_poll_need_without_inflight() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/x-need") + printf '#!/usr/bin/env bash\nexit 0\n' > "$dir/state/x-watch.check.sh" + write_arm_fixture "$dir" actionable + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + expect_code 2 "$status" "an X-mode relay poll need must keep the auto-arm active with zero tasks in flight" + [ -e "$dir/state/arm-ran" ] || fail "hook did not arm for the X-mode poll need" + pass "auto-arm: X-mode poll need arms the cycle even with no tasks in flight" +} + +test_single_flight_admits_exactly_one_owner() { + local dir rc1 rc2 count + dir=$(make_primary_dir "$TMP_ROOT/single-flight") + : > "$dir/state/task.meta" + write_arm_fixture "$dir" slow-actionable + FM_HOME="$dir" "$FAKE_CLAUDE" -c ' + printf "%s\n" "$$" > "$FM_HOME/state/.lock" + printf "%s\n" "{\"session_id\":\"s\"}" | "$FM_HOME/bin/fm-claude-stop-autoarm.sh" >/dev/null 2>"$FM_HOME/state/err1" & + p1=$! + printf "%s\n" "{\"session_id\":\"s\"}" | "$FM_HOME/bin/fm-claude-stop-autoarm.sh" >/dev/null 2>"$FM_HOME/state/err2" & + p2=$! + wait "$p1"; echo $? > "$FM_HOME/state/rc1" + wait "$p2"; echo $? > "$FM_HOME/state/rc2" + ' + rc1=$(cat "$dir/state/rc1") + rc2=$(cat "$dir/state/rc2") + count=$(wc -l < "$dir/state/arm-ran" | tr -d ' ') + [ "$count" -eq 1 ] || fail "concurrent firings must foreground exactly one arm, saw $count" + { [ "$rc1" = 2 ] && [ "$rc2" = 0 ]; } || { [ "$rc1" = 0 ] && [ "$rc2" = 2 ]; } \ + || fail "exactly one firing must translate the close (rc 2) and the other must no-op (rc 0), got rc1=$rc1 rc2=$rc2" + pass "auto-arm: concurrent firings admit one owner and one rewake translation" +} + +test_need_vanished_mid_cycle_closes_quietly() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/vanished") + : > "$dir/state/task.meta" + write_arm_fixture "$dir" meta-vanishes + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + expect_code 0 "$status" "an actionable close after the fleet went idle must not rewake" + [ -z "$out" ] || fail "vanished-need close produced output: $out" + [ "$(epoch_outcome "$dir")" = clean ] || fail "epoch must record outcome=clean, got: $(epoch_outcome "$dir")" + pass "auto-arm: need vanishing mid-cycle closes without a rewake" +} + +test_afk_mid_cycle_suppresses_rewake() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/afk-mid") + : > "$dir/state/task.meta" + write_arm_fixture "$dir" afk-appears + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + expect_code 0 "$status" "AFK appearing mid-cycle must suppress the primary rewake" + [ -z "$out" ] || fail "AFK-suppressed close produced output: $out" + [ "$(epoch_outcome "$dir")" = afk ] || fail "epoch must record outcome=afk, got: $(epoch_outcome "$dir")" + pass "auto-arm: mid-cycle AFK hands triage to the daemon with no rewake" +} + +test_active_in_marked_secondmate_home() { + local dir out status + dir=$(make_secondmate_dir "$TMP_ROOT/secondmate") + : > "$dir/state/task.meta" + write_arm_fixture "$dir" actionable + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + expect_code 2 "$status" "a marked secondmate home must get the same active auto-arm as the main primary" + [ -e "$dir/state/arm-ran" ] || fail "hook did not arm in a marked secondmate home" + [ "$(epoch_outcome "$dir")" = rewake ] || fail "secondmate epoch must record outcome=rewake" + pass "auto-arm: active in a marked secondmate home" +} + +test_fm_lock_status_still_works_with_shared_lib() { + local out + out=$(FM_HOME="$TMP_ROOT/lock-status-home" bash "$ROOT/bin/fm-lock.sh" status 2>&1) + assert_contains "$out" "lock: free" "fm-lock.sh status must keep working after the session-lock lib extraction" + pass "fm-lock: shared session-lock lib preserves the status path" +} + +test_inert_in_child_worktree +test_inert_without_session_lock +test_reclaims_stale_session_lock_before_arming +test_inert_when_lock_held_by_other_harness +test_inert_when_afk +test_stale_lock_recovery_preserves_afk_and_need_gates +test_resolves_outermost_claude_pid_in_nested_bgspare_chain +test_inert_when_fleet_idle +test_actionable_close_rewakes_with_reason +test_failed_close_rewakes_with_failure_banner +test_clean_close_exits_silently +test_arms_for_x_mode_poll_need_without_inflight +test_single_flight_admits_exactly_one_owner +test_need_vanished_mid_cycle_closes_quietly +test_afk_mid_cycle_suppresses_rewake +test_active_in_marked_secondmate_home +test_fm_lock_status_still_works_with_shared_lib diff --git a/tests/fm-composer-ghost.test.sh b/tests/fm-composer-ghost.test.sh index 0076b7a575a..7249574ea3e 100755 --- a/tests/fm-composer-ghost.test.sh +++ b/tests/fm-composer-ghost.test.sh @@ -13,7 +13,8 @@ # normal-intensity, brightly-coloured text. # 2. fm_pane_input_pending reads a ghost-only composer (either style) as NOT # pending, while still treating real (normal/bright) text as pending. -# 3. The human/LLM-facing capture path (fm-peek.sh) stays PLAIN - no escape codes +# 3. The tmux reader structurally scans every row of a multi-row composer. +# 4. The human/LLM-facing capture path (fm-peek.sh) stays PLAIN - no escape codes # ever reach firstmate's context. set -u @@ -35,7 +36,9 @@ ESC=$(printf '\033') # escape-free line for the plain (peek) path. capture-pane returns the styled # fixture verbatim WITH -e (mirrors `tmux capture-pane -e`), and the same content # with SGR sequences stripped WITHOUT -e (mirrors a plain capture). cursor_y comes -# from FM_FAKE_CY. +# from FM_FAKE_CY. The fake deliberately returns the complete fixture for every +# capture, which exercises the structural scan while preserving the historical +# single-row fallback fixtures. make_fake_tmux() { # <dir> local dir=$1 fb="$1/fakebin" mkdir -p "$fb" @@ -48,8 +51,21 @@ case "${1:-}" in printf 'fakepane\n'; exit 0 ;; capture-pane) has_e=0 - for a in "$@"; do [ "$a" = "-e" ] && has_e=1; done + start= end= prev= + for a in "$@"; do + [ "$a" = "-e" ] && has_e=1 + case "$prev" in + -S) start=$a ;; + -E) end=$a ;; + esac + prev=$a + done f="${FM_FAKE_STYLED:-/dev/null}" + if [ -n "${FM_FAKE_ROW:-}" ] \ + && [ "$start" = "${FM_FAKE_CY:-0}" ] \ + && [ "$end" = "${FM_FAKE_CY:-0}" ]; then + f=$FM_FAKE_ROW + fi if [ "$has_e" = 1 ]; then cat "$f" 2>/dev/null else @@ -160,8 +176,8 @@ test_dim_ghost_inside_bordered_composer_is_not_pending() { fb=$(make_fake_tmux "$dir") capture="$dir/styled.txt" # Bordered composer (claude box) holding only dim ghost text. - printf '\xe2\x94\x82 \033[2mtry the other approach instead\033[0m \xe2\x94\x82\n' > "$capture" - if PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=0 \ + printf '╭─────────────────────────────────────╮\n│ \033[2mtry the other approach instead\033[0m │\n╰─────────────────────────────────────╯\n' > "$capture" + if PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=1 \ fm_pane_input_pending "fakepane"; then fail "dim ghost in a bordered composer falsely read as pending" fi @@ -250,6 +266,310 @@ test_real_text_with_trailing_ghost_is_pending() { pass "fm_pane_input_pending: real text plus a trailing ghost run is still pending" } +# --- fm_tmux_composer_state: structural multi-row box scan ------------------ + +test_two_row_composer_reads_text_above_empty_cursor_row() { + local dir fb capture out + dir="$TMP_ROOT/two-row"; mkdir -p "$dir" + fb=$(make_fake_tmux "$dir") + capture="$dir/styled.txt" + cat > "$capture" <<'EOF' +╭────────────────────────────────────────────────────╮ +│ > Read the brief at /tmp/brief.md and follow it. │ +│ │ +╰────────────────────────────────────────────────────╯ +EOF + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=2 \ + fm_tmux_composer_state "fakepane") + [ "$out" = pending ] \ + || fail "two-row composer text above the empty cursor row should be pending, got '$out'" + pass "fm_tmux_composer_state: text above an empty cursor row is pending" +} + +test_wrapped_composer_reads_all_content_rows() { + local dir fb capture out + dir="$TMP_ROOT/wrapped"; mkdir -p "$dir" + fb=$(make_fake_tmux "$dir") + capture="$dir/styled.txt" + cat > "$capture" <<'EOF' +╭────────────────────────────────────────────╮ +│ > This deliberately long instruction wraps │ +│ across a second composer content row and │ +│ across a third composer content row too. │ +│ │ +╰────────────────────────────────────────────╯ +EOF + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=4 \ + fm_tmux_composer_state "fakepane") + [ "$out" = pending ] \ + || fail "a three-row wrapped composer should be pending, got '$out'" + pass "fm_tmux_composer_state: a message wrapped across three rows is pending" +} + +test_bottom_border_cursor_reads_ghost_only_box_as_empty() { + local dir fb capture out + dir="$TMP_ROOT/bottom-border-ghost"; mkdir -p "$dir" + fb=$(make_fake_tmux "$dir") + capture="$dir/styled.txt" + printf '╭────────────────────────╮\n│ ❯ \033[38;2;50;47;70mType a message...\033[0m │\n╰────────────────────────╯\n' > "$capture" + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=2 \ + fm_tmux_composer_state "fakepane") + [ "$out" = empty ] \ + || fail "a ghost-only box with the cursor on its bottom border should be empty, got '$out'" + pass "fm_tmux_composer_state: Grok's bottom-border cursor quirk reads an empty box structurally" +} + +test_bordered_busy_signatures_are_pending() { + local dir fb capture out signature + dir="$TMP_ROOT/bordered-busy-signatures"; mkdir -p "$dir" + fb=$(make_fake_tmux "$dir") + capture="$dir/styled.txt" + for signature in 'Working...' 'Ctrl+c:cancel'; do + printf '╭────────────────────╮\n│ %-18s │\n╰────────────────────╯\n' "$signature" > "$capture" + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=1 \ + fm_tmux_composer_state "fakepane") + [ "$out" = pending ] \ + || fail "typed bordered busy signature '$signature' should be pending, got '$out'" + done + pass "fm_tmux_composer_state: typed Pi and Grok busy signatures inside a box are pending" +} + +test_non_bordered_busy_footer_remains_empty() { + local dir fb capture out + dir="$TMP_ROOT/non-bordered-busy"; mkdir -p "$dir" + fb=$(make_fake_tmux "$dir") + capture="$dir/styled.txt" + printf 'Working...\n' > "$capture" + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=0 \ + fm_tmux_composer_state "fakepane") + [ "$out" = empty ] \ + || fail "a non-bordered busy footer should remain empty, got '$out'" + pass "fm_tmux_composer_state: non-bordered busy footers retain compatibility behavior" +} + +test_clipped_bordered_box_is_unknown() { + local dir fb capture out + dir="$TMP_ROOT/clipped-box"; mkdir -p "$dir" + fb=$(make_fake_tmux "$dir") + capture="$dir/styled.txt" + printf '╭────────────────────╮\n│ > │\n' > "$capture" + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=1 \ + fm_tmux_composer_state "fakepane") + [ "$out" = unknown ] \ + || fail "a bordered box with no readable bottom border should be unknown, got '$out'" + pass "fm_tmux_composer_state: an unbounded bordered box fails closed as unknown" +} + +test_asymmetric_composer_edges_are_unknown() { + local dir fb capture out row + dir="$TMP_ROOT/asymmetric-box"; mkdir -p "$dir" + fb=$(make_fake_tmux "$dir") + capture="$dir/styled.txt" + for row in '│ > text' 'text │' '│ > text ┃' '──────' '+----' 'text +'; do + printf '%s\n' "$row" > "$capture" + out=$(PATH="$fb:$PATH" LC_ALL=C FM_FAKE_STYLED="$capture" FM_FAKE_CY=0 \ + fm_tmux_composer_state "fakepane") + [ "$out" = unknown ] \ + || fail "unbounded composer-edge row '$row' should be unknown, got '$out'" + done + pass "fm_tmux_composer_state: clipped and asymmetric composer edges fail closed" +} + +test_mismatched_box_families_are_unknown() { + local dir fb capture out + dir="$TMP_ROOT/mismatched-box"; mkdir -p "$dir" + fb=$(make_fake_tmux "$dir") + capture="$dir/styled.txt" + printf '╭────────╮\n┃ > ┃\n╰────────╯\n' > "$capture" + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=1 \ + fm_tmux_composer_state "fakepane") + [ "$out" = unknown ] \ + || fail "a box with mismatched border families should be unknown, got '$out'" + pass "fm_tmux_composer_state: inconsistent box geometry fails closed" +} + +test_misaligned_box_is_unknown() { + local dir fb capture out fixture + dir="$TMP_ROOT/misaligned-box"; mkdir -p "$dir" + fb=$(make_fake_tmux "$dir") + capture="$dir/styled.txt" + for fixture in offset width; do + case "$fixture" in + offset) printf ' ╭────╮\n │ │\n╰────╯\n' > "$capture" ;; + width) printf '╭────╮\n│ │\n╰────╯\n' > "$capture" ;; + esac + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=1 \ + fm_tmux_composer_state "fakepane") + [ "$out" = unknown ] \ + || fail "a box with inconsistent $fixture geometry should be unknown, got '$out'" + done + pass "fm_tmux_composer_state: misaligned box bounds fail closed" +} + +test_unproved_empty_geometry_is_unknown() { + local dir fb capture out fixture + dir="$TMP_ROOT/unproved-empty-geometry"; mkdir -p "$dir" + fb=$(make_fake_tmux "$dir") + capture="$dir/styled.txt" + for fixture in ghost idle malformed-top; do + case "$fixture" in + ghost) + printf '╭────────────╮\n│ \033[2mghost\033[0m │\n╰────────────╯\n' > "$capture" + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=1 \ + fm_tmux_composer_state "fakepane") + ;; + idle) + printf '╭────────────╮\n│ idle hint │\n╰────────────╯\n' > "$capture" + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=1 \ + FM_COMPOSER_IDLE_RE='^idle hint$' fm_tmux_composer_state "fakepane") + ;; + malformed-top) + printf '╭────x───────╮\n│ │\n╰────────────╯\n' > "$capture" + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=1 \ + fm_tmux_composer_state "fakepane") + ;; + esac + [ "$out" = unknown ] \ + || fail "unproved empty geometry '$fixture' should be unknown, got '$out'" + done + pass "fm_tmux_composer_state: unproved ghost, idle, and border geometry stays unknown" +} + +test_differing_widths_use_asymmetric_verdicts() { + local dir fb capture out + dir="$TMP_ROOT/differing-widths"; mkdir -p "$dir" + fb=$(make_fake_tmux "$dir") + capture="$dir/styled.txt" + printf '╭──────────╮\n│ > text │\n╰──────────╯\n' > "$capture" + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=1 \ + fm_tmux_composer_state "fakepane") + [ "$out" = pending-unproven ] \ + || fail "text in a differing-width box should be pending-unproven, got '$out'" + printf '╭──────────╮\n│ │\n╰──────────╯\n' > "$capture" + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=1 \ + fm_tmux_composer_state "fakepane") + [ "$out" = unknown ] \ + || fail "an empty differing-width box should be unknown, never empty, got '$out'" + pass "fm_tmux_composer_state: differing widths prefer pending or unknown, never empty" +} + +test_wide_composer_text_is_pending() { + local dir fb capture out text + dir="$TMP_ROOT/wide-text"; mkdir -p "$dir" + fb=$(make_fake_tmux "$dir") + capture="$dir/styled.txt" + for text in '修复登录问题' 'fix the bug 🔧'; do + printf '╭────────────────────╮\n│ > %s │\n╰────────────────────╯\n' "$text" > "$capture" + out=$(PATH="$fb:$PATH" LC_ALL=C FM_FAKE_STYLED="$capture" FM_FAKE_CY=1 \ + fm_tmux_composer_state "fakepane") + [ "$out" = pending-unproven ] \ + || fail "wide composer text '$text' should be pending-unproven, got '$out'" + done + pass "fm_tmux_composer_state: emoji and CJK text remain pending under the C locale" +} + +test_all_tmux_harness_composers_share_classification() { + local dir fb capture out harness + dir="$TMP_ROOT/all-harness-composers"; mkdir -p "$dir" + fb=$(make_fake_tmux "$dir") + capture="$dir/styled.txt" + for harness in claude codex opencode pi pi-signed grok; do + case "$harness" in + claude) printf '╭────────────╮\n│ ❯ \033[2mtry\033[0m │\n╰────────────╯\n' > "$capture" ;; + codex) printf '╭────────────╮\n│ › \033[2mtip\033[0m │\n╰────────────╯\n' > "$capture" ;; + opencode) printf '╭────────────╮\n│ > │\n╰────────────╯\n' > "$capture" ;; + pi|pi-signed) printf '╭────────────╮\n│ │\n╰────────────╯\n' > "$capture" ;; + grok) printf '╭────────────╮\n│ ❯ \033[38;2;50;47;70mType\033[0m │\n╰────────────╯\n' > "$capture" ;; + esac + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=1 \ + fm_tmux_composer_state "fakepane") + [ "$out" = empty ] \ + || fail "$harness aligned idle composer should be empty, got '$out'" + case "$harness" in + claude|grok) printf '╭────────────╮\n│ ❯ fix │\n╰────────────╯\n' > "$capture" ;; + codex) printf '╭────────────╮\n│ › fix │\n╰────────────╯\n' > "$capture" ;; + opencode|pi|pi-signed) printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$capture" ;; + esac + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=1 \ + fm_tmux_composer_state "fakepane") + [ "$out" = pending ] \ + || fail "$harness composer with text should be pending, got '$out'" + done + pass "fm_tmux_composer_state: all tmux harnesses share empty and pending classification" +} + +test_unrecognized_state_defers_input_guard() { + ( + # shellcheck disable=SC2329 + fm_tmux_composer_state() { printf 'future-state'; } + fm_pane_input_pending "fakepane" + ) || fail "an unrecognized composer state should defer the input guard" + pass "fm_pane_input_pending: unrecognized states defer by default" +} + +test_fallback_capture_race_with_edge_is_unknown() { + local dir fb capture row_capture out + dir="$TMP_ROOT/fallback-race"; mkdir -p "$dir" + fb=$(make_fake_tmux "$dir") + capture="$dir/styled.txt" + row_capture="$dir/row.txt" + printf '› deploy staging\n' > "$capture" + printf '│ > │\n' > "$row_capture" + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_ROW="$row_capture" FM_FAKE_CY=0 \ + fm_tmux_composer_state "fakepane") + [ "$out" = unknown ] \ + || fail "an edge appearing between full-pane and fallback captures should be unknown, got '$out'" + pass "fm_tmux_composer_state: fallback capture races cannot admit unbounded edges" +} + +test_legitimate_empty_routes_remain_empty() { + local dir fb capture out fixture cursor + dir="$TMP_ROOT/legitimate-empty"; mkdir -p "$dir" + fb=$(make_fake_tmux "$dir") + capture="$dir/styled.txt" + for fixture in bordered double-bordered agent-prompt blank; do + case "$fixture" in + bordered) printf '╭────╮\n│ │\n╰────╯\n' > "$capture"; cursor=1 ;; + double-bordered) printf '╔════╗\n║ ║\n╚════╝\n' > "$capture"; cursor=1 ;; + agent-prompt) printf '›\n' > "$capture"; cursor=0 ;; + blank) printf '\n' > "$capture"; cursor=0 ;; + esac + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY="$cursor" \ + fm_tmux_composer_state "fakepane") + [ "$out" = empty ] \ + || fail "legitimate empty route '$fixture' should remain empty, got '$out'" + done + pass "fm_tmux_composer_state: only proven structural and non-bordered empty routes stay empty" +} + +test_non_bordered_composer_uses_compatibility_fallback() { + local dir fb capture out + dir="$TMP_ROOT/non-bordered-fallback"; mkdir -p "$dir" + fb=$(make_fake_tmux "$dir") + capture="$dir/styled.txt" + printf '› deploy staging\n' > "$capture" + out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=0 \ + fm_tmux_composer_state "fakepane") + [ "$out" = pending ] \ + || fail "a non-bordered composer should retain cursor-row classification, got '$out'" + pass "fm_tmux_composer_state: panes without bordered structure retain compatibility fallback" +} + +test_non_bordered_interior_edges_are_pending() { + local dir fb capture out row + dir="$TMP_ROOT/non-bordered-interior-edges"; mkdir -p "$dir" + fb=$(make_fake_tmux "$dir") + capture="$dir/styled.txt" + for row in '› cat file | grep x' '› explain │ this glyph'; do + printf '%s\n' "$row" > "$capture" + out=$(PATH="$fb:$PATH" LC_ALL=C FM_FAKE_STYLED="$capture" FM_FAKE_CY=0 \ + fm_tmux_composer_state "fakepane") + [ "$out" = pending ] \ + || fail "non-bordered interior edge row '$row' should be pending, got '$out'" + done + pass "fm_tmux_composer_state: interior edge glyphs retain non-bordered fallback" +} + # --- fm-peek.sh stays escape-free (LLM-facing path) ------------------------- test_peek_output_is_escape_free() { @@ -287,4 +607,22 @@ test_colored_text_with_2_payload_still_pending test_dark_truecolor_ghost_only_composer_is_not_pending test_dark_truecolor_bare_shell_prompt_is_unknown test_real_text_with_trailing_ghost_is_pending +test_two_row_composer_reads_text_above_empty_cursor_row +test_wrapped_composer_reads_all_content_rows +test_bottom_border_cursor_reads_ghost_only_box_as_empty +test_bordered_busy_signatures_are_pending +test_non_bordered_busy_footer_remains_empty +test_clipped_bordered_box_is_unknown +test_asymmetric_composer_edges_are_unknown +test_mismatched_box_families_are_unknown +test_misaligned_box_is_unknown +test_unproved_empty_geometry_is_unknown +test_differing_widths_use_asymmetric_verdicts +test_wide_composer_text_is_pending +test_all_tmux_harness_composers_share_classification +test_unrecognized_state_defers_input_guard +test_fallback_capture_race_with_edge_is_unknown +test_legitimate_empty_routes_remain_empty +test_non_bordered_composer_uses_compatibility_fallback +test_non_bordered_interior_edges_are_pending test_peek_output_is_escape_free diff --git a/tests/fm-continuity-pretool-check.test.sh b/tests/fm-continuity-pretool-check.test.sh deleted file mode 100755 index f4df87f323d..00000000000 --- a/tests/fm-continuity-pretool-check.test.sh +++ /dev/null @@ -1,125 +0,0 @@ -#!/usr/bin/env bash -# Behavior tests for Claude's narrowly scoped watcher-continuity PreToolUse gate. -set -u - -# shellcheck source=tests/lib.sh -. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" - -CHECK="$ROOT/bin/fm-continuity-pretool-check.sh" -WATCH="$ROOT/bin/fm-watch.sh" -TMP_ROOT=$(fm_test_tmproot fm-continuity-pretool-tests) -PRIMARY="$TMP_ROOT/primary" -STATE="$PRIMARY/state" -OUT="$TMP_ROOT/out" -ERR="$TMP_ROOT/err" - -mkdir -p "$PRIMARY/bin" "$STATE" -printf '# fixture\n' > "$PRIMARY/AGENTS.md" -git -C "$PRIMARY" init -q - -run_command() { - local command=$1 rc=0 - : > "$OUT" - : > "$ERR" - FM_ROOT_OVERRIDE="$PRIMARY" FM_HOME="$PRIMARY" FM_STATE_OVERRIDE="$STATE" \ - "$CHECK" --command "$command" > "$OUT" 2> "$ERR" || rc=$? - return "$rc" -} - -expect_allow() { - local label=$1 command=$2 rc=0 - run_command "$command" || rc=$? - [ "$rc" -eq 0 ] || fail "$label must allow, got exit $rc: $(cat "$ERR")" - [ ! -s "$OUT" ] || fail "$label allow wrote stdout: $(cat "$OUT")" - [ ! -s "$ERR" ] || fail "$label allow wrote stderr: $(cat "$ERR")" -} - -expect_deny() { - local label=$1 command=$2 blocked=$3 expected=${4:-} rc=0 actual - run_command "$command" || rc=$? - [ "$rc" -eq 2 ] || fail "$label must deny with exit 2, got $rc" - [ ! -s "$OUT" ] || fail "$label deny wrote stdout: $(cat "$OUT")" - jq -e '.hookSpecificOutput.hookEventName == "PreToolUse" and .hookSpecificOutput.permissionDecision == "deny"' "$ERR" >/dev/null 2>&1 \ - || fail "$label deny omitted Claude's permission decision: $(cat "$ERR")" - [ -n "$expected" ] || expected="[watcher-continuity] tasks are in flight and no live watcher holds this home lock; drain wakes with bin/fm-wake-drain.sh, use fail-closed bin/fm-teardown.sh for completed tasks when needed, then re-arm with bin/fm-watch-arm.sh as a tracked Claude background task before running other fleet commands (blocked: $blocked)" - actual=$(jq -r '.systemMessage' "$ERR") - [ "$actual" = "$expected" ] || fail "$label recovery guidance changed: $actual" -} - -test_gate_scope_and_recovery_exceptions() { - expect_allow "idle fleet command" 'bin/fm-crew-state.sh task' - printf 'project=fixture\n' > "$STATE/task.meta" - - expect_allow "ordinary shell command" 'git status --short' - expect_allow "fleet-script text as data" "rg -n 'bin/fm-send.sh' docs" - expect_allow "wake drain recovery" 'bin/fm-wake-drain.sh' - expect_allow "watch arm recovery" 'bin/fm-watch-arm.sh' - expect_allow "drain then arm recovery" 'bin/fm-wake-drain.sh; bin/fm-watch-arm.sh' - expect_allow "fail-closed teardown recovery" 'bin/fm-teardown.sh task' - unsafe_teardown_reason='[watcher-continuity] tasks are in flight and no live watcher holds this home lock; during recovery only the ordinary literal bin/fm-teardown.sh is allowed, so drop --force and any shell-expanded arguments and retry the literal invocation (blocked: fm-teardown.sh)' - expect_deny "forced teardown is not recovery" 'bin/fm-teardown.sh task --force' 'fm-teardown.sh' "$unsafe_teardown_reason" - expect_deny "nested forced teardown is not recovery" "bash -lc 'bin/fm-teardown.sh task --force'" 'fm-teardown.sh' "$unsafe_teardown_reason" - # shellcheck disable=SC2016 # single quotes are deliberate: "$TEARDOWN_MODE" is literal test data (an unsafe shell-expanded arg the gate must deny), not an expansion here - expect_deny "dynamic teardown mode is not recovery" 'bin/fm-teardown.sh task "$TEARDOWN_MODE"' 'fm-teardown.sh' "$unsafe_teardown_reason" - expect_deny "unrelated fleet command" 'bin/fm-crew-state.sh task' 'fm-crew-state.sh' - expect_deny "recovery bundled with unrelated fleet command" 'bin/fm-wake-drain.sh; bin/fm-send.sh task hi' 'fm-send.sh' - expect_deny "literal nested fleet command" "bash -lc 'bin/fm-bootstrap.sh'" 'fm-bootstrap.sh' - pass "continuity gate allows recovery and ordinary commands but denies only other fleet execution" -} - -test_live_lock_allows_fleet_command_even_with_stale_beacon() { - local holder identity rc=0 - sleep 300 & - holder=$! - identity=$(FM_STATE_OVERRIDE="$STATE" bash -c '. "$1"; fm_pid_identity "$2"' _ "$ROOT/bin/fm-wake-lib.sh" "$holder") \ - || fail "could not identify live continuity fixture" - mkdir -p "$STATE/.watch.lock" - printf '%s\n' "$holder" > "$STATE/.watch.lock/pid" - printf '%s\n' "$PRIMARY" > "$STATE/.watch.lock/fm-home" - printf '%s\n' "$WATCH" > "$STATE/.watch.lock/watcher-path" - printf '%s\n' "$identity" > "$STATE/.watch.lock/pid-identity" - touch -t 200001010000 "$STATE/.last-watcher-beat" - - run_command 'bin/fm-crew-state.sh task' || rc=$? - kill "$holder" 2>/dev/null || true - wait "$holder" 2>/dev/null || true - [ "$rc" -eq 0 ] || fail "identity-matched live lock must allow fleet command even when its beacon is stale" - [ ! -s "$ERR" ] || fail "live-lock allow wrote stderr: $(cat "$ERR")" - pass "continuity gate classifies the lock by live PID identity rather than beacon age" -} - -test_child_worktree_and_malformed_input_fail_open() { - local child="$TMP_ROOT/child" rc=0 - rm -rf "$STATE/.watch.lock" - git -C "$PRIMARY" config user.name fixture - git -C "$PRIMARY" config user.email fixture@example.test - git -C "$PRIMARY" add AGENTS.md - git -C "$PRIMARY" commit -qm fixture - git -C "$PRIMARY" worktree add -q -b fixture-child "$child" - mkdir -p "$child/bin" "$child/state" - FM_ROOT_OVERRIDE="$child" FM_HOME="$child" FM_STATE_OVERRIDE="$child/state" \ - "$CHECK" --command 'bin/fm-send.sh task hi' > "$OUT" 2> "$ERR" || rc=$? - [ "$rc" -eq 0 ] || fail "linked child worktree must be out of continuity-gate scope" - - expect_allow "malformed dynamic shell" "bin/fm-send.sh 'unterminated" - printf '%s' '{not-json' | FM_ROOT_OVERRIDE="$PRIMARY" FM_HOME="$PRIMARY" FM_STATE_OVERRIDE="$STATE" \ - "$CHECK" > "$OUT" 2> "$ERR" || rc=$? - [ "$rc" -eq 0 ] || fail "malformed Claude transport must fail open" - pass "continuity gate excludes child worktrees and fails open on opaque input" -} - -test_claude_hook_registration_preserves_stop_backstop() { - jq -e ' - [.hooks.PreToolUse[] | select(.matcher == "Bash") | .hooks[].command] - | any(contains("fm-continuity-pretool-check.sh")) - ' "$ROOT/.claude/settings.json" >/dev/null || fail "Claude settings omit the continuity PreToolUse hook" - jq -e ' - .hooks.Stop == [{"hooks":[{"type":"command","command":"\"$CLAUDE_PROJECT_DIR\"/bin/fm-turnend-guard.sh"}]}] - ' "$ROOT/.claude/settings.json" >/dev/null || fail "Claude Stop turn-end backstop changed" - pass "Claude wires the continuity gate while preserving the existing Stop backstop byte-for-byte" -} - -test_gate_scope_and_recovery_exceptions -test_live_lock_allows_fleet_command_even_with_stale_beacon -test_child_worktree_and_malformed_input_fail_open -test_claude_hook_registration_preserves_stop_backstop diff --git a/tests/fm-daemon.test.sh b/tests/fm-daemon.test.sh index 6782e09bd79..a1fbd7f5fea 100755 --- a/tests/fm-daemon.test.sh +++ b/tests/fm-daemon.test.sh @@ -138,6 +138,57 @@ test_stale_transient_self_records_marker() { pass "transient stale self-handles and records a persistence marker" } +test_stale_diagnostic_wedge_survives_busy_housekeeping() { + local case_name dir state fakebin key task win pane reason status_line action_log + for case_name in working prior-terminal paused; do + dir=$(make_supercase "stale-diagnostic-$case_name") + state="$dir/state" + fakebin="$dir/fakebin" + task="suffix-$case_name" + win="sess:fm-$task" + pane="$dir/pane.txt" + action_log="$dir/actions.log" + reason="stale: $win (idle 500s, possible wedge, escalation 3, demand-deep-inspection: same pane has wedge-escalated 3 times in a row - do not re-absorb on the run-step/pane state alone)" + fm_write_meta "$state/$task.meta" "window=$win" "backend=tmux" + case "$case_name" in + working) status_line='working: building' ;; + prior-terminal) status_line='done: already surfaced' ;; + paused) status_line='paused: awaiting an external dependency' ;; + esac + printf '%s\n' "$status_line" > "$state/$task.status" + printf 'Working...\n' > "$pane" + key=$(printf '%s' "$task" | tr ':/.' '___') + echo $(( $(date +%s) - 500 )) > "$state/.subsuper-stale-$key" + [ "$case_name" = prior-terminal ] \ + && printf '%s' "$status_line" > "$state/.subsuper-seen-status-$key" + [ "$case_name" = paused ] \ + && echo $(( $(date +%s) - 500 )) > "$state/.subsuper-paused-$key" + + ( + kill() { printf 'kill %s\n' "$*" >> "$action_log"; } + fm_backend_send_text_submit() { printf 'interrupt %s\n' "$*" >> "$action_log"; } + LOG="$dir/daemon.log" FM_STATE_OVERRIDE="$state" handle_wake "$reason" "$state" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$win" FM_FAKE_TMUX_CAPTURE="$pane" \ + FM_STATE_OVERRIDE="$state" FM_ESCALATE_BATCH_SECS=999999 housekeeping "$state" + ) + [ "$(wc -l < "$state/.subsuper-escalations" | tr -d ' ')" = 1 ] \ + || fail "$case_name enriched wedge did not produce exactly one escalation" + grep -F "${reason#stale: }" "$state/.subsuper-escalations" >/dev/null \ + || fail "$case_name enriched wedge lost its demand-deep-inspection detail" + [ ! -e "$state/.subsuper-stale-$key" ] \ + || fail "$case_name enriched wedge retained ordinary stale tracking" + case "$case_name" in + paused) [ -e "$state/.subsuper-paused-$key" ] \ + || fail "paused enriched wedge erased ordinary pause tracking" ;; + *) [ ! -e "$state/.subsuper-paused-$key" ] \ + || fail "$case_name enriched wedge created pause tracking" ;; + esac + [ ! -s "$action_log" ] \ + || fail "$case_name enriched wedge interrupted or killed the busy worker" + done + pass "enriched stale wedges bypass status absorption without disturbing busy workers" +} + test_stale_terminal_escalates() { local dir state out dir=$(make_supercase stale-terminal) @@ -803,23 +854,26 @@ test_pane_input_pending_blank_is_not_pending() { pass "pane_input_pending: blank cursor line is not pending" } -test_pane_input_pending_idle_prompt_not_pending() { - local dir state fakebin capture +test_pane_input_pending_requires_proven_empty_prompt() { + local dir state fakebin capture prompt dir=$(make_supercase pending-prompt) state="$dir/state" fakebin="$dir/fakebin" capture="$dir/pane.txt" - # Cursor line (line 3, cursor_y=2) is a bare prompt ($) → idle → not pending. - printf 'output\noutput\n$ \n' > "$capture" - PATH="$fakebin:$PATH" FM_FAKE_TMUX_CAPTURE="$capture" FM_FAKE_TMUX_CURSOR_Y=2 \ - pane_input_pending "fakepane" \ - && fail "bare prompt falsely detected as pending" - # Bare > prompt also idle. - printf 'output\noutput\n> \n' > "$capture" - PATH="$fakebin:$PATH" FM_FAKE_TMUX_CAPTURE="$capture" FM_FAKE_TMUX_CURSOR_Y=2 \ - pane_input_pending "fakepane" \ - && fail "bare > prompt falsely detected as pending" - pass "pane_input_pending: bare prompts are not pending (idle)" + for prompt in '$' '>'; do + printf 'output\noutput\n%s \n' "$prompt" > "$capture" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_CAPTURE="$capture" FM_FAKE_TMUX_CURSOR_Y=2 \ + pane_input_pending "fakepane" \ + || fail "bare shell prompt '$prompt' should defer as unknown" + done + for prompt in '❯' '›'; do + printf 'output\noutput\n%s \n' "$prompt" > "$capture" + if PATH="$fakebin:$PATH" FM_FAKE_TMUX_CAPTURE="$capture" FM_FAKE_TMUX_CURSOR_Y=2 \ + pane_input_pending "fakepane"; then + fail "proven empty agent prompt '$prompt' should not defer" + fi + done + pass "pane_input_pending: only proven empty agent prompts pass" } # The safety fix at the tmux classifier (task fm-composer-shellglyph-safety): a @@ -848,8 +902,8 @@ test_tmux_composer_state_bordered_and_agent_rows_are_empty() { local dir fakebin capture out dir=$(make_supercase composer-empty-agent) fakebin="$dir/fakebin"; capture="$dir/pane.txt" - printf '%s\n' "│ > │" > "$capture" - out=$(PATH="$fakebin:$PATH" FM_FAKE_TMUX_CAPTURE="$capture" FM_FAKE_TMUX_CURSOR_Y=0 \ + printf '╭────────────────────────╮\n│ > │\n╰────────────────────────╯\n' > "$capture" + out=$(PATH="$fakebin:$PATH" FM_FAKE_TMUX_CAPTURE="$capture" FM_FAKE_TMUX_CURSOR_Y=1 \ fm_tmux_composer_state "fakepane") [ "$out" = empty ] || fail "a bordered '│ > │' composer should read empty, got '$out'" printf '%s\n' "❯ " > "$capture" @@ -883,8 +937,8 @@ test_pane_input_pending_honors_idle_override_after_border_strip() { state="$dir/state" fakebin="$dir/fakebin" capture="$dir/pane.txt" - printf '│ custom idle> │\n' > "$capture" - PATH="$fakebin:$PATH" FM_FAKE_TMUX_CAPTURE="$capture" FM_FAKE_TMUX_CURSOR_Y=0 \ + printf '╭────────────────╮\n│ custom idle> │\n╰────────────────╯\n' > "$capture" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_CAPTURE="$capture" FM_FAKE_TMUX_CURSOR_Y=1 \ FM_COMPOSER_IDLE_RE='^custom idle>$' pane_input_pending "fakepane" \ && fail "FM_COMPOSER_IDLE_RE was not applied after border stripping" pass "pane_input_pending honors FM_COMPOSER_IDLE_RE after border stripping" @@ -991,13 +1045,13 @@ test_pane_input_pending_bordered_idle_not_pending() { local dir state fakebin capture line dir=$(make_supercase pending-bordered-idle) state="$dir/state"; fakebin="$dir/fakebin"; capture="$dir/pane.txt" - for line in \ - "│ > │" \ - "│ ❯ │" \ - "│ > │" \ - "│ │"; do - printf '%s\n' "$line" > "$capture" - if PATH="$fakebin:$PATH" FM_FAKE_TMUX_CAPTURE="$capture" FM_FAKE_TMUX_CURSOR_Y=0 \ + for line in '>' '❯' ''; do + case "$line" in + '>') printf '╭────────────╮\n│ > │\n╰────────────╯\n' > "$capture" ;; + '❯') printf '╭────────────╮\n│ ❯ │\n╰────────────╯\n' > "$capture" ;; + '') printf '╭────────────╮\n│ │\n╰────────────╯\n' > "$capture" ;; + esac + if PATH="$fakebin:$PATH" FM_FAKE_TMUX_CAPTURE="$capture" FM_FAKE_TMUX_CURSOR_Y=1 \ pane_input_pending "fakepane"; then fail "bordered idle composer falsely detected as pending: <$line>" fi @@ -1012,8 +1066,8 @@ test_pane_input_pending_bordered_with_text_is_pending() { local dir state fakebin capture dir=$(make_supercase pending-bordered-text) state="$dir/state"; fakebin="$dir/fakebin"; capture="$dir/pane.txt" - printf '%s\n' "│ > fix findings 1 and 3, skip 2 │" > "$capture" - PATH="$fakebin:$PATH" FM_FAKE_TMUX_CAPTURE="$capture" FM_FAKE_TMUX_CURSOR_Y=0 \ + printf '╭────────────────────────────────────────────────╮\n│ > fix findings 1 and 3, skip 2 │\n╰────────────────────────────────────────────────╯\n' > "$capture" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_CAPTURE="$capture" FM_FAKE_TMUX_CURSOR_Y=1 \ pane_input_pending "fakepane" \ || fail "real text inside a bordered composer was not detected as pending" pass "pane_input_pending: text inside a bordered composer is still pending" @@ -1055,7 +1109,7 @@ test_max_defer_empty_swallow_types_once_and_alarms() { dir=$(make_bordered_case maxdefer-stuck) state="$dir/state"; fakebin="$dir/fakebin" sent="$dir/sent.log"; : > "$sent" - printf '│ > │\n' > "$dir/composer" + printf '╭─────╮\n│ > │\n╰─────╯\n' > "$dir/composer" touch "$dir/.swallow" escalate_add "$state" "needs-decision: pick A" echo $(( $(date +%s) - 600 )) > "$state/.subsuper-escalations.since" @@ -1077,7 +1131,7 @@ test_max_defer_flushes_empty_idle_pane() { dir=$(make_bordered_case maxdefer-recover) state="$dir/state"; fakebin="$dir/fakebin" sent="$dir/sent.log"; : > "$sent" - printf '│ > │\n' > "$dir/composer" + printf '╭─────╮\n│ > │\n╰─────╯\n' > "$dir/composer" escalate_add "$state" "done: PR https://x/y/pull/1" echo $(( $(date +%s) - 600 )) > "$state/.subsuper-escalations.since" afk_enter "$state" @@ -1094,7 +1148,7 @@ test_max_defer_pending_composer_alarms_without_typing() { dir=$(make_bordered_case maxdefer-pending-digest) state="$dir/state"; fakebin="$dir/fakebin" sent="$dir/sent.log"; : > "$sent" - printf '│ > human draft │\n' > "$dir/composer" + printf '╭─────────────────╮\n│ > human draft │\n╰─────────────────╯\n' > "$dir/composer" escalate_add "$state" "needs-decision: pick B" echo $(( $(date +%s) - 600 )) > "$state/.subsuper-escalations.since" afk_enter "$state" @@ -1504,7 +1558,7 @@ test_fm_send_exits_nonzero_on_confirmed_swallow() { FM_SEND_SLEEP=0.05 "$ROOT/bin/fm-send.sh" sess:win 'route this work' >/dev/null 2>"$err" \ || fail "fm-send exited non-zero on a clean submit: $(cat "$err")" # Persistent swallow -> exit non-zero with a clear message. - printf '│ > │\n' > "$dir/composer" + printf '╭─────╮\n│ > │\n╰─────╯\n' > "$dir/composer" touch "$dir/.swallow" if PATH="$fakebin:$PATH" FM_HOME="$dir" FM_STATE_OVERRIDE="$dir/state" FM_FAKE_COMPOSER="$dir/composer" \ FM_FAKE_SWALLOW="$dir/.swallow" FM_FAKE_PERSIST_SWALLOW=1 FM_SEND_SLEEP=0.05 \ @@ -1528,6 +1582,21 @@ test_fm_send_exits_nonzero_on_initial_send_failure() { pass "fm-send exits non-zero when initial text send fails" } +test_fm_send_exits_nonzero_on_unproven_submit() { + local dir fakebin err + dir=$(make_bordered_case send-unproven) + fakebin="$dir/fakebin"; err="$dir/send.err" + touch "$dir/.swallow" + if PATH="$fakebin:$PATH" FM_HOME="$dir" FM_STATE_OVERRIDE="$dir/state" FM_FAKE_COMPOSER="$dir/composer" \ + FM_FAKE_SWALLOW="$dir/.swallow" FM_FAKE_PERSIST_SWALLOW=1 FM_SEND_SLEEP=0.05 \ + "$ROOT/bin/fm-send.sh" sess:win '修复' >/dev/null 2>"$err"; then + fail "fm-send exited zero when submit proof remained pending-unproven" + fi + grep -F 'verdict=pending-unproven' "$err" >/dev/null \ + || fail "fm-send did not preserve the unproven-submit verdict: $(cat "$err")" + pass "fm-send exits non-zero unless delivery is proven empty" +} + # --- herdr backend-awareness (fm-turnend-guard-h6-adjacent transport fix) ---- # Discovery, busy/pending dispatch, and the full inject_msg guard chain must # work through the herdr backend, not just tmux. Env-var prefix assignments @@ -1625,6 +1694,11 @@ test_pane_input_pending_herdr_dispatch() { fail "pane_input_pending should report not-pending for an empty herdr composer" fi ) || fail "herdr pane_input_pending (empty case) subshell failed" + ( + fm_backend_composer_state() { printf 'future-state'; } + pane_input_pending "default:w1:p2" herdr \ + || fail "pane_input_pending should defer on an unrecognized composer state" + ) || fail "herdr pane_input_pending (future-state case) subshell failed" pass "pane_input_pending: dispatches through fm_backend_composer_state for backend=herdr" } @@ -1724,6 +1798,24 @@ test_inject_msg_defers_on_dead_shell_unknown() { pass "inject_msg: defers on a dead-shell/unreadable composer (unknown), never typing the escalation into a shell" } +test_inject_msg_defers_on_unrecognized_composer_state() { + local dir state + dir=$(make_supercase inject-future-composer-state) + state="$dir/state" + afk_enter "$state" + ( + fm_backend_target_exists() { return 0; } + fm_backend_busy_state() { printf 'idle'; } + fm_backend_capture() { printf 'idle prompt\n'; } + fm_backend_composer_state() { printf 'future-state'; } + fm_backend_send_text_submit() { fail "send_text_submit must not run for an unrecognized composer state"; } + if FM_SUPERVISOR_BACKEND=herdr FM_SUPERVISOR_TARGET="default:w1:p2" inject_msg "hello" "$state"; then + fail "inject_msg should defer on an unrecognized composer state" + fi + ) || fail "unrecognized composer-state inject_msg subshell failed" + pass "inject_msg: unrecognized composer states defer by default" +} + test_afk_start_refuses_when_flag_cannot_be_written test_afk_start_ignores_stale_pidfile_without_lock test_afk_start_reclaims_stale_daemon_lock_reused_pid @@ -1732,6 +1824,7 @@ test_classify_routine_signal_self test_classify_terminal_signal_escalates test_classify_check_and_unknown_escalate test_stale_transient_self_records_marker +test_stale_diagnostic_wedge_survives_busy_housekeeping test_stale_terminal_escalates test_stale_paused_classifies_pause test_handle_wake_paused_records_pause_marker @@ -1768,7 +1861,7 @@ test_should_exit_afk_when_afk_inactive test_strip_injection_marker test_pane_input_pending_detects_partial_input test_pane_input_pending_blank_is_not_pending -test_pane_input_pending_idle_prompt_not_pending +test_pane_input_pending_requires_proven_empty_prompt test_tmux_composer_state_bare_shell_is_unknown test_tmux_composer_state_bordered_and_agent_rows_are_empty test_tmux_composer_state_requires_matching_box_borders @@ -1809,6 +1902,7 @@ test_inject_wedge_alarm_fires_active_alert_on_non_tmux_backend test_inject_wedge_alarm_throttles_when_marker_cannot_be_written test_fm_send_exits_nonzero_on_confirmed_swallow test_fm_send_exits_nonzero_on_initial_send_failure +test_fm_send_exits_nonzero_on_unproven_submit test_discover_supervisor_backend_precedence test_discover_supervisor_target_herdr test_pane_is_busy_herdr_native_busy_state @@ -1821,3 +1915,4 @@ test_inject_msg_herdr_composer_guard_defers test_inject_msg_herdr_pane_gone_defers test_inject_msg_herdr_submits_through_backend_dispatch test_inject_msg_defers_on_dead_shell_unknown +test_inject_msg_defers_on_unrecognized_composer_state diff --git a/tests/fm-dispatch-select.test.sh b/tests/fm-dispatch-select.test.sh deleted file mode 100755 index fdcae4b1551..00000000000 --- a/tests/fm-dispatch-select.test.sh +++ /dev/null @@ -1,327 +0,0 @@ -#!/usr/bin/env bash -# Behavior tests for quota-aware crew-dispatch profile selection. -# -# End-user reproduction before the fix: -# - Initiating input: a non-empty profile array in rule use or top-level default. -# - Expected: startup accepts both locations and dispatch consults quota-axi. -# - Observed: startup rejected a default array, while a rule array without -# select silently chose its first element without calling quota-axi. -# - Masking condition: a single profile object worked, and an explicit -# quota-balanced rule array took the quota path. -# - Visible symptom: an actionable-looking startup invalid-config line for a -# valid default array, or first-profile dispatch despite better usable quota. -# - Earliest divergence: bootstrap restricted default to object while use had an -# array normalizer; selector gated quota lookup on explicit select rather than -# the already-normalized input being an array. -# - History: 7a42707 added rule arrays and explicit quota-balanced selection; -# 8cd90fe moved the contract owner without changing that asymmetric behavior. -# - Smallest counterfactual: adding select changed a rule array to quota-aware -# selection but could not make the same array valid under default. -# - Disconfirming evidence: object defaults, object rule uses, and explicit -# quota-balanced rule arrays all followed their proven paths successfully. -set -u - -# shellcheck source=tests/lib.sh -. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" - -BASE_PATH=${FM_TEST_BASE_PATH:-/usr/bin:/bin:/usr/sbin:/sbin} -TMP_ROOT=$(fm_test_tmproot fm-dispatch-select-tests) -mkdir -p "$TMP_ROOT" -RANDOM_ZERO="$TMP_ROOT/random-zero" -RANDOM_ONE="$TMP_ROOT/random-one" -printf '\000\000\000\000' > "$RANDOM_ZERO" -printf '\001\001\001\001' > "$RANDOM_ONE" - -write_quota() { - local file=$1 claude_status=$2 claude_five=$3 claude_week=$4 codex_status=$5 codex_five=$6 codex_week=$7 - mkdir -p "$(dirname "$file")" - cat > "$file" <<JSON -{ - "schemaVersion": 2, - "providers": [ - { - "provider": "claude", - "state": { "status": "$claude_status" }, - "windows": [ - { "id": "five_hour", "kind": "session", "percentRemaining": $claude_five }, - { "id": "seven_day", "kind": "weekly", "percentRemaining": $claude_week }, - { "id": "model:fable", "label": "Fable week", "kind": "model", "percentRemaining": 100 } - ] - }, - { - "provider": "codex", - "state": { "status": "$codex_status" }, - "windows": [ - { "id": "five_hour", "kind": "session", "percentRemaining": $codex_five }, - { "id": "weekly", "kind": "weekly", "percentRemaining": $codex_week }, - { "id": "model:codex_bengalfox:5h", "label": "GPT-5.3-Codex-Spark session", "kind": "model", "percentRemaining": 100 } - ] - } - ] -} -JSON -} - -profiles='[{"harness":"claude","model":"claude-sonnet-5","effort":"high"},{"harness":"codex","model":"gpt-5.5","effort":"high"}]' - -assert_profile() { - local actual=$1 expected=$2 message=$3 - [ "$actual" = "$expected" ] || fail "$message, got: $actual" -} - -test_implicit_array_picks_higher_min_provider() { - local quota out err - quota="$TMP_ROOT/higher.json" - write_quota "$quota" fresh 80 30 fresh 70 60 - out=$(FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ZERO" "$ROOT/bin/fm-dispatch-select.sh" --quota-json "$quota" "$profiles" 2>"$TMP_ROOT/higher.err") - err=$(cat "$TMP_ROOT/higher.err") - assert_profile "$out" '{"harness":"codex","model":"gpt-5.5","effort":"high"}' "higher-min provider should win" - assert_contains "$err" "selection basis: quota-selected" "quota selection basis was not exposed" - pass "every profile array implicitly picks the least constrained scorable provider" -} - -test_rule_array_without_select_invokes_quota_axi() { - local fakebin marker out rule - fakebin=$(fm_fakebin "$TMP_ROOT/implicit-command") - marker="$TMP_ROOT/implicit-command/called" - cat > "$fakebin/quota-axi" <<SH -#!/usr/bin/env bash -printf '%s\n' "\$*" > '$marker' -cat <<'JSON' -{"schemaVersion":2,"providers":[{"provider":"claude","state":{"status":"fresh"},"windows":[{"id":"five_hour","kind":"session","percentRemaining":10}]},{"provider":"codex","state":{"status":"fresh"},"windows":[{"id":"five_hour","kind":"session","percentRemaining":90}]}]} -JSON -SH - chmod +x "$fakebin/quota-axi" - rule='{"when":"big work","use":[{"harness":"claude"},{"harness":"codex"}]}' - out=$(PATH="$fakebin:$BASE_PATH" FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ZERO" "$ROOT/bin/fm-dispatch-select.sh" "$rule" 2>/dev/null) - assert_profile "$out" '{"harness":"codex"}' "implicit rule array should use quota data" - assert_contains "$(cat "$marker")" "--json" "implicit array did not invoke quota-axi --json" - pass "rule arrays need no select property to invoke installed quota-axi" -} - -test_legacy_explicit_selector_stays_compatible() { - local quota out - quota="$TMP_ROOT/legacy.json" - write_quota "$quota" fresh 90 80 fresh 70 60 - out=$(FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ZERO" "$ROOT/bin/fm-dispatch-select.sh" --select quota-balanced --quota-json "$quota" "$profiles" 2>/dev/null) - assert_profile "$out" '{"harness":"claude","model":"claude-sonnet-5","effort":"high"}' "legacy explicit selector changed behavior" - - out=$(FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ZERO" "$ROOT/bin/fm-dispatch-select.sh" --quota-json "$quota" \ - '{"when":"big work","use":[{"harness":"claude"},{"harness":"codex"}],"select":"quota-balanced"}' 2>/dev/null) - assert_profile "$out" '{"harness":"claude"}' "legacy rule selector changed behavior" - pass "legacy select quota-balanced forms remain compatible" -} - -test_equal_winners_use_os_random_tie_break() { - local quota first second - quota="$TMP_ROOT/tie.json" - write_quota "$quota" fresh 90 50 fresh 60 50 - first=$(FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ZERO" "$ROOT/bin/fm-dispatch-select.sh" --quota-json "$quota" "$profiles" 2>/dev/null) - second=$(FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ONE" "$ROOT/bin/fm-dispatch-select.sh" --quota-json "$quota" "$profiles" 2>/dev/null) - assert_profile "$first" '{"harness":"claude","model":"claude-sonnet-5","effort":"high"}' "zero random fixture should choose first tie winner" - assert_profile "$second" '{"harness":"codex","model":"gpt-5.5","effort":"high"}' "nonzero random fixture should choose second tie winner" - pass "equal quota winners use the OS-backed random tie-break" -} - -test_provider_and_product_mapping_through_wrappers() { - local quota out - quota="$TMP_ROOT/routes.json" - cat > "$quota" <<'JSON' -{ - "schemaVersion": 2, - "providers": [ - {"provider":"claude","state":{"status":"fresh"},"windows":[{"id":"five_hour","kind":"session","percentRemaining":45},{"id":"seven_day","kind":"weekly","percentRemaining":40}]}, - {"provider":"codex","state":{"status":"fresh"},"windows":[{"id":"five_hour","kind":"session","percentRemaining":55},{"id":"weekly","kind":"weekly","percentRemaining":50}]}, - {"provider":"grok","state":{"status":"fresh"},"windows":[ - {"id":"credits","kind":"credits","percentRemaining":1}, - {"id":"product:api","kind":"credits","percentRemaining":75}, - {"id":"product:grok_build","kind":"credits","percentRemaining":25} - ]} - ] -} -JSON - out=$(FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ZERO" "$ROOT/bin/fm-dispatch-select.sh" --quota-json "$quota" \ - '[{"harness":"claude"},{"harness":"pi","model":"openai-codex/gpt-5.5"}]' 2>/dev/null) - assert_profile "$out" '{"harness":"pi","model":"openai-codex/gpt-5.5"}' "Pi OpenAI Codex route was not scored as Codex" - - out=$(FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ZERO" "$ROOT/bin/fm-dispatch-select.sh" --quota-json "$quota" \ - '[{"harness":"pi","model":"anthropic/claude-sonnet-5"},{"harness":"codex"}]' 2>/dev/null) - assert_profile "$out" '{"harness":"codex"}' "Pi Anthropic route was not scored as Claude" - - out=$(FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ZERO" "$ROOT/bin/fm-dispatch-select.sh" --quota-json "$quota" \ - '[{"harness":"pi","model":"xai/grok-4.5"},{"harness":"grok","model":"grok-4.5"}]' 2>/dev/null) - assert_profile "$out" '{"harness":"pi","model":"xai/grok-4.5"}' "Pi xAI API should use product:api rather than Grok Build or aggregate credits" - - out=$(FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ZERO" "$ROOT/bin/fm-dispatch-select.sh" --quota-json "$quota" \ - '[{"harness":"grok"},{"harness":"claude"}]' 2>/dev/null) - assert_profile "$out" '{"harness":"claude"}' "direct Grok should use product:grok_build" - pass "direct and Pi-wrapped candidates map to consumed Claude, Codex, xAI API, and Grok Build quota" -} - -test_most_constrained_relevant_window_scores_candidate() { - local quota out - quota="$TMP_ROOT/scoped.json" - cat > "$quota" <<'JSON' -{"schemaVersion":2,"providers":[ - {"provider":"claude","state":{"status":"fresh"},"windows":[ - {"id":"five_hour","kind":"session","percentRemaining":90}, - {"id":"seven_day","kind":"weekly","percentRemaining":80}, - {"id":"model:fable","label":"Fable week","kind":"model","percentRemaining":5} - ]}, - {"provider":"codex","state":{"status":"fresh"},"windows":[ - {"id":"five_hour","kind":"session","percentRemaining":30}, - {"id":"weekly","kind":"weekly","percentRemaining":30}, - {"id":"code_review_five_hour","label":"code review session","kind":"session","percentRemaining":1}, - {"id":"code_review_weekly","label":"code review week","kind":"weekly","percentRemaining":1}, - {"id":"model:other:5h","label":"Unrelated preview session","kind":"model","percentRemaining":1} - ]} -]} -JSON - out=$(FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ZERO" "$ROOT/bin/fm-dispatch-select.sh" --quota-json "$quota" \ - '[{"harness":"claude","model":"claude-fable-5"},{"harness":"codex","model":"gpt-5.5"}]' 2>/dev/null) - assert_profile "$out" '{"harness":"codex","model":"gpt-5.5"}' "matching model window was not included or unrelated model window was included" - pass "candidate score uses its most constrained general or matching model quota window" -} - -test_grok_aggregate_fallback_requires_no_product_windows() { - local quota out - quota="$TMP_ROOT/grok-partial-products.json" - cat > "$quota" <<'JSON' -{"schemaVersion":2,"providers":[ - {"provider":"grok","state":{"status":"fresh"},"windows":[ - {"id":"credits","kind":"credits","percentRemaining":100}, - {"id":"product:grok_build","kind":"credits","percentRemaining":90} - ]}, - {"provider":"claude","state":{"status":"fresh"},"windows":[ - {"id":"five_hour","kind":"session","percentRemaining":5}, - {"id":"seven_day","kind":"weekly","percentRemaining":5} - ]} -]} -JSON - out=$(FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ZERO" "$ROOT/bin/fm-dispatch-select.sh" --quota-json "$quota" \ - '[{"harness":"pi","model":"xai/grok-4.5"},{"harness":"claude"}]' 2>/dev/null) - assert_profile "$out" '{"harness":"claude"}' "xAI API route used aggregate credits despite exposed product windows" - pass "Grok aggregate credits are used only when product windows are absent" -} - -test_stale_cache_needs_clear_margin_to_beat_fresh() { - local quota out - quota="$TMP_ROOT/stale-margin.json" - write_quota "$quota" stale 85 70 fresh 65 60 - out=$(FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ZERO" "$ROOT/bin/fm-dispatch-select.sh" --quota-json "$quota" "$profiles" 2>/dev/null) - assert_profile "$out" '{"harness":"codex","model":"gpt-5.5","effort":"high"}' "fresh provider should win below stale margin" - - write_quota "$quota" stale 90 85 fresh 65 60 - out=$(FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ZERO" "$ROOT/bin/fm-dispatch-select.sh" --quota-json "$quota" "$profiles" 2>/dev/null) - assert_profile "$out" '{"harness":"claude","model":"claude-sonnet-5","effort":"high"}' "stale provider should win after clearing margin" - pass "stale cached quota retains the documented freshness margin" -} - -test_partial_quota_data_prefers_scorable_candidate() { - local quota out - quota="$TMP_ROOT/partial.json" - cat > "$quota" <<'JSON' -{"schemaVersion":2,"providers":[{"provider":"codex","state":{"status":"fresh"},"windows":[{"id":"five_hour","kind":"session","percentRemaining":4}]}]} -JSON - out=$(FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ZERO" "$ROOT/bin/fm-dispatch-select.sh" --quota-json "$quota" "$profiles" 2>/dev/null) - assert_profile "$out" '{"harness":"codex","model":"gpt-5.5","effort":"high"}' "unscorable first candidate beat usable Codex data" - pass "partial quota data picks the best scorable candidate instead of an unscorable candidate" -} - -assert_random_fallback_chooses_second() { - local out_file=$1 err_file=$2 - shift 2 - FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ONE" "$@" >"$out_file" 2>"$err_file" - assert_profile "$(cat "$out_file")" '{"harness":"codex","model":"gpt-5.5","effort":"high"}' "random fallback fixture should choose the second candidate" - assert_contains "$(cat "$err_file")" "selection basis: random fallback" "random fallback basis was not exposed" -} - -test_operational_quota_failures_use_uniform_random_fallback() { - local fakebin quota - fakebin=$(fm_fakebin "$TMP_ROOT/missing") - assert_random_fallback_chooses_second "$TMP_ROOT/missing.out" "$TMP_ROOT/missing.err" \ - env PATH="$fakebin:$BASE_PATH" "$ROOT/bin/fm-dispatch-select.sh" "$profiles" - assert_contains "$(cat "$TMP_ROOT/missing.err")" "quota-axi missing" "missing quota-axi reason was not logged" - - fakebin=$(fm_fakebin "$TMP_ROOT/error") - cat > "$fakebin/quota-axi" <<'SH' -#!/usr/bin/env bash -exit 42 -SH - chmod +x "$fakebin/quota-axi" - assert_random_fallback_chooses_second "$TMP_ROOT/error.out" "$TMP_ROOT/error.err" \ - env PATH="$fakebin:$BASE_PATH" "$ROOT/bin/fm-dispatch-select.sh" "$profiles" - assert_contains "$(cat "$TMP_ROOT/error.err")" "quota-axi exited 42" "quota-axi error reason was not logged" - - quota="$TMP_ROOT/bad.json" - printf '%s\n' not-json > "$quota" - assert_random_fallback_chooses_second "$TMP_ROOT/bad.out" "$TMP_ROOT/bad.err" \ - "$ROOT/bin/fm-dispatch-select.sh" --quota-json "$quota" "$profiles" - assert_contains "$(cat "$TMP_ROOT/bad.err")" "unparseable JSON" "bad quota JSON reason was not logged" - - printf '%s\n' '{"schemaVersion":2,"providers":[]}' > "$quota" - assert_random_fallback_chooses_second "$TMP_ROOT/empty.out" "$TMP_ROOT/empty.err" \ - "$ROOT/bin/fm-dispatch-select.sh" --quota-json "$quota" "$profiles" - assert_contains "$(cat "$TMP_ROOT/empty.err")" "no usable quota windows" "wholly unusable quota reason was not logged" - pass "missing, failed, malformed, and wholly unusable quota data use OS-backed random fallback" -} - -test_single_profile_and_one_element_array() { - local fakebin marker out err - fakebin=$(fm_fakebin "$TMP_ROOT/single") - marker="$TMP_ROOT/single/called" - cat > "$fakebin/quota-axi" <<SH -#!/usr/bin/env bash -printf called > '$marker' -exit 1 -SH - chmod +x "$fakebin/quota-axi" - - out=$(PATH="$fakebin:$BASE_PATH" "$ROOT/bin/fm-dispatch-select.sh" '{"harness":"grok","model":"grok-4.5","effort":"high"}' 2>/dev/null) - assert_profile "$out" '{"harness":"grok","model":"grok-4.5","effort":"high"}' "single profile object should resolve to itself" - [ ! -e "$marker" ] || fail "single profile object should not invoke quota-axi" - - out=$(PATH="$fakebin:$BASE_PATH" FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ONE" "$ROOT/bin/fm-dispatch-select.sh" \ - '[{"harness":"grok","model":"grok-4.5","effort":"high"}]' 2>"$TMP_ROOT/one.err") - err=$(cat "$TMP_ROOT/one.err") - assert_profile "$out" '{"harness":"grok","model":"grok-4.5","effort":"high"}' "one-element array should remain selectable" - [ -e "$marker" ] || fail "one-element array should invoke quota-axi" - assert_contains "$err" "selection basis: random fallback" "one-element operational fallback basis was not logged" - pass "single objects remain backward compatible and one-element arrays remain quota-aware" -} - -test_malformed_profile_arrays_are_validation_errors() { - local body expect out status n - n=0 - while IFS='^' read -r body expect; do - n=$((n + 1)) - out=$(FM_DISPATCH_RANDOM_SOURCE="$RANDOM_ONE" "$ROOT/bin/fm-dispatch-select.sh" "$body" 2>&1) - status=$? - expect_code 2 "$status" "malformed profile array should exit 2" - assert_contains "$out" "$expect" "malformed profile array did not explain validation error" - assert_not_contains "$out" "random fallback" "malformed profile array incorrectly used operational fallback" - done <<'ROWS' -[]^must not be empty -["claude"]^must be an object -[{"model":"claude-sonnet-5"}]^needs a non-empty harness -[{"harness":"claude","model":3}]^model must be a non-empty string -[{"harness":"spaceship"}]^contains an unverified harness -[{"harness":"codex","effort":"max"}]^contains an unsupported harness/effort pair -ROWS - pass "malformed arrays stay actionable validation errors and never enter random fallback" -} - -test_implicit_array_picks_higher_min_provider -test_rule_array_without_select_invokes_quota_axi -test_legacy_explicit_selector_stays_compatible -test_equal_winners_use_os_random_tie_break -test_provider_and_product_mapping_through_wrappers -test_most_constrained_relevant_window_scores_candidate -test_grok_aggregate_fallback_requires_no_product_windows -test_stale_cache_needs_clear_margin_to_beat_fresh -test_partial_quota_data_prefers_scorable_candidate -test_operational_quota_failures_use_uniform_random_fallback -test_single_profile_and_one_element_array -test_malformed_profile_arrays_are_validation_errors - -echo "# all fm-dispatch-select tests passed" diff --git a/tests/fm-documentation-audiences.test.sh b/tests/fm-documentation-audiences.test.sh new file mode 100755 index 00000000000..90222802f6a --- /dev/null +++ b/tests/fm-documentation-audiences.test.sh @@ -0,0 +1,141 @@ +#!/usr/bin/env bash +# Structural regression tests for the tracked documentation audience inventory. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +CHECK="$ROOT/bin/fm-doc-audience-check.sh" +INVENTORY="$ROOT/docs/documentation-audiences.json" +TMP_ROOT=$(mktemp -d "${TMPDIR:-/tmp}/fm-doc-audiences.XXXXXX") +trap 'rm -rf "$TMP_ROOT"' EXIT + +run_expect_failure() { + local expected=$1 + shift + local out rc + set +e + out=$("$@" 2>&1) + rc=$? + set -e + [ "$rc" -ne 0 ] || fail "expected failure containing '$expected'" + assert_contains "$out" "$expected" "failure did not explain '$expected'" +} + +mutate_inventory() { + local source=$1 destination=$2 mode=$3 + python3 - "$source" "$destination" "$mode" <<'PY' +import json +import sys +from pathlib import Path + +source, destination, mode = map(Path, sys.argv[1:]) +data = json.loads(source.read_text(encoding="utf-8")) +if mode.name == "duplicate": + data["surfaces"].append(dict(data["surfaces"][0])) +elif mode.name == "bad-setup-audience": + for entry in data["surfaces"]: + if entry["path"] == "docs/tmux-backend.md": + entry["audience"] = "maintainer-verification" + break +elif mode.name == "missing-owner-pointer": + data["requiredOwnerPointers"][0] = { + "source": "README.md", + "target": "docs/sessionstart-nudge.md", + } +elif mode.name == "shrink-scope": + data["scope"]["trackedPatterns"] = ["README.md"] +else: + raise SystemExit(f"unknown mode: {mode.name}") +destination.write_text(json.dumps(data, indent=2) + "\n", encoding="utf-8") +PY +} + +test_repository_inventory_passes() { + local out + out=$("$CHECK") || fail "repository documentation audience check failed" + assert_contains "$out" "fm-doc-audience-check: ok surfaces=" \ + "audience check did not report exact surface coverage" + assert_contains "$out" "local_links=" \ + "audience check did not report local-link validation" + pass "documentation inventory classifies every maintained prose surface exactly once" +} + +test_duplicate_and_setup_classification_fail() { + local duplicate="$TMP_ROOT/duplicate.json" + local bad_setup="$TMP_ROOT/bad-setup.json" + local shrink_scope="$TMP_ROOT/shrink-scope.json" + mutate_inventory "$INVENTORY" "$duplicate" duplicate + mutate_inventory "$INVENTORY" "$bad_setup" bad-setup-audience + mutate_inventory "$INVENTORY" "$shrink_scope" shrink-scope + run_expect_failure "surfaces classified more than once" \ + "$CHECK" --inventory "$duplicate" + run_expect_failure "README setup target docs/tmux-backend.md has disallowed audience" \ + "$CHECK" --inventory "$bad_setup" + run_expect_failure "scope.trackedPatterns must match the fixed maintained-prose scope" \ + "$CHECK" --inventory "$shrink_scope" + pass "classification, setup routing, and maintained-prose scope fail safely" +} + +test_required_pointer_fails() { + local missing_pointer="$TMP_ROOT/missing-pointer.json" + mutate_inventory "$INVENTORY" "$missing_pointer" missing-owner-pointer + run_expect_failure "required owner pointer missing" \ + "$CHECK" --inventory "$missing_pointer" + pass "required documentation owner pointers cannot silently disappear" +} + +write_fixture_inventory() { + local repo=$1 + cat > "$repo/docs/documentation-audiences.json" <<'JSON' +{ + "version": 1, + "scope": {"trackedPatterns": ["*.md", "*.mdx", "*.rst", "*.txt", "docs/examples/*"]}, + "allowedAudiences": ["public-product", "operator-current", "maintainer-verification"], + "setupAudiences": ["public-product", "operator-current"], + "readmeSetupTargets": ["docs/setup.md"], + "requiredOwnerPointers": [ + {"source": "README.md", "target": "docs/policy.md"} + ], + "surfaces": [ + {"path": "README.md", "audience": "public-product"}, + {"path": "docs/evidence.md", "audience": "maintainer-verification"}, + {"path": "docs/policy.md", "audience": "operator-current"}, + {"path": "docs/setup.md", "audience": "operator-current"} + ] +} +JSON +} + +test_local_links_and_no_keyword_heuristic() { + local repo="$TMP_ROOT/fixture" + mkdir -p "$repo/docs" + git -C "$repo" init -q + printf '%s\n' '[Setup](docs/setup.md) [Policy](docs/policy.md)' > "$repo/README.md" + printf '%s\n' '# Setup' > "$repo/docs/setup.md" + printf '%s\n' '# Policy' > "$repo/docs/policy.md" + cat > "$repo/docs/evidence.md" <<'MD' +# Incident verification on 2026-07-23 + +```sh +/tmp/task-worktree/bin/tool --version +``` + +Observed version 1.2.3 on branch `fm/example`. +MD + write_fixture_inventory "$repo" + git -C "$repo" add README.md docs + "$CHECK" --root "$repo" >/dev/null \ + || fail "structural checker rejected legitimate maintainer evidence prose" + + printf '%s\n' '[Setup](docs/setup.md) [Policy](docs/policy.md) [Broken](docs/missing.bin)' \ + > "$repo/README.md" + git -C "$repo" add README.md + run_expect_failure "unresolved local link" "$CHECK" --root "$repo" + pass "local links resolve while dates, versions, commands, and incident prose remain semantically reviewed" +} + +test_repository_inventory_passes +test_duplicate_and_setup_classification_fail +test_required_pointer_fails +test_local_links_and_no_keyword_heuristic diff --git a/tests/fm-gate-refuse.test.sh b/tests/fm-gate-refuse.test.sh index f5b166566e4..aff57ab18e1 100755 --- a/tests/fm-gate-refuse.test.sh +++ b/tests/fm-gate-refuse.test.sh @@ -223,8 +223,10 @@ case "${1:-}" in done printf 'send-keys target=%s literal=%s arg=%s\n' "$target" "$literal" "${1:-}" >> "$FM_TMUX_LOG" exit 0 ;; - display-message) printf '%%1\n'; exit 0 ;; - capture-pane) printf '\xe2\x94\x82 \xe2\x94\x82\n'; exit 0 ;; + display-message) + for a in "$@"; do case "$a" in *cursor_y*) printf '1\n'; exit 0 ;; esac; done + printf '%%1\n'; exit 0 ;; + capture-pane) printf '╭────╮\n│ │\n╰────╯\n'; exit 0 ;; esac exit 0 SH @@ -316,7 +318,8 @@ SH git -C "$case_dir/wt" push -q origin fm/task-x1 git -C "$case_dir/project" fetch -q origin fm_write_meta "$case_dir/state/task-x1.meta" \ - "window=fm-task-x1" "worktree=$case_dir/wt" "project=$case_dir/project" \ + "window=firstmate:fm-task-x1" "endpoint_task_id=task-x1" \ + "worktree=$case_dir/wt" "project=$case_dir/project" \ "kind=ship" "mode=no-mistakes" touch "$case_dir/state/.last-watcher-beat" printf '%s\n' "$case_dir" @@ -358,36 +361,6 @@ test_teardown_refuses_and_admits() { pass "fm-teardown: refuses on marker and gate-worktree backstop; a normal teardown is unaffected" } -# --- tracked .no-mistakes.yaml ---------------------------------------------- - -test_no_mistakes_yaml_disables_project_settings() { - local file="$ROOT/.no-mistakes.yaml" val tab - assert_present "$file" "tracked .no-mistakes.yaml is missing" - git -C "$ROOT" ls-files --error-unmatch .no-mistakes.yaml >/dev/null 2>&1 \ - || fail ".no-mistakes.yaml is not tracked by git" - - # Parse with a real YAML loader and assert the field is boolean true, so a - # malformed file or a stringy "true" fails where a naive grep would pass. - if command -v python3 >/dev/null 2>&1 && python3 -c 'import yaml' >/dev/null 2>&1; then - val=$(python3 -c 'import yaml,sys; print(yaml.safe_load(open(sys.argv[1])).get("disable_project_settings"))' "$file") \ - || fail ".no-mistakes.yaml did not parse as YAML (python3)" - [ "$val" = "True" ] || fail "disable_project_settings is not boolean true (python3 read: $val)" - elif command -v ruby >/dev/null 2>&1; then - ruby -ryaml -e 'exit((YAML.safe_load(File.read(ARGV[0]))["disable_project_settings"] == true) ? 0 : 1)' "$file" \ - || fail ".no-mistakes.yaml did not parse or disable_project_settings != true (ruby)" - else - # No YAML loader: fall back to a strict structural check - no tab indentation - # (YAML forbids it) and the top-level key mapped to the bare boolean true. - tab=$(printf '\t') - case "$(cat "$file")" in - *"$tab"*) fail ".no-mistakes.yaml uses a tab (invalid YAML indentation)" ;; - esac - grep -qxE 'disable_project_settings:[[:space:]]+true' "$file" \ - || fail "top-level 'disable_project_settings: true' not found in .no-mistakes.yaml" - fi - pass ".no-mistakes.yaml parses and sets disable_project_settings: true (trusted-only gate opt-out)" -} - test_helper_env_marker_refuses test_helper_empty_env_marker_refuses test_helper_path_backstop_refuses @@ -395,4 +368,3 @@ test_helper_normal_is_noop test_spawn_refuses_and_admits test_send_refuses_and_admits test_teardown_refuses_and_admits -test_no_mistakes_yaml_disables_project_settings diff --git a/tests/fm-gitignore-config.test.sh b/tests/fm-gitignore-config.test.sh new file mode 100755 index 00000000000..5b864dd6466 --- /dev/null +++ b/tests/fm-gitignore-config.test.sh @@ -0,0 +1,47 @@ +#!/usr/bin/env bash +# .gitignore must ignore config/ as a directory, not by exact filename. +# +# A name-by-name list silently stops ignoring any new or home-local file under +# config/ (fm-gitignore-config-name-by-name): an unrecognized file there makes +# the working tree read as dirty, which then blocks guarded sync paths that +# refuse to touch a dirty home. +set -u + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" + +fail() { + printf 'not ok - %s\n' "$1" >&2 + exit 1 +} + +pass() { + printf 'ok - %s\n' "$1" +} + +random_leaf() { + printf '%s-%s' "$1" "$$-$RANDOM-$RANDOM" +} + +test_config_dir_ignored_as_category() { + local direct nested sample + direct="$(random_leaf config/unlisted-key)" + nested="config/$(random_leaf nested-dir)/$(random_leaf deep-file)" + for sample in "$direct" "$nested" config/some-new-key.admin; do + git -C "$ROOT" check-ignore -q "$sample" \ + || fail "git does not ignore $sample (config/ must be ignored as a directory)" + done + pass "config/ is ignored as a directory, covering unlisted and nested paths" +} + +test_unrelated_path_stays_visible() { + # Control: a path outside config/ must remain visible to Git, so the + # coverage above is proven by contrast rather than an always-ignoring rule. + local sibling + sibling="$(random_leaf not-config)" + git -C "$ROOT" check-ignore -q "$sibling" \ + && fail "git unexpectedly ignores $sibling (outside config/)" + pass "an unrelated path outside config/ remains visible to git" +} + +test_config_dir_ignored_as_category +test_unrelated_path_stays_visible diff --git a/tests/fm-gotmp.test.sh b/tests/fm-gotmp.test.sh index 03f0afafa7c..2fc7c8d78cc 100755 --- a/tests/fm-gotmp.test.sh +++ b/tests/fm-gotmp.test.sh @@ -5,10 +5,10 @@ # gotmp/, exports GOTMPDIR into the crewmate pane, and records tasktmp= in the task's # meta. fm-teardown reads tasktmp= and removes the whole root on cleanup. # -# These tests exercise behavior directly: fm-teardown is run as a subprocess against a -# fake FM_HOME/FM_ROOT (built so the real script resolves into it), with stub helper scripts. -# Nothing is sourced. The fm-spawn side is verified both structurally (the source has -# the contract lines) and behaviorally (the mkdir + meta-write pattern it uses). +# These tests exercise fm-teardown directly as a subprocess against a fake FM_HOME/FM_ROOT +# built so the real script resolves into it, with stub helper scripts. +# The isolated fm-spawn subprocess in fm-kimi-harness.test.sh covers temp-root creation, +# metadata publication, and the pane environment export. set -u # This suite does not source tests/lib.sh, so exempt its teardown subprocess from @@ -18,7 +18,6 @@ set -u export FM_GATE_REFUSE_BYPASS=1 ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" -SPAWN="$ROOT/bin/fm-spawn.sh" TEARDOWN="$ROOT/bin/fm-teardown.sh" fail() { @@ -96,40 +95,6 @@ META printf '%s' "$fake" } -# --- fm-spawn side --- - -test_spawn_contract_and_mkdir_pattern() { - # Structural: fm-spawn must create the gotmp dir, record tasktmp in meta, and export - # GOTMPDIR into the pane. Assert the contract lines are present in the source. - # shellcheck disable=SC2016 # single quotes are deliberate: these are literal source strings - grep -F 'mkdir -p "$TASK_TMP/gotmp"' "$SPAWN" >/dev/null \ - || fail "fm-spawn missing: mkdir of gotmp under TASK_TMP" - # shellcheck disable=SC2016 # single quotes are deliberate: literal source string - grep -F 'echo "tasktmp=$TASK_TMP"' "$SPAWN" >/dev/null \ - || fail "fm-spawn missing: tasktmp= line in meta write" - grep -F 'export GOTMPDIR=' "$SPAWN" >/dev/null \ - || fail "fm-spawn missing: GOTMPDIR export into pane" - # Behavioral: the mkdir + meta-write pattern spawn uses must produce a gotmp dir and - # a meta line whose value the teardown grep (tasktmp=, cut -d= -f2-) reads back whole. - local id=spawn-sim-z1 - local sim_root="$TMP_ROOT/$id-root" - local task_tmp="$sim_root/tmp/fm-$id" - mkdir -p "$sim_root/state" - # Replicate spawn's exact mkdir + meta-write lines. - TASK_TMP="$task_tmp" - mkdir -p "$TASK_TMP/gotmp" - { - echo "tasktmp=$TASK_TMP" - } > "$sim_root/state/$id.meta" - [ -d "$task_tmp/gotmp" ] || fail "simulated spawn did not create gotmp dir" - # Teardown reads tasktmp= with `grep '^tasktmp=' | cut -d= -f2-`; round-trip it. - local read_back - read_back=$(grep '^tasktmp=' "$sim_root/state/$id.meta" | cut -d= -f2-) - [ "$read_back" = "$task_tmp" ] \ - || fail "tasktmp value not round-tripped by teardown's grep|cut (got '$read_back')" - pass "fm-spawn creates gotmp dir and records tasktmp in meta" -} - # --- fm-teardown side (real subprocess) --- test_teardown_removes_tasktmp_dir() { @@ -207,7 +172,6 @@ test_teardown_skips_gracefully_when_dir_missing() { pass "fm-teardown skips gracefully when tasktmp= points to a nonexistent dir" } -test_spawn_contract_and_mkdir_pattern test_teardown_removes_tasktmp_dir test_teardown_skips_gracefully_without_tasktmp test_teardown_skips_gracefully_when_dir_missing diff --git a/tests/fm-grok-stop-live-e2e.test.sh b/tests/fm-grok-stop-live-e2e.test.sh new file mode 100755 index 00000000000..c9d4bcc9a64 --- /dev/null +++ b/tests/fm-grok-stop-live-e2e.test.sh @@ -0,0 +1,219 @@ +#!/usr/bin/env bash +# Opt-in real-process Grok Stop compatibility matrix. +# +# Requires exact official binary paths for one native-capable build and one +# genuine pre-native build. Each cell gets a unique scratch repo/home, dedicated +# tmux socket, socket-bound wrapper, target window, and independent control +# window. Cleanup uses only creation-time pane/process identities. +set -u + +if [ "${FM_GROK_STOP_LIVE_E2E:-0}" != 1 ]; then + echo "skip: set FM_GROK_STOP_LIVE_E2E=1 with FM_GROK_NATIVE_BIN and FM_GROK_LEGACY_BIN" + exit 0 +fi + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +NATIVE_BIN=${FM_GROK_NATIVE_BIN:-} +LEGACY_BIN=${FM_GROK_LEGACY_BIN:-} +AUTH=${FM_GROK_AUTH_FILE:-$HOME/.grok/auth.json} +REAL_TMUX=$(command -v tmux || true) +ACTIVE_LAB= + +[ -x "$NATIVE_BIN" ] || fail "FM_GROK_NATIVE_BIN must be an exact executable path" +[ -x "$LEGACY_BIN" ] || fail "FM_GROK_LEGACY_BIN must be an exact executable path" +[ -f "$AUTH" ] || fail "FM_GROK_AUTH_FILE must name the already-managed auth artifact" +[ -n "$REAL_TMUX" ] || fail "tmux not found" +command -v jq >/dev/null 2>&1 || fail "jq not found" + +NATIVE_VERSION=$($NATIVE_BIN --version) +LEGACY_VERSION=$($LEGACY_BIN --version) + +cleanup_exact_cell() { + local lab=$1 session=$2 target=$3 pane_expected pid_expected pane_live pid_live windows + pane_expected=$(sed -n 's/^target-pane=//p' "$lab/target.identity") + pid_expected=$(sed -n 's/^target-pid=//p' "$lab/target.identity") + [ -n "$pane_expected" ] && [ -n "$pid_expected" ] || { + echo "blocked: missing recorded target identity; preserving $lab" >&2 + return 1 + } + pane_live=$(cd "$lab" && env -u TMUX -u TMUX_PANE "$REAL_TMUX" -S dedicated.sock \ + display-message -p -t "$session:$target" '#{pane_id}' 2>/dev/null) || { + echo "blocked: cannot verify target pane identity; preserving $lab" >&2 + return 1 + } + pid_live=$(cd "$lab" && env -u TMUX -u TMUX_PANE "$REAL_TMUX" -S dedicated.sock \ + display-message -p -t "$session:$target" '#{pane_pid}' 2>/dev/null) || { + echo "blocked: cannot verify target process identity; preserving $lab" >&2 + return 1 + } + [ "$pane_live" = "$pane_expected" ] && [ "$pid_live" = "$pid_expected" ] || { + echo "blocked: live target identity differs from creation record; preserving $lab" >&2 + return 1 + } + FM_E2E_TMUX_SOCKET_ID="$lab/dedicated.sock" PATH="$lab/bin:$PATH" \ + env -u TMUX -u TMUX_PANE "$lab/bin/tmux" kill-window -t "$session:$target" || return 1 + windows=$(cd "$lab" && env -u TMUX -u TMUX_PANE "$REAL_TMUX" -S dedicated.sock \ + list-windows -t "$session" -F '#{window_name}') || return 1 + printf '%s\n' "$windows" | grep -Fqx control || { + echo "blocked: control window did not survive exact target cleanup; preserving $lab" >&2 + return 1 + } + printf '%s\n' "$windows" | grep -Fqx "$target" && { + echo "blocked: exact target survived cleanup; preserving $lab" >&2 + return 1 + } + FM_E2E_TMUX_SOCKET_ID="$lab/dedicated.sock" PATH="$lab/bin:$PATH" \ + env -u TMUX -u TMUX_PANE "$lab/bin/tmux" kill-server || return 1 + rm -rf -- "$lab" + ACTIVE_LAB= +} + +preserve_on_failure() { + local rc=$? + if [ "$rc" -ne 0 ] && [ -n "$ACTIVE_LAB" ]; then + echo "blocked: preserving failed isolated Grok cell at $ACTIVE_LAB" >&2 + fi + exit "$rc" +} +trap preserve_on_failure EXIT + +run_cell() { # <native|legacy> <exact-binary> + local kind=$1 binary=$2 lab tmp_base session target prompt i child_count payload_count unique_sessions + local stop_values outer_turns outer_text + tmp_base=${TMPDIR:-/tmp} + tmp_base=${tmp_base%/} + lab=$(mktemp -d "$tmp_base/fm-grok-stop-$kind.XXXXXX") || return 1 + session="$kind-e2e" + target="fm-$kind" + ACTIVE_LAB=$lab + umask 077 + mkdir -p "$lab"/{home,grok-home,bin,fmhome/state,fmhome/config} + git clone -q --no-hardlinks "$ROOT" "$lab/project" || return 1 + # Before the candidate is committed, clone sees HEAD only. Apply the current + # tracked diff so this opt-in gate always exercises the code under review. + git -C "$ROOT" diff --binary HEAD -- > "$lab/candidate.patch" || return 1 + [ ! -s "$lab/candidate.patch" ] \ + || git -C "$lab/project" apply --whitespace=nowarn "$lab/candidate.patch" || return 1 + ln -s "$AUTH" "$lab/grok-home/auth.json" + printf '%s\n' "$lab/dedicated.sock" > "$lab/tmux.socket-identity" + + cat > "$lab/bin/tmux" <<EOF +#!/usr/bin/env bash +set -eu +expected='$lab/dedicated.sock' +[ -z "\${TMUX:-}" ] && [ -z "\${TMUX_PANE:-}" ] || exit 91 +[ "\${FM_E2E_TMUX_SOCKET_ID:-}" = "\$expected" ] || exit 92 +[ "\$(cat '$lab/tmux.socket-identity')" = "\$expected" ] || exit 93 +for arg in "\$@"; do case "\$arg" in -S|-L) exit 94 ;; esac; done +printf 'tmux' >> '$lab/tmux-wrapper.log' +printf ' <%s>' "\$@" >> '$lab/tmux-wrapper.log' +printf '\n' >> '$lab/tmux-wrapper.log' +cd '$lab' +exec '$REAL_TMUX' -S dedicated.sock "\$@" +EOF + cat > "$lab/bin/grok" <<'EOF' +#!/usr/bin/env bash +printf 'active=%s' "${GROK_TURNEND_GUARD_ACTIVE:-}" >> "${FM_GROK_E2E_ROOT:?}/resume-invocations.log" +printf ' <%s>' "$@" >> "${FM_GROK_E2E_ROOT:?}/resume-invocations.log" +printf '\n' >> "${FM_GROK_E2E_ROOT:?}/resume-invocations.log" +exec "${FM_GROK_E2E_BIN:?}" "$@" +EOF + chmod +x "$lab/bin/tmux" "$lab/bin/grok" + : > "$lab/tmux-wrapper.log" + : > "$lab/resume-invocations.log" + + if [ "$kind" = native ]; then + cat > "$lab/project/.grok/hooks/00-fm-stop-live-e2e.json" <<EOF +{"hooks":{"Stop":[{"hooks":[{"type":"command","command":"bash -lc 'cat >> \"$lab/payloads.jsonl\"'","timeout":30}]}]}} +EOF + prompt='This is an isolated regression test. Reply exactly NATIVE_BASE. If Stop-hook feedback arrives, do not use tools; acknowledge it by replying exactly NATIVE_CONTINUED, then stop.' + else + mkdir -p "$lab/grok-home/hooks" + cp "$ROOT/.grok/hooks/fm-primary-turnend-guard.json" "$lab/grok-home/hooks/fm-primary-turnend-guard.json" + rm -f "$lab/project/.grok/hooks/fm-primary-turnend-guard.json" + cat > "$lab/grok-home/hooks/00-fm-stop-live-e2e.json" <<EOF +{"hooks":{"Stop":[{"hooks":[{"type":"command","command":"bash -lc 'cat >> \"$lab/payloads.jsonl\"'","timeout":30}]}]}} +EOF + prompt='This is an isolated regression test. Reply exactly LEGACY_BASE. If a resumed turn receives guard feedback, do not use tools; acknowledge it by replying exactly LEGACY_RESUMED, then stop.' + fi + + cat > "$lab/fmhome/state/$kind.meta" <<EOF +window=$session:$target +endpoint_task_id=$kind +worktree=$lab/project +project=$lab/project +harness=grok +kind=scout +mode=no-mistakes +EOF + cat > "$lab/run.sh" <<EOF +#!/usr/bin/env bash +set -u +cd '$lab/project' || exit 70 +env -u TMUX -u TMUX_PANE HOME='$lab/home' GROK_HOME='$lab/grok-home' GROK_AGENT=1 \ + FM_HOME='$lab/fmhome' FM_ROOT_OVERRIDE='$lab/project' FM_GROK_E2E_ROOT='$lab' \ + FM_GROK_E2E_BIN='$binary' FM_E2E_TMUX_SOCKET_ID='$lab/dedicated.sock' PATH='$lab/bin':"\$PATH" \ + '$binary' $([ "$kind" = native ] && printf '%s' '--trust ')--always-approve --reasoning-effort low \ + --output-format json --leader-socket '$lab/leader.sock' -p $(printf '%q' "$prompt") \ + > '$lab/outer.json' 2> '$lab/outer.err' & +child=\$! +printf 'child_pid=%s\n' "\$child" > '$lab/process.identity' +pgid=\$(ps -o pgid= -p "\$child" 2>/dev/null | tr -d ' ') +printf 'child_pgid=%s\n' "\$pgid" >> '$lab/process.identity' +wait "\$child"; rc=\$? +printf 'rc=%s\n' "\$rc" > '$lab/done' +sleep 1800 +EOF + chmod +x "$lab/run.sh" + + ( cd "$lab" && env -u TMUX -u TMUX_PANE "$REAL_TMUX" -S dedicated.sock \ + new-session -d -s "$session" -n control 'sleep 1800' ) || return 1 + ( cd "$lab" && env -u TMUX -u TMUX_PANE "$REAL_TMUX" -S dedicated.sock \ + new-window -d -t "$session:" -n "$target" "$lab/run.sh" ) || return 1 + printf 'target-pane=%s\ntarget-pid=%s\n' \ + "$(cd "$lab" && env -u TMUX -u TMUX_PANE "$REAL_TMUX" -S dedicated.sock display-message -p -t "$session:$target" '#{pane_id}')" \ + "$(cd "$lab" && env -u TMUX -u TMUX_PANE "$REAL_TMUX" -S dedicated.sock display-message -p -t "$session:$target" '#{pane_pid}')" \ + > "$lab/target.identity" + + i=0 + while [ "$i" -lt "${FM_GROK_STOP_LIVE_TIMEOUT:-600}" ] && [ ! -f "$lab/done" ]; do + sleep 1 + i=$((i + 1)) + done + [ -f "$lab/done" ] || fail "$kind Grok cell timed out" + grep -qx 'rc=0' "$lab/done" || fail "$kind Grok process returned nonzero" + [ ! -s "$lab/tmux-wrapper.log" ] || fail "$kind model path invoked tmux unexpectedly" + payload_count=$(jq -s 'length' "$lab/payloads.jsonl") + unique_sessions=$(jq -s '[.[].sessionId] | unique | length' "$lab/payloads.jsonl") + [ "$unique_sessions" -eq 1 ] || fail "$kind Stop payloads changed session identity" + child_count=$(grep -c '^active=' "$lab/resume-invocations.log" || true) + outer_text=$(jq -r '.text // empty' "$lab/outer.json") + + if [ "$kind" = native ]; then + [ "$payload_count" -eq 2 ] || fail "native path expected two Stop payloads, got $payload_count" + stop_values=$(jq -sc '[.[].stopHookActive]' "$lab/payloads.jsonl") + [ "$stop_values" = '[false,true]' ] || fail "native capability sequence was $stop_values" + outer_turns=$(jq -r '.num_turns // 0' "$lab/outer.json") + [ "$outer_turns" -eq 2 ] || fail "native path expected two model turns, got $outer_turns" + [ "$child_count" -eq 0 ] || fail "native path started grok --resume" + case "$outer_text" in *NATIVE_BASE*NATIVE_CONTINUED*) ;; *) fail "native model did not receive guard feedback" ;; esac + printf 'ok - %s native Stop kept one session across false->true, two model turns, and zero resume processes\n' "$NATIVE_VERSION" + else + [ "$payload_count" -eq 2 ] || fail "legacy path expected two Stop payloads, got $payload_count" + jq -se 'all(.[]; (has("stopHookActive") | not) and (has("stop_hook_active") | not))' "$lab/payloads.jsonl" >/dev/null \ + || fail "legacy payload unexpectedly exposed native capability" + [ "$child_count" -eq 1 ] || fail "legacy path expected exactly one grok --resume, got $child_count" + grep -q '^active=1 ' "$lab/resume-invocations.log" || fail "legacy resume lacked the recursion guard" + case "$outer_text" in *LEGACY_BASE*) ;; *) fail "legacy outer turn did not finish normally" ;; esac + printf 'ok - %s legacy Stop omitted capability, resumed exactly once, and stopped normally\n' "$LEGACY_VERSION" + fi + + cleanup_exact_cell "$lab" "$session" "$target" || return 1 +} + +run_cell native "$NATIVE_BIN" +run_cell legacy "$LEGACY_BIN" +trap - EXIT +echo "ok - Grok adaptive Stop real-process matrix passed with exact target cleanup and control-window survival" diff --git a/tests/fm-herdr-lab.test.sh b/tests/fm-herdr-lab.test.sh index 6d9c14521e2..14ab7497a09 100755 --- a/tests/fm-herdr-lab.test.sh +++ b/tests/fm-herdr-lab.test.sh @@ -12,7 +12,7 @@ FAKE_LOG="$TMP_ROOT/herdr.log" TRIPWIRES="$TMP_ROOT/tripwires" REAL_SLEEP=$(command -v sleep) mkdir -p "$FAKE_STATE" -printf '%s\n' '/Users/test/.config/herdr/herdr.sock' > "$FAKE_STATE/default-socket" +printf '%s\n' '/home/test/.config/herdr/herdr.sock' > "$FAKE_STATE/default-socket" : > "$FAKE_LOG" cat > "$FAKEBIN/herdr" <<'SH' @@ -176,7 +176,7 @@ test_changed_default_trips_after_teardown() { run_with_fake fm_herdr_lab_teardown "$name" >/dev/null 2>&1 || status=$? expect_code 1 "$status" "changed default fleet state must fail teardown" assert_present "$TRIPWIRES/$name.fleet-state.json" "failed tripwire should retain evidence" - printf '%s\n' '/Users/test/.config/herdr/herdr.sock' > "$FAKE_STATE/default-socket" + printf '%s\n' '/home/test/.config/herdr/herdr.sock' > "$FAKE_STATE/default-socket" rm -f "$TRIPWIRES/$name.fleet-state.json" pass "fm-herdr-lab: changed default fleet state is a hard failure" } diff --git a/tests/fm-herdr-session-cleanup-e2e.test.sh b/tests/fm-herdr-session-cleanup-e2e.test.sh new file mode 100755 index 00000000000..7a4a49aa008 --- /dev/null +++ b/tests/fm-herdr-session-cleanup-e2e.test.sh @@ -0,0 +1,146 @@ +#!/usr/bin/env bash +# Real restored-shell E2E for home-local session-start Herdr projection cleanup. +# Every CLI operation is routed through one guarded named non-default lab, and +# lab teardown verifies that the default fleet session is byte-identical. +set -u + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +HERDR_LAB_HELPER=${HERDR_LAB_HELPER:-$ROOT/bin/fm-herdr-lab.sh} + +fail() { printf 'not ok - %s\n' "$1" >&2; exit 1; } +pass() { printf 'ok - %s\n' "$1"; } + +command -v herdr >/dev/null 2>&1 || { echo 'skip: herdr not found'; exit 0; } +command -v jq >/dev/null 2>&1 || { echo 'skip: jq not found'; exit 0; } +command -v python3 >/dev/null 2>&1 || { echo 'skip: python3 not found'; exit 0; } +[ -x "$HERDR_LAB_HELPER" ] || { echo "skip: Herdr lab helper not executable at $HERDR_LAB_HELPER"; exit 0; } + +REAL_HERDR=$(command -v herdr) +HERDR_ORIGINAL_PATH=$PATH +TMP_ROOT=$(mktemp -d "$(cd "${TMPDIR:-/tmp}" && pwd -P)/fm-herdr-session-cleanup-e2e.XXXXXX") +FAKEBIN="$TMP_ROOT/fakebin" +HOME_DIR="$TMP_ROOT/home" +mkdir -p "$FAKEBIN" "$HOME_DIR/state" "$HOME_DIR/config" +touch "$HOME_DIR/config/herdr-presentation-spaces" +printf '%s\n' herdr > "$HOME_DIR/config/backend" + +HERDR_LAB_SESSION=$("$HERDR_LAB_HELPER" name fm-herdr-session-start-stale-projection-cleanup-r1) +export HERDR_LAB_HELPER HERDR_LAB_SESSION REAL_HERDR HERDR_ORIGINAL_PATH +cleanup() { + local status=$? + env PATH="$HERDR_ORIGINAL_PATH" "$HERDR_LAB_HELPER" teardown "$HERDR_LAB_SESSION" || status=1 + rm -rf "$TMP_ROOT" + exit "$status" +} +trap cleanup EXIT +"$HERDR_LAB_HELPER" provision "$HERDR_LAB_SESSION" + +# Keep the lab helper as the only CLI transport. Production adapter calls have +# already appended the exact session; this shim strips that pair, refuses every +# other caller-supplied session, and delegates the command to helper run. +cat > "$FAKEBIN/herdr" <<'SH' +#!/usr/bin/env bash +set -u +args=("$@") +last=$((${#args[@]} - 1)) +flag=$((last - 1)) +if [ "${#args[@]}" -ge 2 ] \ + && [ "${args[$flag]}" = --session ] \ + && [ "${args[$last]}" = "$HERDR_LAB_SESSION" ]; then + unset "args[$last]" "args[$flag]" +fi +set -- "${args[@]}" +for arg in "$@"; do + case "$arg" in --session|--session=*) exit 9 ;; esac +done +if [ "${1:-}" = --version ]; then + exec env PATH="$HERDR_ORIGINAL_PATH" "$REAL_HERDR" "$@" --session "$HERDR_LAB_SESSION" +fi +exec env PATH="$HERDR_ORIGINAL_PATH" "$HERDR_LAB_HELPER" run "$HERDR_LAB_SESSION" "$@" +SH +chmod +x "$FAKEBIN/herdr" + +lab() { env PATH="$HERDR_ORIGINAL_PATH" "$HERDR_LAB_HELPER" run "$HERDR_LAB_SESSION" "$@"; } +production_process_proof() { + FM_HOME="$HOME_DIR" FM_BACKEND=herdr HERDR_SESSION="$HERDR_LAB_SESSION" \ + FM_HERDR_SESSION_CLEANUP_SOURCE_ONLY=1 PATH="$FAKEBIN:$HERDR_ORIGINAL_PATH" \ + bash -c '. "$1"; fm_herdr_cleanup_process_is_idle_shell "$2" "$3"' \ + _ "$ROOT/bin/fm-herdr-session-cleanup.sh" "$HERDR_LAB_SESSION" "$PANE" +} +focus_snapshot() { + local list workspace tab tabs + list=$(lab workspace list) || return 1 + workspace=$(printf '%s' "$list" | jq -er '[.result.workspaces[] | select(.focused == true)] | select(length == 1) | .[0].workspace_id') || return 1 + tab=$(printf '%s' "$list" | jq -er --arg workspace "$workspace" '[.result.workspaces[] | select(.workspace_id == $workspace)] | select(length == 1) | .[0].active_tab_id') || return 1 + tabs=$(lab tab list --workspace "$workspace") || return 1 + printf '%s' "$tabs" | jq -e --arg tab "$tab" '([.result.tabs[] | select(.focused == true)] | length) == 1 and ([.result.tabs[] | select(.focused == true)][0].tab_id == $tab)' >/dev/null || return 1 + printf '%s\t%s' "$workspace" "$tab" +} + +ANCHOR=$(lab workspace create --cwd "$ROOT" --label captain-anchor --focus) || fail 'could not create focus anchor' +ANCHOR_TAB=$(printf '%s' "$ANCHOR" | jq -r '.result.tab.tab_id') +TOKEN=AbCdEfGhIjKlMnOpQrStUv +ID=restored-idle-shell +TITLE="└ $ID · p:$TOKEN" +CANDIDATE=$(lab workspace create --cwd "$ROOT" --label "$TITLE" --no-focus) || fail 'could not create projected child fixture' +WS=$(printf '%s' "$CANDIDATE" | jq -r '.result.workspace.workspace_id') +PANE=$(printf '%s' "$CANDIDATE" | jq -r '.result.root_pane.pane_id') +{ + printf 'version=1\n' + printf 'task_id=%s\n' "$ID" + printf 'projection_id=%s\n' "$TOKEN" +} > "$HOME_DIR/state/$ID.herdr-presentation" + +"$HERDR_LAB_HELPER" stop "$HERDR_LAB_SESSION" >/dev/null || fail 'could not stop named lab for restored-shell reproduction' +"$HERDR_LAB_HELPER" provision "$HERDR_LAB_SESSION" || fail 'could not restore named lab layout' +lab tab focus "$ANCHOR_TAB" >/dev/null || fail 'could not restore the anchor focus after lab restart' +BEFORE_FOCUS=$(focus_snapshot) || fail 'could not capture exact pre-cleanup focus' +[ "$BEFORE_FOCUS" = "$(printf '%s\t%s' "$(printf '%s' "$ANCHOR" | jq -r '.result.workspace.workspace_id')" "$ANCHOR_TAB")" ] \ + || fail 'anchor focus does not match the exact intended workspace and tab' + +WORKSPACES=$(lab workspace list) || fail 'could not inspect restored workspaces' +TABS=$(lab tab list --workspace "$WS") || fail 'could not inspect restored tabs' +PANES=$(lab pane list --workspace "$WS") || fail 'could not inspect restored panes' +[ "$(printf '%s' "$WORKSPACES" | jq --arg title "$TITLE" '[.result.workspaces[] | select(.label == $title)] | length')" = 1 ] \ + || fail 'restored projected title is not unique' +[ "$(printf '%s' "$TABS" | jq '.result.tabs | length')" = 1 ] || fail 'restored child is not one tab' +[ "$(printf '%s' "$PANES" | jq '.result.panes | length')" = 1 ] || fail 'restored child is not one pane' +if lab agent get "$PANE" >/dev/null 2>&1; then + fail 'restored child unexpectedly retained a registered agent' +fi +attempt=0 +while [ "$attempt" -lt 50 ]; do + if production_process_proof; then + break + fi + sleep 0.1 + attempt=$((attempt + 1)) +done +[ "$attempt" -lt 50 ] || fail 'restored child did not converge to the exact childless idle-shell process-group shape' +pass 'real named lab reproduced the exact restored one-tab one-pane childless no-agent shell shape' + +FM_HOME="$HOME_DIR" FM_BACKEND=herdr HERDR_SESSION="$HERDR_LAB_SESSION" \ + PATH="$FAKEBIN:$HERDR_ORIGINAL_PATH" "$ROOT/bin/fm-herdr-session-cleanup.sh" \ + || fail 'session-start cleanup command failed' +AFTER_FOCUS=$(focus_snapshot) || fail 'could not capture exact post-cleanup focus' +[ "$AFTER_FOCUS" = "$BEFORE_FOCUS" ] || fail 'exact workspace/tab focus changed during cleanup' +if lab pane get "$PANE" >/dev/null 2>&1; then + fail 'exact stale pane survived cleanup' +fi +if lab workspace get "$WS" >/dev/null 2>&1; then + fail 'last-pane side effect did not remove the stale projected child workspace' +fi +[ ! -e "$HOME_DIR/state/$ID.herdr-presentation" ] || fail 'matching journal survived confirmed exact pane closure' +pass 'real named lab cleanup closes only the exact stale pane and preserves exact focus' + +FM_HOME="$HOME_DIR" FM_BACKEND=herdr HERDR_SESSION="$HERDR_LAB_SESSION" \ + PATH="$FAKEBIN:$HERDR_ORIGINAL_PATH" "$ROOT/bin/fm-herdr-session-cleanup.sh" \ + || fail 'idempotent repeat failed' +[ "$(focus_snapshot)" = "$BEFORE_FOCUS" ] || fail 'idempotent repeat changed focus' +lab pane get "$(printf '%s' "$ANCHOR" | jq -r '.result.root_pane.pane_id')" >/dev/null \ + || fail 'anchor pane was touched by cleanup' +STATUS=$(lab status --json) || fail 'could not read final named-lab version evidence' +pass 'real named lab cleanup is idempotent and leaves the default fleet session to the teardown tripwire' +printf 'evidence: herdr=%s protocol=%s default-session-tripwire=armed\n' \ + "$(printf '%s' "$STATUS" | jq -r '.client.version')" \ + "$(printf '%s' "$STATUS" | jq -r '.server.protocol')" diff --git a/tests/fm-herdr-session-cleanup.test.sh b/tests/fm-herdr-session-cleanup.test.sh new file mode 100755 index 00000000000..f4c1df153b4 --- /dev/null +++ b/tests/fm-herdr-session-cleanup.test.sh @@ -0,0 +1,309 @@ +#!/usr/bin/env bash +# Focused safety tests for bin/fm-herdr-session-cleanup.sh. +# Covers one exact cleanup, every title/journal/topology/agent/process refusal, +# locked revalidation races, focus refusal, read errors, and repeat idempotence. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +TMP_ROOT=$(fm_test_tmproot fm-herdr-session-cleanup) +FM_TEST_CLEANUP_DIRS+=("$TMP_ROOT") +trap fm_test_cleanup EXIT + +export FM_HOME="$TMP_ROOT/home" +export FM_STATE_OVERRIDE="$FM_HOME/state" +export FM_CONFIG_OVERRIDE="$FM_HOME/config" +mkdir -p "$FM_STATE_OVERRIDE" "$FM_CONFIG_OVERRIDE" +touch "$FM_CONFIG_OVERRIDE/herdr-presentation-spaces" +printf '%s\n' herdr > "$FM_CONFIG_OVERRIDE/backend" +FAKEBIN=$(fm_fakebin "$TMP_ROOT") +fm_fake_exit0 "$FAKEBIN" herdr +export PATH="$FAKEBIN:$PATH" +export FM_HERDR_SESSION_CLEANUP_SOURCE_ONLY=1 +# shellcheck source=/dev/null +. "$ROOT/bin/fm-herdr-session-cleanup.sh" +unset FM_HERDR_SESSION_CLEANUP_SOURCE_ONLY + +LINUX_PROCESS_INFO='{"result":{"process_info":{"foreground_processes":[{"argv":["/bin/sh"],"name":"sh","pid":67}]}}}' +[ "$(fm_herdr_cleanup_process_argv0 "$LINUX_PROCESS_INFO")" = /bin/sh ] \ + || fail "Linux Herdr process argv array was not accepted" +if fm_herdr_cleanup_process_argv0 \ + '{"result":{"process_info":{"foreground_processes":[{"argv":[67],"name":"sh","pid":67}]}}}' \ + >/dev/null 2>&1; then + fail "non-string Herdr process argv was accepted" +fi +pass "process proof reads Linux Herdr argv arrays and rejects malformed executable identities" + +TOKEN=AbCdEfGhIjKlMnOpQrStUv +ID=task +WS=w2 +TAB=w2:t1 +PANE=w2:p1 +TITLE="└ task · p:$TOKEN" +FIXTURE_DIR="$TMP_ROOT/fixture" +LOCK_LOG="$TMP_ROOT/locks.log" +CLOSE_LOG="$TMP_ROOT/closes.log" +mkdir -p "$FIXTURE_DIR" + +fm_backend_name() { printf herdr; } +fm_backend_herdr_session() { printf test; } +fm_backend_herdr_presentation_session_lock_path() { printf '%s/presentation.lock' "$TMP_ROOT"; } +fm_lock_try_acquire() { + printf '%s\n' "$1" >> "$LOCK_LOG" + mkdir "$1" 2>/dev/null +} +fm_lock_release() { rm -rf -- "$1"; } +fm_herdr_cleanup_process_is_idle_shell() { [ ! -e "$FIXTURE_DIR/process-unsafe" ]; } +fm_backend_herdr_projection_focus_snapshot() { + [ ! -e "$FIXTURE_DIR/focus-unreadable" ] || return 1 + printf 'w1\t%s' "$(cat "$FIXTURE_DIR/active-tab")" +} + +fixture_workspace_json() { + local title=$1 tabs=$2 panes=$3 focused=false active=$TAB + [ "$(cat "$FIXTURE_DIR/active-tab")" = "$TAB" ] && focused=true + printf '{"workspace_id":"%s","label":"%s","focused":%s,"active_tab_id":"%s","tab_count":%s,"pane_count":%s}' \ + "$WS" "$title" "$focused" "$active" "$tabs" "$panes" +} + +fixture_workspaces() { + local title tabs panes + title=$(cat "$FIXTURE_DIR/title") + tabs=$(cat "$FIXTURE_DIR/tabs") + panes=$(cat "$FIXTURE_DIR/panes") + printf '[{"workspace_id":"w1","label":"firstmate","focused":%s,"active_tab_id":"w1:t1","tab_count":1,"pane_count":1},' \ + "$( [ "$(cat "$FIXTURE_DIR/active-tab")" = w1:t1 ] && printf true || printf false )" + fixture_workspace_json "$title" "$tabs" "$panes" + if [ -e "$FIXTURE_DIR/duplicate-token" ]; then + printf ',{"workspace_id":"w3","label":"└ copy · p:%s","focused":false,"active_tab_id":"w3:t1","tab_count":1,"pane_count":1}' "$TOKEN" + fi + printf ']' +} + +fixture_tabs() { + local count i + count=$(cat "$FIXTURE_DIR/tabs") + printf '[' + i=1 + while [ "$i" -le "$count" ]; do + [ "$i" -eq 1 ] || printf ',' + printf '{"tab_id":"%s:t%s","workspace_id":"%s","focused":false,"label":"fm-task"}' "$WS" "$i" "$WS" + i=$((i + 1)) + done + printf ']' +} + +fixture_panes() { + local count i + count=$(cat "$FIXTURE_DIR/panes") + printf '[' + i=1 + while [ "$i" -le "$count" ]; do + [ "$i" -eq 1 ] || printf ',' + printf '{"pane_id":"%s:p%s","tab_id":"%s:t%s","workspace_id":"%s","agent_status":"unknown"}' \ + "$WS" "$i" "$WS" "$i" "$WS" + i=$((i + 1)) + done + printf ']' +} + +fm_backend_herdr_cli() { + local _session=$1 first=${2:-} second=${3:-} title tabs panes + shift + [ ! -e "$FIXTURE_DIR/error-${first}-${second}" ] || return 1 + if [ -e "$FIXTURE_DIR/closed" ]; then + case "$first $second" in + "pane get") printf '%s\n' '{"error":{"code":"pane_not_found"}}' >&2; return 1 ;; + "workspace list") printf '%s\n' '{"result":{"workspaces":[{"workspace_id":"w1","label":"firstmate","focused":true,"active_tab_id":"w1:t1","tab_count":1,"pane_count":1}]}}'; return 0 ;; + esac + fi + if [ -e "$FIXTURE_DIR/race" ] && [ -e "$FIXTURE_DIR/snapshotted" ] && [ "$first $second" = "workspace list" ]; then + printf '%s\n' '{"result":{"workspaces":[{"workspace_id":"w1","label":"firstmate","focused":true,"active_tab_id":"w1:t1","tab_count":1,"pane_count":1},{"workspace_id":"w2","label":"renamed","focused":false,"active_tab_id":"w2:t1","tab_count":1,"pane_count":1}]}}' + return 0 + fi + title=$(cat "$FIXTURE_DIR/title") + tabs=$(cat "$FIXTURE_DIR/tabs") + panes=$(cat "$FIXTURE_DIR/panes") + case "$first $second" in + "workspace list") + printf '{"result":{"workspaces":'; fixture_workspaces; printf '}}\n' + ;; + "workspace get") + printf '{"result":{"workspace":'; fixture_workspace_json "$title" "$tabs" "$panes"; printf '}}\n' + ;; + "tab list") + printf '{"result":{"tabs":'; fixture_tabs; printf '}}\n' + ;; + "pane list") + printf '{"result":{"panes":'; fixture_panes; printf '}}\n' + ;; + "pane get") + printf '{"result":{"pane":{"pane_id":"%s","tab_id":"%s","workspace_id":"%s"}}}\n' "$PANE" "$TAB" "$WS" + ;; + "agent get") + case "$(cat "$FIXTURE_DIR/agent")" in + absent) printf '%s\n' '{"error":{"code":"agent_not_found"}}' >&2; return 1 ;; + live) printf '%s\n' '{"result":{"agent":{"agent_status":"idle"}}}' ;; + unknown) printf '%s\n' '{"error":{"code":"internal_error"}}' >&2; return 1 ;; + esac + ;; + "api snapshot") + : > "$FIXTURE_DIR/snapshotted" + printf '{"result":{"snapshot":{"focused_workspace_id":"w1","focused_tab_id":"%s","focused_pane_id":"w1:p1","workspaces":' "$(cat "$FIXTURE_DIR/active-tab")" + fixture_workspaces + printf ',"tabs":'; fixture_tabs + printf ',"panes":'; fixture_panes + printf '}}}\n' + ;; + "session list") + printf '%s\n' '{"sessions":[{"name":"test","running":true,"socket_path":"/tmp/fake.sock"}]}' + ;; + "pane close") + : > "$FIXTURE_DIR/closed" + ;; + *) return 1 ;; + esac +} + +fm_backend_herdr_projection_close_pane_focus_preserving() { + [ ! -e "$FIXTURE_DIR/focus-refuse" ] || return 1 + [ "${3:-}" = no-agent ] || return 1 + printf '%s\n' "$*" >> "$CLOSE_LOG" + : > "$FIXTURE_DIR/closed" +} + +write_v1() { # <id> [token] + local id=$1 token=${2:-$TOKEN} + { + printf 'version=1\n' + printf 'task_id=%s\n' "$id" + printf 'projection_id=%s\n' "$token" + } > "$FM_STATE_OVERRIDE/$id.herdr-presentation" +} + +write_v2() { # <home> <workspace> <tab> <pane> + local home=$1 workspace=$2 tab=$3 pane=$4 + { + printf 'version=2\n' + printf 'task_id=%s\n' "$ID" + printf 'projection_id=%s\n' "$TOKEN" + printf 'home=%s\n' "$home" + printf 'session=test\nworkspace_id=%s\ntab_id=%s\npane_id=%s\n' "$workspace" "$tab" "$pane" + printf 'parent_workspace_id=w1\nparent_label=firstmate\nworkspace_label=%s\ntask_label=fm-%s\n' "$TITLE" "$ID" + } > "$FM_STATE_OVERRIDE/$ID.herdr-presentation" +} + +write_cross_home_v2() { + mkdir -p "$TMP_ROOT/other-home" + write_v2 "$TMP_ROOT/other-home" "$WS" "$TAB" "$PANE" +} + +reset_fixture() { + rm -rf "$FIXTURE_DIR" "$TMP_ROOT"/*.lock "${FM_STATE_OVERRIDE:?}/"* + mkdir -p "$FIXTURE_DIR" + : > "$LOCK_LOG"; : > "$CLOSE_LOG" + printf '%s\n' "$TITLE" > "$FIXTURE_DIR/title" + printf '1\n' > "$FIXTURE_DIR/tabs" + printf '1\n' > "$FIXTURE_DIR/panes" + printf 'w1:t1\n' > "$FIXTURE_DIR/active-tab" + printf 'absent\n' > "$FIXTURE_DIR/agent" + write_v1 "$ID" +} + +assert_preserved() { # <case> + local name=$1 had_journal=0 + [ -f "$FM_STATE_OVERRIDE/$ID.herdr-presentation" ] && had_journal=1 + fm_herdr_session_cleanup >/dev/null 2>&1 + if [ "$had_journal" -eq 1 ]; then + [ -f "$FM_STATE_OVERRIDE/$ID.herdr-presentation" ] || fail "$name retired the journal" + fi + [ ! -s "$CLOSE_LOG" ] || fail "$name closed the pane" + pass "$name preserves the candidate" +} + +reset_fixture +fm_herdr_session_cleanup >/dev/null 2>&1 +[ ! -e "$FM_STATE_OVERRIDE/$ID.herdr-presentation" ] || fail "positive cleanup kept the journal" +[ "$(wc -l < "$CLOSE_LOG" | tr -d ' ')" = 1 ] || fail "positive cleanup did not close exactly once" +[ "$(sed -n '1p' "$LOCK_LOG")" = "$FM_STATE_OVERRIDE/.spawn-$ID.lock" ] || fail "task lock was not acquired first" +[ "$(sed -n '2p' "$LOCK_LOG")" = "$TMP_ROOT/presentation.lock" ] || fail "presentation lock was not acquired second" +pass "exact stale projection closes one exact pane under task then presentation locks" +fm_herdr_session_cleanup >/dev/null 2>&1 +[ "$(wc -l < "$CLOSE_LOG" | tr -d ' ')" = 1 ] || fail "repeat cleanup closed again" +pass "successful cleanup is idempotent on repeat" + +reset_fixture; printf '%s\n' '└ malformed p:AbCdEfGhIjKlMnOpQrStUv' > "$FIXTURE_DIR/title"; assert_preserved "malformed title" +reset_fixture; printf '%s\n' '└ missing-token' > "$FIXTURE_DIR/title"; assert_preserved "missing token" +reset_fixture; printf 'version=1\ntask_id=%s\nprojection_id=short\n' "$ID" > "$FM_STATE_OVERRIDE/$ID.herdr-presentation"; assert_preserved "malformed journal" +reset_fixture; : > "$FIXTURE_DIR/duplicate-token"; assert_preserved "duplicate token" +reset_fixture; printf '%s\n' "└ task · p:$TOKEN p:$TOKEN" > "$FIXTURE_DIR/title"; assert_preserved "duplicate title token" +reset_fixture; rm -f "$FM_STATE_OVERRIDE/$ID.herdr-presentation"; assert_preserved "zero journal match" +reset_fixture; write_v1 fm-task; assert_preserved "multiple journal matches" +reset_fixture; rm -f "$FM_STATE_OVERRIDE/$ID.herdr-presentation"; write_cross_home_v2; assert_preserved "cross-home journal" +reset_fixture; write_v2 "$FM_HOME" w9 "$TAB" "$PANE"; assert_preserved "v2 workspace binding mismatch" +reset_fixture; write_v2 "$FM_HOME" "$WS" w9:t1 "$PANE"; assert_preserved "v2 tab binding mismatch" +reset_fixture; write_v2 "$FM_HOME" "$WS" "$TAB" w9:p1; assert_preserved "v2 pane binding mismatch" +reset_fixture; write_v2 "$FM_HOME" "$WS" "$TAB" "$PANE" +fm_herdr_session_cleanup >/dev/null 2>&1 +[ ! -e "$FM_STATE_OVERRIDE/$ID.herdr-presentation" ] || fail "matching v2 cleanup kept the journal" +[ "$(wc -l < "$CLOSE_LOG" | tr -d ' ')" = 1 ] || fail "matching v2 cleanup did not close exactly once" +pass "v2 cleanup requires and accepts the exact journal endpoint binding" +reset_fixture; : > "$FM_STATE_OVERRIDE/$ID.meta"; assert_preserved "current task metadata" +reset_fixture; printf 'live\n' > "$FIXTURE_DIR/agent"; assert_preserved "registered agent" +reset_fixture; printf 'unknown\n' > "$FIXTURE_DIR/agent"; assert_preserved "unknown agent" +reset_fixture; printf '2\n' > "$FIXTURE_DIR/tabs"; printf '2\n' > "$FIXTURE_DIR/panes"; assert_preserved "multiple tabs" +reset_fixture; printf '2\n' > "$FIXTURE_DIR/panes"; assert_preserved "multiple panes" +reset_fixture; : > "$FIXTURE_DIR/process-unsafe"; assert_preserved "non-idle shell" +reset_fixture; : > "$FIXTURE_DIR/process-unsafe"; assert_preserved "child process or shell job" +reset_fixture; : > "$FIXTURE_DIR/error-api-snapshot"; assert_preserved "unreadable snapshot" +reset_fixture; : > "$FIXTURE_DIR/error-workspace-get"; assert_preserved "unreadable topology check" +reset_fixture; : > "$FIXTURE_DIR/race"; assert_preserved "revalidation race" +reset_fixture; printf '%s\n' "$TAB" > "$FIXTURE_DIR/active-tab"; assert_preserved "active target" +reset_fixture; : > "$FIXTURE_DIR/focus-refuse"; assert_preserved "focus refusal" + +INTEGRATION_ROOT="$TMP_ROOT/bootstrap-integration" +mkdir -p "$INTEGRATION_ROOT/home/state" "$INTEGRATION_ROOT/home/data" "$INTEGRATION_ROOT/home/config" +cp -R "$ROOT/bin" "$INTEGRATION_ROOT/bin" +TRACE="$INTEGRATION_ROOT/cleanup.trace" +cat > "$INTEGRATION_ROOT/bin/fm-herdr-session-cleanup.sh" <<'SH' +#!/usr/bin/env bash +printf '%s\n' "${FM_HOME:?}" >> "${FM_HERDR_CLEANUP_TRACE:?}" +SH +chmod +x "$INTEGRATION_ROOT/bin/fm-herdr-session-cleanup.sh" +printf '%s\n' manual > "$INTEGRATION_ROOT/home/config/backlog-backend" +FM_HOME="$INTEGRATION_ROOT/home" FM_HERDR_CLEANUP_TRACE="$TRACE" FM_BOOTSTRAP_DETECT_ONLY=1 \ + "$INTEGRATION_ROOT/bin/fm-bootstrap.sh" >/dev/null 2>&1 +[ ! -e "$TRACE" ] || fail "detect-only bootstrap ran stale projection cleanup" +FM_HOME="$INTEGRATION_ROOT/home" FM_HERDR_CLEANUP_TRACE="$TRACE" \ + "$INTEGRATION_ROOT/bin/fm-bootstrap.sh" >/dev/null 2>&1 +[ ! -e "$TRACE" ] || fail "standalone bootstrap ran lock-owned stale projection cleanup" +pass "standalone bootstrap cannot run lock-owned stale projection cleanup" + +cat > "$INTEGRATION_ROOT/bin/fm-lock.sh" <<'SH' +#!/usr/bin/env bash +printf '%s\n' 'lock acquired' +SH +chmod +x "$INTEGRATION_ROOT/bin/fm-lock.sh" +FM_HOME="$INTEGRATION_ROOT/home" FM_ROOT_OVERRIDE="$INTEGRATION_ROOT" \ + FM_HERDR_CLEANUP_TRACE="$TRACE" \ + "$INTEGRATION_ROOT/bin/fm-session-start.sh" >/dev/null 2>&1 \ + || fail "lock-owning session start failed" +[ "$(cat "$TRACE")" = "$INTEGRATION_ROOT/home" ] \ + || fail "lock-owning session start did not run cleanup for its exact home" + +: > "$TRACE" +cat > "$INTEGRATION_ROOT/bin/fm-lock.sh" <<'SH' +#!/usr/bin/env bash +printf '%s\n' 'error: another live firstmate session holds the lock' >&2 +exit 1 +SH +chmod +x "$INTEGRATION_ROOT/bin/fm-lock.sh" +FM_HOME="$INTEGRATION_ROOT/home" FM_ROOT_OVERRIDE="$INTEGRATION_ROOT" \ + FM_HERDR_CLEANUP_TRACE="$TRACE" \ + "$INTEGRATION_ROOT/bin/fm-session-start.sh" >/dev/null 2>&1 \ + || fail "read-only session start failed" +[ ! -s "$TRACE" ] || fail "read-only session start ran stale projection cleanup" +pass "session start runs cleanup only after acquiring its home lock" + +printf 'all fm-herdr-session-cleanup tests passed\n' diff --git a/tests/fm-install-herdr.test.sh b/tests/fm-install-herdr.test.sh deleted file mode 100755 index cc5a70ada88..00000000000 --- a/tests/fm-install-herdr.test.sh +++ /dev/null @@ -1,106 +0,0 @@ -#!/usr/bin/env bash -# Contract tests for the pinned Herdr / Treehouse CI installers and the -# bounded Herdr lab cleanup helper. These tests do not download release assets -# and never start or stop the captain's default Herdr session. -set -u - -# shellcheck source=tests/lib.sh -. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" - -HERDR_INSTALL="$ROOT/bin/fm-install-herdr.sh" -TREEHOUSE_INSTALL="$ROOT/bin/fm-install-treehouse.sh" -CLEANUP="$ROOT/bin/fm-herdr-ci-cleanup.sh" -CI="$ROOT/.github/workflows/ci.yml" - -assert_present "$HERDR_INSTALL" "bin/fm-install-herdr.sh is missing" -assert_present "$TREEHOUSE_INSTALL" "bin/fm-install-treehouse.sh is missing" -assert_present "$CLEANUP" "bin/fm-herdr-ci-cleanup.sh is missing" -[ -x "$HERDR_INSTALL" ] || fail "fm-install-herdr.sh must be executable" -[ -x "$TREEHOUSE_INSTALL" ] || fail "fm-install-treehouse.sh must be executable" -[ -x "$CLEANUP" ] || fail "fm-herdr-ci-cleanup.sh must be executable" - -test_herdr_installer_pins_exact_version_and_checksums() { - assert_grep 'FM_HERDR_CI_VERSION=0.7.4' "$HERDR_INSTALL" \ - "Herdr installer must pin suite-verified 0.7.4" - assert_grep 'FM_HERDR_CI_MIN_PROTOCOL=16' "$HERDR_INSTALL" \ - "Herdr installer must require protocol floor 16" - assert_grep 'ogulcancelik/herdr' "$HERDR_INSTALL" \ - "Herdr installer must use the official GitHub release source" - assert_grep 'herdr-linux-x86_64' "$HERDR_INSTALL" \ - "Herdr installer must name the Linux x86_64 release asset" - assert_grep 'bc0fc02d4ba500f9cac2353a43e67fe036785ecca6eb55378e050fac3c103059' "$HERDR_INSTALL" \ - "Herdr installer must pin the Linux x86_64 SHA-256" - assert_grep 'sha256sum' "$HERDR_INSTALL" \ - "Herdr installer must verify a SHA-256 checksum" - assert_grep '--max-filesize' "$HERDR_INSTALL" \ - "Herdr installer must bound the download size" - assert_no_grep 'brew install' "$HERDR_INSTALL" \ - "Herdr installer must not use a floating package-manager install" - assert_no_grep 'apt-get install' "$HERDR_INSTALL" \ - "Herdr installer must not use a floating package-manager install" - pass "Herdr installer pins exact version, asset, checksum, and protocol floor" -} - -test_treehouse_installer_pins_exact_version_and_checksums() { - assert_grep 'FM_TREEHOUSE_CI_VERSION=2.0.1' "$TREEHOUSE_INSTALL" \ - "Treehouse installer must pin the suite-verified 2.0.1 release" - assert_grep 'kunchenguid/treehouse' "$TREEHOUSE_INSTALL" \ - "Treehouse installer must use the official GitHub release source" - assert_grep 'linux-amd64.tar.gz' "$TREEHOUSE_INSTALL" \ - "Treehouse installer must name the Linux amd64 archive" - assert_grep '1d5a32751ab921670103fd201ddb2b91b47338cb13976f45642b827cf8976af2' "$TREEHOUSE_INSTALL" \ - "Treehouse installer must pin the Linux amd64 SHA-256" - assert_grep '--max-filesize' "$TREEHOUSE_INSTALL" \ - "Treehouse installer must bound the download size" - assert_no_grep 'brew install' "$TREEHOUSE_INSTALL" \ - "Treehouse installer must not use a floating package-manager install" - pass "Treehouse installer pins exact version, asset, and checksum" -} - -test_cleanup_only_targets_job_owned_lab_sessions() { - assert_grep 'fm-lab-' "$CLEANUP" \ - "cleanup must only consider fm-lab-* session names" - assert_grep 'default == false' "$CLEANUP" \ - "cleanup must refuse default sessions" - assert_grep 'snapshot' "$CLEANUP" \ - "cleanup must support a pre-suite snapshot" - assert_grep 'teardown' "$CLEANUP" \ - "cleanup must support post-suite teardown of the delta" - # Must not call ambient server stop. - assert_no_grep 'server stop' "$CLEANUP" \ - "cleanup must never call ambient herdr server stop" - pass "cleanup is bounded to job-owned fm-lab-* sessions" -} - -test_ci_wires_installers_and_required_lane() { - assert_grep 'tests-herdr:' "$CI" "CI must define the required Herdr Behavior job" - assert_grep 'fm-install-herdr.sh' "$CI" "CI must call the Herdr installer" - assert_grep 'fm-install-treehouse.sh' "$CI" "CI must call the Treehouse installer" - assert_grep 'fm-herdr-ci-cleanup.sh snapshot' "$CI" "CI must snapshot sessions before the suite" - assert_grep 'fm-herdr-ci-cleanup.sh teardown' "$CI" "CI must teardown job-owned sessions after" - assert_grep "fail-on-gate-skip 'herdr not found'" "$CI" \ - "CI Herdr lane must fail on herdr-not-found" - assert_grep 'family real-herdr-gated' "$CI" \ - "CI Herdr lane must run only the real-herdr-gated family" - assert_grep 'lane portable-parallel-1' "$CI" \ - "portable CI must run parallel shard 1" - assert_grep 'lane portable-parallel-2' "$CI" \ - "portable CI must run parallel shard 2" - assert_grep 'lane portable-serial' "$CI" \ - "portable CI must run the serial remainder" - assert_grep 'fm-test-run.sh --check-coverage' "$CI" \ - "CI must prove portable lanes and Herdr partition the complete inventory" - # Live harness credential tests must stay out of the default Herdr lane. - assert_no_grep 'live-harness-optin' "$CI" \ - "CI must not run live-harness-optin in the required Herdr lane" - assert_no_grep 'FM_AFK_PI_HERDR_E2E' "$CI" \ - "CI must not enable live Pi/Herdr credential tests" - assert_no_grep 'FM_SEND_MARKER_HERDR_E2E' "$CI" \ - "CI must not enable live marker Herdr credential tests" - pass "CI wires pinned installers into a required serial Herdr lane" -} - -test_herdr_installer_pins_exact_version_and_checksums -test_treehouse_installer_pins_exact_version_and_checksums -test_cleanup_only_targets_job_owned_lab_sessions -test_ci_wires_installers_and_required_lane diff --git a/tests/fm-instruction-owners.test.sh b/tests/fm-instruction-owners.test.sh deleted file mode 100755 index 196511bffb4..00000000000 --- a/tests/fm-instruction-owners.test.sh +++ /dev/null @@ -1,260 +0,0 @@ -#!/usr/bin/env bash -# Static contract tests for conditional instruction owners introduced before the -# AGENTS.md reduction pass. -# shellcheck disable=SC2016 -set -u - -# shellcheck source=tests/lib.sh -. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" - -DIAG="$ROOT/.agents/skills/diagnostic-reasoning/SKILL.md" -PROJECT="$ROOT/.agents/skills/project-management/SKILL.md" -HARNESS="$ROOT/.agents/skills/harness-adapters/SKILL.md" -CODING="$ROOT/.agents/skills/firstmate-coding-guidelines/SKILL.md" -RECOVERY="$ROOT/.agents/skills/stuck-crewmate-recovery/SKILL.md" -SECONDMATE="$ROOT/.agents/skills/secondmate-provisioning/SKILL.md" -CONFIG="$ROOT/docs/configuration.md" -AGENTS="$ROOT/AGENTS.md" -BRIEF="$ROOT/bin/fm-brief.sh" - -test_new_skill_metadata_and_triggers() { - local skill name count - for pair in "diagnostic-reasoning:$DIAG" "project-management:$PROJECT"; do - name=${pair%%:*} - skill=${pair#*:} - assert_present "$skill" "$name skill is missing" - assert_grep "name: $name" "$skill" "$name skill metadata has the wrong name" - assert_grep "user-invocable: false" "$skill" "$name skill must not be user-invocable" - assert_grep " internal: true" "$skill" "$name skill must be internal" - count=$(grep -Fc -- "- \`$name\` -" "$ROOT/AGENTS.md") - [ "$count" -eq 1 ] || fail "$name must have exactly one AGENTS.md trigger entry, found $count" - done - assert_grep 'Use before scoping a reported bug and before acting on a diagnostic report.' "$DIAG" \ - "diagnostic skill metadata lost its precise load trigger" - assert_grep '`diagnostic-reasoning` - load before scoping a reported bug and before acting on a diagnostic report.' "$ROOT/AGENTS.md" \ - "AGENTS.md lost the diagnostic-reasoning trigger" - assert_grep 'Use before adding, creating, removing, or initializing a project.' "$PROJECT" \ - "project-management skill metadata lost its precise load trigger" - assert_grep '`project-management` - load before adding, creating, removing, or initializing a project.' "$ROOT/AGENTS.md" \ - "AGENTS.md lost the project-management trigger" - pass "new internal skills have one precise AGENTS.md trigger each" -} - -test_diagnostic_owner_covers_causal_procedure() { - assert_grep "single owner of Firstmate's bug-diagnosis reasoning procedure" "$DIAG" \ - "diagnostic skill does not declare ownership" - for phrase in \ - "end-to-end reproduction aligned with the real user path" \ - "initiating trigger" \ - "masking condition" \ - "visible symptom" \ - "proven path" \ - "relevant history" \ - "smallest counterfactual" \ - "disconfirming evidence"; do - assert_grep "$phrase" "$DIAG" "diagnostic owner is missing '$phrase'" - done - assert_grep "evidence, not authorization to change code" "$DIAG" \ - "diagnostic owner lost the diagnosis-only authority boundary" - pass "diagnostic-reasoning owns the approved evidence procedure" -} - -test_project_management_owner_covers_guarded_operations() { - assert_grep "single owner of Firstmate's project-management procedure" "$PROJECT" \ - "project-management skill does not declare ownership" - for phrase in \ - 'bin/fm-project-mode.sh' \ - '`no-mistakes`' \ - '`direct-PR`' \ - '`local-only`' \ - 'Default it off' \ - 'Creating a GitHub repository is outward-facing.' \ - "captain's explicit consent" \ - 'Never issue a raw removal command from Firstmate.' \ - 'no-mistakes init && no-mistakes doctor'; do - assert_grep "$phrase" "$PROJECT" "project-management owner is missing '$phrase'" - done - pass "project-management owns registry, delivery posture, consent, initialization, and removal safety" -} - -test_generic_effort_fallback_respects_precedence() { - local section - section=$(awk ' - /^Effort precedence is / { found = 1 } - found && /^The supported launch-profile flags / { exit } - found { print } - ' "$HARNESS") - assert_contains "$section" "explicit per-task captain instruction first" \ - "effort rubric lost per-task captain precedence" - assert_contains "$section" "standing dispatch profile or secondmate pin" \ - "effort rubric lost standing configuration precedence" - assert_contains "$section" 'Use `low` for well-understood work' \ - "effort rubric lost its low fallback" - assert_contains "$section" '`xhigh` for ambiguous investigation or design' \ - "effort rubric lost its xhigh fallback" - assert_contains "$section" "Choose intermediate levels proportionally" \ - "effort rubric lost proportional intermediate levels" - assert_contains "$section" 'Never select `max` from this fallback' \ - "effort rubric permits max without an explicit captain preference" - if printf '%s\n' "$section" | grep -qi sol; then - fail "generic effort fallback must not contain Sol-specific policy" - fi - pass "generic effort fallback applies only below captain and standing configuration" -} - -test_shared_authoring_requirements_are_owned() { - assert_grep "review every affected supported primary harness and runtime backend" "$CODING" \ - "coding guidance lost the supported compatibility matrix review" - assert_grep "prefer deterministic and idempotent enforcement over relying on agent memory alone" "$CODING" \ - "coding guidance lost deterministic idempotent enforcement" - assert_grep "critical safety, routing, startup, and supervision infrastructure" "$CODING" \ - "coding guidance lost the critical infrastructure scope" - pass "firstmate-coding-guidelines owns compatibility review and deterministic enforcement" -} - -test_secondmate_registry_contract_stays_concise() { - local guidance routing_section schema_line - routing_section=$(awk ' - /^## Routing table$/ { found = 1 } - found && /^## Charter and seed$/ { exit } - found { print } - ' "$SECONDMATE") - guidance=$(awk ' - /^## Routing table$/ { found = 1 } - found && /^## Backlog handoff$/ { exit } - found { print } - ' "$SECONDMATE") - schema_line="- <id> - <one-sentence charter summary> (home: <absolute-home-path>; scope: <natural-language responsibility>; projects: <project-a>, <project-b>; added <date>)" - assert_contains "$routing_section" "$schema_line" \ - "secondmate routing table lost the parser-compatible single-line schema" - assert_contains "$routing_section" "Each registry entry stays concise and single-line" \ - "secondmate routing table no longer requires concise single-line entries" - assert_contains "$routing_section" "genuinely domain-specific hard rules" \ - "secondmate routing table no longer limits extra prose to domain-specific hard rules" - assert_contains "$routing_section" "The home-seeded \`data/charter.md\` is the sole owner of boilerplate idle-by-default behavior, the normal delegation lifecycle, and standard escalation contracts" \ - "secondmate routing table lost the explicit charter ownership pointer" - assert_contains "$routing_section" "no extra registry pointer field is needed" \ - "secondmate routing table no longer explains why the existing home field is the charter pointer" - for phrase in \ - "go idle and wait silently" \ - "Act only on tasks" \ - "never spawn a survey" \ - "run normal firstmate bootstrap" \ - "escalation back to the main firstmate status file" \ - "requests-from-main-firstmate contract" \ - "waits for routed tasks, never self-initiating a survey or audit" \ - "marked supervisor requests return through status" \ - "unmarked captain messages stay conversational"; do - if printf '%s\n' "$guidance" | grep -F "$phrase" >/dev/null; then - fail "secondmate provisioning guidance restated charter boilerplate: $phrase" - fi - done - pass "secondmate registry guidance keeps concise routes and points to the charter" -} - -test_state_startup_and_ordinary_recovery_placement() { - assert_grep "single owner of the top-level operational-home layout" "$CONFIG" \ - "configuration docs do not own the operational state layout" - assert_grep "header is the single owner of session-start ordering" "$CONFIG" \ - "session-start mechanism is not assigned to the script header" - assert_grep "Ordinary dead-direct-report recovery is owned by \`stuck-crewmate-recovery\`" "$CONFIG" \ - "D05 ordinary recovery placement is missing" - assert_grep "## Session-start reconciliation for a dead ordinary direct report" "$RECOVERY" \ - "stuck-crewmate-recovery lacks the dead ordinary direct-report procedure" - assert_grep "treehouse status" "$RECOVERY" \ - "ordinary recovery lost treehouse inventory inspection" - assert_grep "recorded \`orca_worktree_id=\` and \`terminal=\`" "$RECOVERY" \ - "ordinary recovery lost Orca inventory inspection" - assert_grep "session-start digest reports an ordinary direct report's endpoint dead or its metadata has no window" "$AGENTS" \ - "AGENTS.md does not trigger ordinary dead-report recovery" - pass "state, startup, and ordinary recovery have focused owners and triggers" -} - -test_compressed_agents_owner_map() { - assert_grep '`docs/configuration.md` is the single owner of the top-level operational-home layout' "$AGENTS" \ - "AGENTS.md lost the state-layout owner pointer" - assert_grep 'header is the single owner of composed commands, ordering, and digest contents' "$AGENTS" \ - "AGENTS.md lost the session-start owner pointer" - assert_grep '`docs/configuration.md` owns dispatch-profile and runtime-backend schemas' "$AGENTS" \ - "AGENTS.md lost the dispatch-schema owner pointer" - assert_grep 'That skill owns registry syntax, delivery-mode selection' "$AGENTS" \ - "AGENTS.md lost the project-management owner pointer" - assert_grep 'The delivery lifecycle is an always-loaded operational contract' "$AGENTS" \ - "AGENTS.md no longer owns the delivery lifecycle" - assert_grep 'Fleet supervision is an always-loaded operational contract' "$AGENTS" \ - "AGENTS.md no longer owns fleet supervision" - assert_grep '`.tasks.toml`, `docs/configuration.md`, and current `tasks-axi --help` own the backlog schema' "$AGENTS" \ - "AGENTS.md lost the backlog-mechanics owner pointer" - assert_grep '`bin/fm-brief.sh` and its help own scaffold syntax' "$AGENTS" \ - "AGENTS.md lost the brief-mechanics owner pointer" - assert_grep '`docs/configuration.md` owns activation, generated state, cadence, wire protocol' "$AGENTS" \ - "AGENTS.md lost the X-mode mechanics owner pointer" - pass "compressed AGENTS.md records the approved one-owner map" -} - -test_intake_reuses_evidence_and_parallelizes_safe_work() { - for phrase in \ - 'consult existing reports and established evidence' \ - 'remaining bounded research inside it' \ - 'unresolved uncertainty could materially change whether or what to build' \ - 'relay it without a design-only scout' \ - 'ask one concise implementation question when useful' \ - 'Never both present a likely-enough solution' \ - 'overlap as a risk signal rather than an automatic reason to wait' \ - 'independently implemented and validated' \ - 'selected delivery path can reconcile ordinary rebases or conflicts' \ - 'Serialize only for a true semantic dependency' \ - 'shared mutable external state' \ - 'incompatible concurrent migration' \ - 'same-file editing alone is insufficient' \ - 'genuine blockers remain durable'; do - assert_grep "$phrase" "$AGENTS" "intake contract lost '$phrase'" - done - assert_grep 'dispatch isolated work immediately with no concurrency cap' "$AGENTS" \ - "intake contract lost unbounded safe parallel dispatch" - assert_grep 'captain explicitly requests a separate knowledge or design deliverable' "$AGENTS" \ - "intake contract lost captain-requested separate scouts" - assert_grep 'When implementation is separately authorized, promote the existing scout' "$AGENTS" \ - "intake contract lost genuine scout promotion" - pass "intake reuses evidence, reserves scouts for uncertainty, and parallelizes safe work" -} - -test_compressed_agents_retains_authority_and_supervision_safety() { - for phrase in \ - 'A lock-refused session must not spawn, steer, merge, drain the wake queue' \ - 'A diagnostic request, report, recommendation, or implementation-ready finding is evidence, not authorization to change code.' \ - 'The selected delivery path owns its own rigor.' \ - 'When no-mistakes is selected, no-mistakes alone owns review, fixes, tests, documentation, push, PR, and CI; otherwise follow the faster path without adding an independent reviewer.' \ - 'Never hold work outside no-mistakes for a manual clean verdict, stack serial manual reviews, or infer authority for one from security, architecture, or risk alone.' \ - 'A separate review or audit is allowed only when the captain explicitly requests that deliverable or the authorized task is a knowledge-only review; one named question remains scoped to that question.' \ - 'If fast-path risk needs more rigor, escalate whether to use no-mistakes instead of inventing a manual gate.' \ - '**local-only** has the worker stop with a clean ready branch, then waits for the configured merge authority' \ - 'A status line is a wake event, not current state' \ - 'keep exactly one live supervision cycle' \ - 'Never broadly kill watchers' \ - 'While `state/.afk` exists, the daemon owns supervision' \ - 'post the final completion follow-up before teardown'; do - assert_grep "$phrase" "$AGENTS" "compressed AGENTS.md lost safety phrase '$phrase'" - done - assert_no_grep 'Firstmate does not personally review code or deliverables' "$AGENTS" \ - "AGENTS.md retained the weaker duplicate review prohibition" - assert_no_grep 'firstmate reviews your branch' "$AGENTS" \ - "AGENTS.md retained a personal branch-review requirement" - assert_no_grep 'firstmate reviews, captain approves' "$BRIEF" \ - "generated brief retained a stacked personal-review requirement" - if grep -q "$(printf '\342\200\224')" "$AGENTS"; then - fail "AGENTS.md contains an em dash" - fi - pass "compressed AGENTS.md retains authority, supervision, AFK, and X safety" -} - -test_new_skill_metadata_and_triggers -test_diagnostic_owner_covers_causal_procedure -test_project_management_owner_covers_guarded_operations -test_generic_effort_fallback_respects_precedence -test_shared_authoring_requirements_are_owned -test_secondmate_registry_contract_stays_concise -test_state_startup_and_ordinary_recovery_placement -test_compressed_agents_owner_map -test_intake_reuses_evidence_and_parallelizes_safe_work -test_compressed_agents_retains_authority_and_supervision_safety diff --git a/tests/fm-kimi-harness.test.sh b/tests/fm-kimi-harness.test.sh new file mode 100755 index 00000000000..8e27052d8ce --- /dev/null +++ b/tests/fm-kimi-harness.test.sh @@ -0,0 +1,676 @@ +#!/usr/bin/env bash +# Behavior tests for the verified Kimi Code CLI crewmate adapter. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +SPAWN="$ROOT/bin/fm-spawn.sh" +TEARDOWN="$ROOT/bin/fm-teardown.sh" +KIMI_HOOK="$ROOT/bin/fm-kimi-turnend-hook.sh" +TMP_ROOT=$(fm_test_tmproot fm-kimi-harness) +KIMI_RUNTIME_TASK_TMP= +PYTHON_BIN=$(command -v python3) || fail "test needs python3" +PYTHON_BIN_DIR=$(dirname "$PYTHON_BIN") +JQ_BIN=$(command -v jq) || fail "test needs jq" +BASE_PATH=${FM_TEST_BASE_PATH:-$PYTHON_BIN_DIR:/usr/bin:/bin:/usr/sbin:/sbin} + +cleanup_kimi_harness() { + [ -z "$KIMI_RUNTIME_TASK_TMP" ] || rm -rf "$KIMI_RUNTIME_TASK_TMP" + rm -rf "$TMP_ROOT" +} +trap cleanup_kimi_harness EXIT + +make_spawn_fakebin() { + local dir=$1 fakebin + fakebin=$(fm_fakebin "$dir") + cat > "$fakebin/tmux" <<'SH' +#!/usr/bin/env bash +set -u +printf '%s\n' "$*" >> "$FM_FAKE_TMUX_CALL_LOG" +state=$(cat "$FM_FAKE_KIMI_STATE" 2>/dev/null || true) +fake_screen() { + case "$state" in + ready) + printf 'Welcome to Kimi Code!\ncontext: 0%% (0/256k)\n╭────────────────────────────────╮\n│ > │\n╰────────────────────────────────╯\n' + ;; + pointer-typed) + printf 'context: 0%% (0/256k)\n╭────────────────────────────────╮\n│ > Read the brief and follow it │\n│ │\n╰────────────────────────────────╯\n' + ;; + delivered) + printf '✨ Read the brief at %s and follow it exactly.\ncontext: 1%% (2k/256k)\n╭────────────────────────────────╮\n│ > │\n╰────────────────────────────────╯\n' "$FM_FAKE_BRIEF_REAL" + ;; + *) + printf 'shell starting\n$ \n' + ;; + esac +} +fake_cursor_y() { + case "$state" in + pointer-typed) printf '3\n' ;; + ready|delivered) printf '3\n' ;; + *) printf '1\n' ;; + esac +} +case "$*" in + *"#{pane_current_path}"*) printf '%s\n' "$FM_FAKE_PANE_PATH"; exit 0 ;; + *"#{cursor_y}"*) fake_cursor_y; exit 0 ;; +esac +case "${1:-}" in + display-message) printf 'firstmate\n'; exit 0 ;; + list-windows) exit 0 ;; + has-session|new-session|new-window|kill-window) exit 0 ;; + send-keys) + prev= + literal= + for arg in "$@"; do + if [ "$prev" = -l ]; then literal=$arg; break; fi + prev=$arg + done + if [ -n "$literal" ]; then + case "$literal" in + *' --auto') + printf '%s\n' "$literal" >> "$FM_FAKE_LAUNCH_LOG" + printf 'launched\n' > "$FM_FAKE_KIMI_STATE" + ;; + *) + printf '%s\n' "$literal" >> "$FM_FAKE_POINTER_LOG" + printf 'pointer-typed\n' > "$FM_FAKE_KIMI_STATE" + ;; + esac + exit 0 + fi + case " $* " in + *' Enter '*) + case "$state" in + launched) + if [ "${FM_FAKE_KIMI_READY:-yes}" = yes ]; then + printf 'ready\n' > "$FM_FAKE_KIMI_STATE" + fi + ;; + pointer-typed) + if [ "${FM_FAKE_KIMI_DELIVERY:-yes}" = yes ]; then + if [ "${FM_FAKE_KIMI_SWALLOW_FIRST:-no}" = yes ] \ + && [ ! -f "$FM_FAKE_KIMI_SWALLOWED" ]; then + : > "$FM_FAKE_KIMI_SWALLOWED" + else + printf 'delivered\n' > "$FM_FAKE_KIMI_STATE" + fi + else + printf 'ready\n' > "$FM_FAKE_KIMI_STATE" + fi + ;; + esac + ;; + esac + exit 0 + ;; + capture-pane) + start= end= prev= + for arg in "$@"; do + case "$prev" in + -S) start=$arg ;; + -E) end=$arg ;; + esac + case "$arg" in -S|-E) prev=$arg ;; *) prev= ;; esac + done + case "$start:$end" in + *[!0-9:]*|'':*|*:'') fake_screen ;; + *) fake_screen | awk -v start="$start" -v end="$end" \ + 'NR - 1 >= start && NR - 1 <= end' ;; + esac + exit 0 + ;; +esac +exit 0 +SH + chmod +x "$fakebin/tmux" + fm_fake_exit0 "$fakebin" treehouse gh-axi gh + fm_fake_exit0 "$fakebin" kimi + ln -s "$JQ_BIN" "$fakebin/jq" + printf '%s\n' "$fakebin" +} + +make_spawn_case() { + local name=$1 id=$2 case_dir home proj wt fakebin + case_dir="$TMP_ROOT/$name" + home="$case_dir/home" + proj="$case_dir/project" + wt="$case_dir/wt" + fakebin=$(make_spawn_fakebin "$case_dir/fake") + mkdir -p "$home/data/$id" "$home/projects" "$home/state" "$home/config" "$home/.kimi-code" + printf '# Kimi test config\ndefault_model = "test"\n' > "$home/.kimi-code/config.toml" + printf 'brief for kimi\n' > "$home/data/$id/brief.md" + printf 'kimi\n' > "$home/config/crew-harness" + fm_git_worktree "$proj" "$wt" "wt-$name" + touch "$home/state/.last-watcher-beat" + : > "$case_dir/launch.log" + : > "$case_dir/pointer.log" + : > "$case_dir/kimi.state" + : > "$case_dir/tmux-calls.log" + printf '%s\n' "$case_dir|$home|$proj|$wt|$fakebin" +} + +run_spawn() { + local case_dir=$1 home=$2 proj=$3 wt=$4 fakebin=$5 id=$6 + shift 6 + HOME="$home" FM_ROOT_OVERRIDE='' FM_HOME="$home" \ + FM_STATE_OVERRIDE="$home/state" FM_DATA_OVERRIDE="$home/data" \ + FM_PROJECTS_OVERRIDE="$home/projects" FM_CONFIG_OVERRIDE="$home/config" \ + FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$wt" TMUX="fake,1,0" \ + FM_FAKE_LAUNCH_LOG="$case_dir/launch.log" \ + FM_FAKE_POINTER_LOG="$case_dir/pointer.log" \ + FM_FAKE_KIMI_STATE="$case_dir/kimi.state" \ + FM_FAKE_KIMI_SWALLOWED="$case_dir/kimi.swallowed" \ + FM_FAKE_KIMI_SWALLOW_FIRST="${FM_FAKE_KIMI_SWALLOW_FIRST:-no}" \ + FM_FAKE_TMUX_CALL_LOG="$case_dir/tmux-calls.log" \ + FM_FAKE_BRIEF_REAL="$(cd "$home/data/$id" && pwd -P)/brief.md" \ + FM_KIMI_READY_POLLS=2 FM_KIMI_DELIVERY_POLLS=2 FM_KIMI_POLL_INTERVAL=0 \ + PATH="$fakebin:$BASE_PATH" \ + "$SPAWN" "$id" "$proj" --harness kimi "$@" 2>&1 +} + +read_spawn_record() { + IFS='|' read -r CASE_DIR HOME_DIR PROJ_DIR WT_DIR FAKEBIN_DIR <<EOF +$1 +EOF +} + +test_kimi_launch_then_send_is_verified() { + local id rec out rc launch pointer brief_real meta task_tmp + id="kimi-success-z1-$$" + task_tmp="/tmp/fm-$id" + KIMI_RUNTIME_TASK_TMP=$task_tmp + rm -rf "$task_tmp" + rec=$(make_spawn_case success "$id") + read_spawn_record "$rec" + out=$(FM_FAKE_KIMI_SWALLOW_FIRST=yes run_spawn \ + "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id" \ + --model kimi-code/k3 --effort high) + rc=$? + expect_code 0 "$rc" "verified kimi launch-then-send should succeed" + assert_contains "$out" "spawned $id harness=kimi" "kimi spawn did not report success" + + launch=$(cat "$CASE_DIR/launch.log") + [ "$launch" = "'$FAKEBIN_DIR/kimi' --model 'kimi-code/k3' --auto" ] \ + || fail "kimi launch did not use the absolute binary, model, and --auto only: $launch" + assert_not_contains "$launch" "--effort" "kimi launch emitted a nonexistent effort flag" + assert_not_contains "$launch" "turn-ended" "kimi launch embedded a turn-end path" + assert_not_contains "$launch" "__TURNEND__" "kimi launch retained a turn-end placeholder" + + brief_real="$(cd "$HOME_DIR/data/$id" && pwd -P)/brief.md" + pointer=$(cat "$CASE_DIR/pointer.log") + [ "$pointer" = "Read the brief at $brief_real and follow it exactly." ] \ + || fail "kimi pointer was not the exact absolute-path-only instruction: $pointer" + meta="$HOME_DIR/state/$id.meta" + assert_grep 'model=kimi-code/k3' "$meta" "kimi meta lost the requested model" + assert_grep 'effort=high' "$meta" "kimi meta did not retain the unsupported effort axis" + assert_grep "tasktmp=$task_tmp" "$meta" "kimi meta did not record its task temp root" + assert_present "$task_tmp/gotmp" "kimi spawn did not create its Go temp directory" + assert_grep "export GOTMPDIR=$task_tmp/gotmp" "$CASE_DIR/tmux-calls.log" \ + "kimi spawn did not export its Go temp directory into the pane" + assert_grep 'BEGIN FIRSTMATE KIMI TURN-END HOOK' "$HOME_DIR/.kimi-code/config.toml" \ + "kimi spawn did not install its guarded global hook region" + assert_grep 'token=' "$WT_DIR/.fm-kimi-turnend" "kimi spawn did not write its token pointer" + assert_present "$HOME_DIR/state/$id.kimi-turnend-token" "kimi spawn did not record its token" + pass "fm-spawn: kimi launches, delivers its brief, and registers a guarded turn-end token" +} + +test_kimi_hook_install_is_surgical_idempotent_and_removable() { + local home config original once stripped count + home="$TMP_ROOT/config-surgery" + config="$home/.kimi-code/config.toml" + original="$home/original.toml" + once="$home/once.toml" + stripped="$home/stripped.toml" + mkdir -p "$home/.kimi-code" + cat > "$config" <<'EOF' +# Captain's leading comment stays exactly here. + +[ui] +theme = "night" # inline comment +show_usage = true + +# Foreign hook with intentionally unusual key ordering. +[[hooks]] +timeout=17 +command = "printf foreign" +matcher="" +event = "Stop" + +[providers.example] +model = "some/model" +# Final comment and blank line follow. + +EOF + cp "$config" "$original" + + HOME="$home" "$KIMI_HOOK" install || fail "Kimi hook install refused a realistic config" + cp "$config" "$once" + HOME="$home" "$KIMI_HOOK" install || fail "second Kimi hook install failed" + cmp -s "$once" "$config" || fail "second Kimi hook install changed config bytes" + count=$(grep -c '^# BEGIN FIRSTMATE KIMI TURN-END HOOK' "$config") + [ "$count" -eq 1 ] || fail "idempotent install left $count Firstmate regions" + + HOME="$home" "$KIMI_HOOK" remove || fail "Kimi hook removal failed" + cp "$config" "$stripped" + cmp -s "$original" "$stripped" \ + || fail "config with the Firstmate region excised was not byte-identical to the original" + assert_absent "$home/.kimi-code/fm-turn-end.sh" "removal left the Firstmate hook script" + assert_absent "$home/.kimi-code/fm-turn-end.d" "removal left the Firstmate registry" + pass "Kimi hook install is idempotent and removal restores every foreign config byte" +} + +test_kimi_hook_remove_preserves_owned_newline_boundary() { + local appended config expected home original + home="$TMP_ROOT/config-owned-newline" + config="$home/.kimi-code/config.toml" + original="$home/original.toml" + expected="$home/expected.toml" + appended="$home/appended.toml" + mkdir -p "$home/.kimi-code" + printf 'default_model = "test"' > "$config" + cp "$config" "$original" + + HOME="$home" "$KIMI_HOOK" install || fail "Kimi hook install refused config without a final newline" + HOME="$home" "$KIMI_HOOK" remove || fail "Kimi hook removal failed without appended config" + cmp -s "$original" "$config" \ + || fail "pristine removal did not restore the absent final newline byte-identically" + + HOME="$home" "$KIMI_HOOK" install || fail "second Kimi hook install refused config without a final newline" + printf '[captain]\nenabled = true\n' > "$appended" + cat "$appended" >> "$config" + HOME="$home" "$KIMI_HOOK" remove || fail "Kimi hook removal joined config appended after its region" + { + cat "$original" + printf '\n' + cat "$appended" + } > "$expected" + cmp -s "$expected" "$config" \ + || fail "removal did not preserve appended captain config on its own line" + "$PYTHON_BIN" - "$config" <<'PY' || fail "config with appended captain TOML did not parse after removal" +import sys +import tomllib + +with open(sys.argv[1], "rb") as stream: + tomllib.load(stream) +PY + pass "Kimi hook removal preserves owned newline boundaries and pristine bytes" +} + +test_kimi_hook_fails_closed_on_missing_malformed_or_partial_config() { + local missing malformed partial out rc + missing="$TMP_ROOT/config-missing" + malformed="$TMP_ROOT/config-malformed" + partial="$TMP_ROOT/config-partial" + mkdir -p "$missing/.kimi-code" "$malformed/.kimi-code" "$partial/.kimi-code" + + rc=0 + out=$(HOME="$missing" "$KIMI_HOOK" install 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "missing Kimi config was accepted" + assert_contains "$out" "Kimi config is missing" "missing config refusal lacked its concrete reason" + assert_absent "$missing/.kimi-code/fm-turn-end.sh" "missing config refusal wrote the hook script" + + printf '[broken\n' > "$malformed/.kimi-code/config.toml" + cp "$malformed/.kimi-code/config.toml" "$malformed/before" + rc=0 + out=$(HOME="$malformed" "$KIMI_HOOK" install 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "malformed Kimi config was accepted" + assert_contains "$out" "malformed TOML" "malformed config refusal lacked its concrete reason" + cmp -s "$malformed/before" "$malformed/.kimi-code/config.toml" \ + || fail "malformed config refusal changed config bytes" + assert_absent "$malformed/.kimi-code/fm-turn-end.sh" "malformed config refusal wrote the hook script" + + printf '# BEGIN FIRSTMATE KIMI TURN-END HOOK\n' > "$partial/.kimi-code/config.toml" + cp "$partial/.kimi-code/config.toml" "$partial/before" + rc=0 + out=$(HOME="$partial" "$KIMI_HOOK" install 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "partial Firstmate marker was accepted" + assert_contains "$out" "partial, duplicated, or altered" "partial marker refusal lacked its concrete reason" + cmp -s "$partial/before" "$partial/.kimi-code/config.toml" \ + || fail "partial marker refusal changed config bytes" + pass "Kimi hook install refuses missing, malformed, and surprising config without writing" +} + +test_kimi_hook_install_refuses_without_jq() { + local home config before fakebin out rc + home="$TMP_ROOT/config-no-jq" + config="$home/.kimi-code/config.toml" + before="$home/config-before.toml" + fakebin=$(fm_fakebin "$home/no-jq") + mkdir -p "$home/.kimi-code" + printf '# Captain config\nmodel = "test"\n' > "$config" + cp "$config" "$before" + ln -s "$(command -v bash)" "$fakebin/bash" + ln -s "$(command -v python3)" "$fakebin/python3" + + rc=0 + out=$(HOME="$home" PATH="$fakebin" "$KIMI_HOOK" install 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "Kimi hook install succeeded without jq" + assert_contains "$out" "jq is required" "missing-jq refusal did not name jq" + cmp -s "$before" "$config" || fail "missing-jq refusal changed config bytes" + assert_absent "$home/.kimi-code/fm-turn-end.sh" "missing-jq refusal wrote the hook script" + assert_absent "$home/.kimi-code/fm-turn-end.d" "missing-jq refusal wrote the registry" + pass "Kimi hook install refuses without jq before any config write" +} + +test_kimi_hook_is_silent_and_requires_registered_workspace_token() { + local id rec out rc hook target token no_token snapshot_before snapshot_after fakebin + id=kimi-hook-auth-z6 + rec=$(make_spawn_case hook-auth "$id") + read_spawn_record "$rec" + out=$(run_spawn "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id") + rc=$? + expect_code 0 "$rc" "Kimi spawn should succeed before hook authentication checks" + hook="$HOME_DIR/.kimi-code/fm-turn-end.sh" + target="$HOME_DIR/state/$id.turn-ended" + token=$(sed -n 's/^token=//p' "$WT_DIR/.fm-kimi-turnend") + assert_present "$HOME_DIR/.kimi-code/fm-turn-end.d/$token" "Kimi registry token is missing" + + no_token="$CASE_DIR/no-token-workspace" + mkdir -p "$no_token" + snapshot_before=$(find "$no_token" -mindepth 1 -print) + out=$(printf '{"hook_event_name":"Stop","session_id":"ordinary","cwd":"%s","stop_hook_active":false}\n' "$no_token" \ + | HOME="$HOME_DIR" bash "$hook" 2>&1) + rc=$? + expect_code 0 "$rc" "Kimi hook must never block a tokenless session" + [ -z "$out" ] || fail "Kimi hook printed into a tokenless session: $out" + snapshot_after=$(find "$no_token" -mindepth 1 -print) + [ "$snapshot_before" = "$snapshot_after" ] || fail "Kimi hook wrote inside a tokenless workspace" + assert_absent "$target" "tokenless Kimi hook invocation touched a task marker" + + printf 'token=%s\n' "$token" > "$WT_DIR/.fm-kimi-turnend" + out=$(printf '{"hook_event_name":"Stop","session_id":"crew","cwd":"%s","stop_hook_active":false}\n' "$WT_DIR" \ + | HOME="$HOME_DIR" bash "$hook" 2>&1) + rc=$? + expect_code 0 "$rc" "registered Kimi hook invocation did not exit zero" + [ -z "$out" ] || fail "registered Kimi hook invocation printed output: $out" + assert_present "$target" "registered Kimi hook invocation did not touch the turn-end marker" + + rm "$target" + fakebin=$(fm_fakebin "$CASE_DIR/no-jq") + ln -s "$(command -v bash)" "$fakebin/bash" + out=$(printf '{"hook_event_name":"Stop","session_id":"crew","cwd":"%s","stop_hook_active":false}\n' "$WT_DIR" \ + | HOME="$HOME_DIR" PATH="$fakebin" "$hook" 2>&1) + rc=$? + expect_code 0 "$rc" "Kimi hook without jq must still exit zero" + [ -z "$out" ] || fail "Kimi hook without jq printed output: $out" + assert_absent "$target" "Kimi hook without jq touched the turn-end marker" + pass "Kimi hook stays silent and inert without a Firstmate registry token" +} + +test_kimi_spawn_refuses_unsafe_global_config_before_pane_creation() { + local id rec out rc + id=kimi-config-refuse-z7 + rec=$(make_spawn_case config-refuse "$id") + read_spawn_record "$rec" + printf '[malformed\n' > "$HOME_DIR/.kimi-code/config.toml" + rc=0 + out=$(run_spawn "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id") || rc=$? + [ "$rc" -ne 0 ] || fail "Kimi spawn accepted malformed global config" + assert_contains "$out" "malformed TOML" "Kimi spawn omitted the concrete config refusal" + if grep -Eq '(^| )new-(session|window)( |$)' "$CASE_DIR/tmux-calls.log"; then + fail "unsafe Kimi config refusal created a tmux container or pane" + fi + pass "fm-spawn: unsafe Kimi global config refuses before pane creation" +} + +test_kimi_teardown_removes_pointer_and_registry_token() { + local id rec out rc token + id=kimi-teardown-z8 + rec=$(make_spawn_case teardown "$id") + read_spawn_record "$rec" + out=$(run_spawn "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id") + rc=$? + expect_code 0 "$rc" "Kimi spawn should succeed before teardown" + token=$(sed -n 's/^token=//p' "$WT_DIR/.fm-kimi-turnend") + + HOME="$HOME_DIR" FM_ROOT_OVERRIDE="$ROOT" FM_HOME="$HOME_DIR" \ + FM_STATE_OVERRIDE="$HOME_DIR/state" FM_DATA_OVERRIDE="$HOME_DIR/data" \ + FM_PROJECTS_OVERRIDE="$HOME_DIR/projects" FM_CONFIG_OVERRIDE="$HOME_DIR/config" \ + FM_SPAWN_NO_GUARD=1 PATH="$FAKEBIN_DIR:$BASE_PATH" \ + "$TEARDOWN" "$id" --force >/dev/null 2>&1 || fail "Kimi teardown failed" + assert_absent "$WT_DIR/.fm-kimi-turnend" "Kimi token pointer survived teardown" + assert_absent "$HOME_DIR/.kimi-code/fm-turn-end.d/$token" "Kimi registry token survived teardown" + assert_absent "$HOME_DIR/state/$id.kimi-turnend-token" "Kimi token state survived teardown" + pass "fm-teardown: Kimi task pointer and registry token are removed" +} + +test_kimi_falls_back_to_expanded_home_binary() { + local id rec out rc launch fallback + id=kimi-fallback-z4 + rec=$(make_spawn_case fallback "$id") + read_spawn_record "$rec" + rm "$FAKEBIN_DIR/kimi" + fallback="$HOME_DIR/.kimi-code/bin/kimi" + mkdir -p "$(dirname "$fallback")" + fm_fake_exit0 "$(dirname "$fallback")" kimi + out=$(run_spawn "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id") + rc=$? + expect_code 0 "$rc" "Kimi HOME fallback spawn should succeed" + launch=$(cat "$CASE_DIR/launch.log") + [ "$launch" = "'$fallback' --auto" ] \ + || fail "Kimi fallback did not expand HOME into an absolute executable: $launch" + pass "fm-spawn: Kimi fallback expands the active HOME" +} + +test_kimi_missing_binary_refuses_before_pane_creation() { + local id rec out rc fallback + id=kimi-missing-z5 + rec=$(make_spawn_case missing "$id") + read_spawn_record "$rec" + rm "$FAKEBIN_DIR/kimi" + fallback="$HOME_DIR/.kimi-code/bin/kimi" + rc=0 + out=$(run_spawn "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id") || rc=$? + [ "$rc" -ne 0 ] || fail "missing Kimi executable should refuse the spawn" + assert_contains "$out" "searched PATH for 'kimi'" "missing Kimi diagnostic omitted PATH" + assert_contains "$out" "fallback '$fallback'" "missing Kimi diagnostic omitted expanded fallback" + if grep -Eq '(^| )new-(session|window)( |$)' "$CASE_DIR/tmux-calls.log"; then + fail "missing Kimi executable created a tmux container or pane" + fi + pass "fm-spawn: missing Kimi executable refuses before pane creation" +} + +test_kimi_unconfirmed_delivery_fails_loudly() { + local id rec out rc + id=kimi-drop-z2 + rec=$(make_spawn_case drop "$id") + read_spawn_record "$rec" + rc=0 + out=$(FM_FAKE_KIMI_DELIVERY=no run_spawn \ + "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id") || rc=$? + [ "$rc" -ne 0 ] || fail "an unconfirmed kimi delivery should fail" + assert_contains "$out" "kimi brief pointer delivery was not confirmed" \ + "unconfirmed kimi delivery lacked a loud diagnostic" + assert_grep 'failed: kimi brief pointer delivery was not confirmed' "$HOME_DIR/state/$id.status" \ + "unconfirmed kimi delivery did not leave a supervisor-visible failure" + pass "fm-spawn: kimi treats a silent pointer drop as a failed spawn" +} + +test_kimi_readiness_gate_precedes_pointer() { + local id rec out rc + id=kimi-not-ready-z3 + rec=$(make_spawn_case not-ready "$id") + read_spawn_record "$rec" + rc=0 + out=$(FM_FAKE_KIMI_READY=no run_spawn \ + "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id") || rc=$? + [ "$rc" -ne 0 ] || fail "kimi spawn without a ready signal should fail" + assert_contains "$out" "kimi did not show a verified ready signal" \ + "kimi readiness failure lacked a loud diagnostic" + [ ! -s "$CASE_DIR/pointer.log" ] || fail "kimi pointer was sent before readiness" + pass "fm-spawn: kimi never sends the brief pointer before an observable ready signal" +} + +test_kimi_detection_uses_ancestry_after_markers() { + local dir fakebin cfg out + dir="$TMP_ROOT/detection" + fakebin=$(fm_fakebin "$dir") + cfg="$dir/config" + mkdir -p "$cfg" + cat > "$fakebin/ps" <<'SH' +#!/usr/bin/env bash +set -u +field= +pid= +prev= +for arg in "$@"; do + [ "$prev" = -o ] && field=$arg + [ "$prev" = -p ] && pid=$arg + prev=$arg +done +case "$field:$pid" in + comm=:4242) printf '/opt/kimi/bin/kimi\n' ;; + comm=:*) printf '/bin/bash\n' ;; + ppid=:4242) printf '1\n' ;; + ppid=:*) printf '4242\n' ;; + args=:*) printf 'bash\n' ;; +esac +SH + chmod +x "$fakebin/ps" + + out=$(env -u CLAUDECODE -u PI_CODING_AGENT -u GROK_AGENT \ + PATH="$fakebin:$BASE_PATH" FM_CONFIG_OVERRIDE="$cfg" "$ROOT/bin/fm-harness.sh") + [ "$out" = kimi ] || fail "kimi ancestry detection returned '$out'" + out=$(CLAUDECODE=1 PATH="$fakebin:$BASE_PATH" FM_CONFIG_OVERRIDE="$cfg" "$ROOT/bin/fm-harness.sh") + [ "$out" = claude ] || fail "verified env-marker precedence changed, got '$out'" + pass "fm-harness: markerless kimi is detected by ancestry after env-marker precedence" +} + +test_kimi_session_lock_identity() { + local home fakebin out + home="$TMP_ROOT/session-lock-home" + fakebin=$(fm_fakebin "$TMP_ROOT/session-lock-fake") + mkdir -p "$home/state" + cat > "$fakebin/ps" <<'SH' +#!/usr/bin/env bash +case "$*" in + *"comm="*) printf '%s\n' '/opt/kimi/bin/kimi'; exit 0 ;; + *"args="*) printf '%s\n' 'kimi'; exit 0 ;; +esac +exit 1 +SH + chmod +x "$fakebin/ps" + + FM_HOME="$home" PATH="$fakebin:$BASE_PATH" "$ROOT/bin/fm-lock.sh" \ + || fail "fm-lock did not acquire from Kimi ancestry" + case "$(cat "$home/state/.lock")" in + ''|*[!0-9]*) fail "fm-lock did not record the Kimi harness ancestor" ;; + esac + printf '%s\n' "$$" > "$home/state/.lock" + out=$(FM_HOME="$home" PATH="$fakebin:$BASE_PATH" "$ROOT/bin/fm-lock.sh" status) + assert_contains "$out" "lock: held by live harness pid" \ + "fm-lock did not recognize Kimi as a live holder" + pass "fm-lock recognizes Kimi ancestry and live lock holders" +} + +test_kimi_busy_signature_is_scoped_to_spinner_lines() { + local capture + # shellcheck source=/dev/null + . "$ROOT/bin/fm-tmux-lib.sh" + unset FM_BUSY_REGEX + capture="$TMP_ROOT/busy-pane" + tmux() { + case "${1:-}" in + capture-pane) cat "$capture" ;; + *) return 0 ;; + esac + } + # These fixtures reproduce the observed spinner shape rather than byte-exact + # transcriptions. Leading whitespace is deliberately varied; separator whitespace + # follows the captured contract. + local phase + for phase in 🌑 🌒 🌓 🌔 🌕 🌖 🌗 🌘; do + printf ' %s · Tip: Kimi is working\n│ > │\n' "$phase" > "$capture" + fm_pane_is_busy fake kimi || fail "Kimi spinner phase $phase was not recognized as busy" + done + printf 'ordinary response ending with 🌕\n│ > │\n' > "$capture" + if fm_pane_is_busy fake kimi; then + fail "a moon outside Kimi's spinner-line shape was misread as busy" + fi + printf '🌕 Full moon details\n│ > │\n' > "$capture" + if fm_pane_is_busy fake kimi; then + fail "moon-led Kimi output without the middot separator was misread as busy" + fi + printf ' 🌗 · Tip: /plugins: manage plugins ...\n│ > │\n' > "$capture" + if fm_pane_is_busy fake codex; then + fail "Kimi's real spinner signature leaked into another harness" + fi + printf 'tip: ctrl+c: cancel\n│ > │\n' > "$capture" + if fm_pane_is_busy fake kimi; then + fail "kimi's independently rotating idle tip was misread as busy" + fi + printf 'Ctrl+c:cancel\n│ > │\n' > "$capture" + if fm_pane_is_busy fake kimi; then + fail "Grok's exact busy token leaked into Kimi's harness-scoped matcher" + fi + printf 'auto K2.7 Coding thinking /some/path\n│ > │\n' > "$capture" + if fm_pane_is_busy fake kimi; then + fail "Kimi's idle thinking-effort status label was misread as busy" + fi + pass "busy detection: real Kimi moon-plus-middot captures require its harness while idle labels stay idle" +} + +test_watcher_scopes_moon_spinner_to_recorded_kimi_task() ( + local state="$TMP_ROOT/watch-state" busy_capture=' 🌑 · Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"' + mkdir -p "$state" + printf 'window=fake\nharness=kimi\n' > "$state/kimi-watch.meta" + unset FM_BUSY_REGEX + FM_HOME="$TMP_ROOT/watch-home" + FM_STATE_OVERRIDE="$state" + export FM_HOME FM_STATE_OVERRIDE + # shellcheck source=/dev/null + . "$ROOT/bin/fm-watch.sh" + # shellcheck disable=SC2329 # Runtime override called by the sourced watcher. + fm_backend_busy_state() { printf 'unknown'; } + window_is_busy fake "$busy_capture" \ + || fail "fm-watch did not recognize the real Kimi spinner-line shape" + printf 'window=fake\nharness=codex\n' > "$state/kimi-watch.meta" + if window_is_busy fake "$busy_capture"; then + fail "fm-watch applied Kimi's real spinner signature to a recorded Codex task" + fi + printf 'window=fake\nharness=kimi\n' > "$state/kimi-watch.meta" + if window_is_busy fake 'ordinary response ending with 🌕'; then + fail "fm-watch treated an ordinary Kimi moon as a spinner line" + fi + if window_is_busy fake '🌕 Full moon details'; then + fail "fm-watch treated moon-led Kimi output without the middot separator as busy" + fi + if window_is_busy fake 'auto K2.7 Coding thinking /some/path'; then + fail "fm-watch treated Kimi's idle thinking-effort status label as busy" + fi + if window_is_busy fake 'Ctrl+c:cancel'; then + fail "fm-watch let Grok's exact busy token classify a recorded Kimi task busy" + fi + pass "fm-watch: Kimi spinner matching is metadata-scoped and ignores Grok's busy token" +) + +test_kimi_bordered_prompt_needs_no_override() { + local out + # shellcheck source=/dev/null + . "$ROOT/bin/fm-composer-lib.sh" + out=$(fm_composer_classify_content 1 '>') + [ "$out" = empty ] || fail "kimi's bordered bare > composer should read empty, got '$out'" + out=$(fm_composer_classify_content 0 '>') + [ "$out" = unknown ] || fail "an unbordered dead-shell > must stay unknown, got '$out'" + pass "composer classifier: kimi's existing bordered > shape is already safe without an override" +} + +test_kimi_hook_install_is_surgical_idempotent_and_removable +test_kimi_hook_remove_preserves_owned_newline_boundary +test_kimi_hook_fails_closed_on_missing_malformed_or_partial_config +test_kimi_hook_install_refuses_without_jq +test_kimi_launch_then_send_is_verified +test_kimi_hook_is_silent_and_requires_registered_workspace_token +test_kimi_spawn_refuses_unsafe_global_config_before_pane_creation +test_kimi_teardown_removes_pointer_and_registry_token +test_kimi_falls_back_to_expanded_home_binary +test_kimi_missing_binary_refuses_before_pane_creation +test_kimi_unconfirmed_delivery_fails_loudly +test_kimi_readiness_gate_precedes_pointer +test_kimi_detection_uses_ancestry_after_markers +test_kimi_session_lock_identity +test_kimi_busy_signature_is_scoped_to_spinner_lines +test_watcher_scopes_moon_spinner_to_recorded_kimi_task +test_kimi_bordered_prompt_needs_no_override diff --git a/tests/fm-lint.test.sh b/tests/fm-lint.test.sh index 4a1b18d7dcb..17fb097f758 100755 --- a/tests/fm-lint.test.sh +++ b/tests/fm-lint.test.sh @@ -18,11 +18,7 @@ set -u . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" LINT="$ROOT/bin/fm-lint.sh" -CI="$ROOT/.github/workflows/ci.yml" -NM="$ROOT/.no-mistakes.yaml" INSTALLER="$ROOT/bin/fm-install-shellcheck.sh" -# The authoritative file set the one owner must run. -CANON='ROOTS=(bin/*.sh bin/backends/*.sh tests/*.sh)' # The pinned version, read from the single source (the one owner itself). REQUIRED=$("$LINT" --required-version) @@ -33,34 +29,13 @@ pinned_ready() { [ "$(shellcheck --version | awk '/^version:/ {print $2; exit}')" = "$REQUIRED" ] } -test_owner_exists_and_executable() { - assert_present "$LINT" "bin/fm-lint.sh is missing" - [ -x "$LINT" ] || fail "bin/fm-lint.sh must be executable so CI/gate can run it directly" - pass "one-owner lint script exists and is executable" -} - -test_owner_defines_canonical_set() { - assert_grep "$CANON" "$LINT" "fm-lint.sh must run the canonical shellcheck file set" - # It must not weaken CI: no severity downgrade and no blanket disable/exclude - # that would hide findings CI fails on. - assert_no_grep '--severity' "$LINT" "fm-lint.sh must not lower severity below the CI default" - assert_no_grep '--exclude' "$LINT" "fm-lint.sh must not blanket-exclude checks CI enforces" - assert_grep "\"\$FM_LINT_SHELLCHECK\" --norc --external-sources -- \"\${roots[@]}\"" "$LINT" "every bounded worker must ignore ambient config and preserve annotated production sources" - [ "$(grep -Fc -- '--norc --external-sources' "$LINT")" -eq 1 ] || fail "the one worker command must own ShellCheck configuration" - assert_grep "JOBS=\${FM_LINT_JOBS:-2}" "$LINT" "canonical lint must default to two bounded workers" - pass "fm-lint.sh is the sole authoritative definition at CI-default severity" -} - -test_ci_invokes_the_owner() { - grep -Eq '^ - run: bin/fm-lint\.sh$' "$CI" || fail "CI lint job must invoke the one-owner script as a run step" - # Guard against regression to an inline re-spelling of the command. - assert_no_grep 'run: shellcheck' "$CI" "CI must call fm-lint.sh, not re-spell shellcheck inline" - pass "CI lint job calls the one-owner script, not an inline command" -} - -test_nomistakes_invokes_the_owner() { - grep -Fqx " lint: 'bin/fm-lint.sh'" "$NM" || fail "no-mistakes commands.lint must map exactly to the one-owner script" - pass "no-mistakes pre-push lint calls the one-owner script" +test_list_files_reports_the_shell_inventory() { + local listed expected + listed=$("$LINT" --list-files) + expected=$(find bin bin/backends tests -maxdepth 1 -type f -name '*.sh' -print | LC_ALL=C sort) + [ "$(printf '%s\n' "$listed" | LC_ALL=C sort)" = "$expected" ] \ + || fail "fm-lint.sh --list-files did not return the complete shell inventory" + pass "fm-lint.sh --list-files reports the complete shell inventory" } test_pins_an_explicit_version() { @@ -71,17 +46,6 @@ test_pins_an_explicit_version() { pass "fm-lint.sh pins an explicit ShellCheck version ($REQUIRED)" } -test_ci_installs_and_logs_the_pinned_version() { - # CI must derive the version from the one owner (never hardcode a divergent - # number) and log the resolved version as parity evidence. - assert_grep "VERSION=\"\$(\"\$ROOT/bin/fm-lint.sh\" --required-version)\"" "$INSTALLER" "installer must read the version fm-lint.sh pins" - [ "$(grep -Fc "bin/fm-install-shellcheck.sh \"\$RUNNER_TEMP/bin\"" "$CI")" -eq 4 ] || fail "lint and all three portable behavior jobs must use the shared ShellCheck installer" - assert_grep "ACTUAL_SHA256=\$(sha256sum" "$INSTALLER" "installer must calculate the ShellCheck archive checksum" - assert_grep "[ \"\$ACTUAL_SHA256\" = \"\$SHA256\" ]" "$INSTALLER" "installer must verify the ShellCheck archive checksum" - assert_grep "\"\$DESTINATION/shellcheck\" --version" "$INSTALLER" "installer must log the resolved ShellCheck version as evidence" - pass "CI installs and logs the pinned ShellCheck version from the one owner" -} - test_installer_retries_transient_download_failure() { local tmp fakebin destination out tmp=$(fm_test_tmproot fm-shellcheck-download) @@ -238,26 +202,6 @@ SH pass "fm-lint.sh passes a clean fixture" } -test_source_graph_boundaries_keep_every_owner() { - local adapter file production_context_tests="" - [ "$(grep -Fc '# shellcheck source=/dev/null' "$ROOT/bin/fm-backend.sh")" -eq 5 ] \ - || fail "the dispatcher must stop static source following at all five dynamic adapters" - for adapter in tmux herdr zellij orca cmux; do - assert_present "$ROOT/bin/backends/$adapter.sh" "canonical adapter root is missing: $adapter" - done - assert_present "$ROOT/bin/fm-push-transition-lib.sh" "narrow push-transition owner is missing" - assert_grep '# shellcheck source=bin/fm-push-transition-lib.sh' "$ROOT/bin/fm-watch.sh" "the watcher must consume the narrow push-transition owner" - assert_grep ". \"\$ROOT/bin/fm-push-transition-lib.sh\"" "$ROOT/tests/fm-backend-herdr-eventwait-smoke.test.sh" "the Herdr event-wait smoke must consume the narrow production owner" - assert_no_grep '# shellcheck source=bin/fm-watch.sh' "$ROOT/tests/fm-backend-herdr-eventwait-smoke.test.sh" "the event-wait smoke must not re-import the whole watcher graph" - for file in "$ROOT"/tests/*.sh; do - grep -q '^[[:space:]]*# shellcheck source=bin/' "$file" || continue - production_context_tests="${production_context_tests}$(basename "$file")|" - done - [ "$production_context_tests" = 'fm-backend-herdr.test.sh|fm-daemon.test.sh|fm-pending-reply.test.sh|fm-secondmate-sync.test.sh|' ] \ - || fail "only callback/variable interop tests may retain production source context: $production_context_tests" - pass "dispatcher, adapters, production owner, and tests have explicit lint boundaries" -} - test_jobs_are_deterministic_and_complete() { if ! pinned_ready; then pass "SKIP (ShellCheck $REQUIRED not resolved): deterministic bounded jobs check" @@ -482,18 +426,13 @@ SH pass "seeded dispatcher, adapter, production-owner, and test-local diagnostics preserve parity" } -test_owner_exists_and_executable -test_owner_defines_canonical_set -test_ci_invokes_the_owner -test_nomistakes_invokes_the_owner +test_list_files_reports_the_shell_inventory test_pins_an_explicit_version -test_ci_installs_and_logs_the_pinned_version test_installer_retries_transient_download_failure test_rejects_wrong_shellcheck_version test_catches_a_real_lint_defect test_ignores_ambient_shellcheck_opts test_clean_fixture_passes -test_source_graph_boundaries_keep_every_owner test_jobs_are_deterministic_and_complete test_worker_trees_stop_on_signal test_seeded_module_boundary_parity diff --git a/tests/fm-nm-test-contract.test.sh b/tests/fm-nm-test-contract.test.sh deleted file mode 100755 index 54c19eab2f8..00000000000 --- a/tests/fm-nm-test-contract.test.sh +++ /dev/null @@ -1,127 +0,0 @@ -#!/usr/bin/env bash -# Contract: local no-mistakes Test is intent-targeted; CI owns broad regression. -# -# Firstmate must not configure commands.test as a complete tests/*.test.sh walk -# (that duplicated CI and burned local pipeline time). Lint stays pinned to -# bin/fm-lint.sh. Remote CI owns broad regression through separate portable and -# required real-Herdr Behavior lanes composed around bin/fm-test-run.sh. -set -u - -# shellcheck source=tests/lib.sh -. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" - -NM="$ROOT/.no-mistakes.yaml" -CI="$ROOT/.github/workflows/ci.yml" - -test_nm_yaml_tracked() { - assert_present "$NM" "tracked .no-mistakes.yaml is missing" - git -C "$ROOT" ls-files --error-unmatch .no-mistakes.yaml >/dev/null 2>&1 \ - || fail ".no-mistakes.yaml is not tracked by git" - pass ".no-mistakes.yaml is present and tracked" -} - -test_nm_keeps_lint_pin() { - grep -Fqx " lint: 'bin/fm-lint.sh'" "$NM" \ - || fail "commands.lint must remain exactly bin/fm-lint.sh" - pass "commands.lint stays pinned to bin/fm-lint.sh" -} - -# True when the YAML maps a non-empty commands.test (string or mapping value). -# Empty / null / absent is the intended targeted-Test posture. -nm_commands_test_value() { - if command -v python3 >/dev/null 2>&1 && python3 -c 'import yaml' >/dev/null 2>&1; then - python3 -c ' -import yaml, sys -doc = yaml.safe_load(open(sys.argv[1])) or {} -cmds = doc.get("commands") or {} -val = cmds.get("test") if isinstance(cmds, dict) else None -if val is None or val is False: - print("") -elif isinstance(val, str): - print(val) -else: - print(repr(val)) -' "$NM" - return - fi - if command -v ruby >/dev/null 2>&1; then - ruby -ryaml -e ' -doc = YAML.safe_load(File.read(ARGV[0])) || {} -cmds = doc["commands"] || {} -val = cmds.is_a?(Hash) ? cmds["test"] : nil -if val.nil? || val == false - puts "" -elsif val.is_a?(String) - puts val -else - puts val.inspect -end -' "$NM" - return - fi - # Structural fallback: any commands.test line under the commands block. - awk ' - /^commands:[[:space:]]*$/ { in_cmds=1; next } - in_cmds && /^[^[:space:]#]/ { in_cmds=0 } - in_cmds && /^[[:space:]]+test:[[:space:]]*/ { - sub(/^[[:space:]]+test:[[:space:]]*/, "") - gsub(/^['\''"]|['\''"]$/, "") - print - exit - } - ' "$NM" -} - -test_nm_has_no_complete_local_test_command() { - local val - val=$(nm_commands_test_value) || fail "failed to read commands.test from .no-mistakes.yaml" - if [ -n "$val" ]; then - case "$val" in - *'tests/*.test.sh'*|*'tests/'*'.test.sh'*) - fail "commands.test must not walk the complete tests/*.test.sh suite; got: $val" - ;; - *) - # Any non-empty override still steers Test away from intent-targeted default. - fail "commands.test must be absent or empty so Test stays intent-targeted; got: $val" - ;; - esac - fi - # Also refuse a commented-out full-suite remnant that could be re-enabled by habit. - if grep -E '^[[:space:]]*#?[[:space:]]*test:[[:space:]].*tests/\*\.test\.sh' "$NM" >/dev/null 2>&1; then - fail ".no-mistakes.yaml still documents a full-suite commands.test line (active or comment)" - fi - pass "no-mistakes does not configure a complete local Test command" -} - -test_ci_still_runs_broad_behavior_suite() { - assert_present "$CI" "ci.yml is missing" - # Portable shards and the serial remainder cover every portable behavior - # script through the one owner, with a deterministic inventory guard. - grep -Fq 'bin/fm-test-run.sh --lane portable-parallel-1' "$CI" \ - || fail "CI must invoke portable parallel shard 1 through fm-test-run.sh" - grep -Fq 'bin/fm-test-run.sh --lane portable-parallel-2' "$CI" \ - || fail "CI must invoke portable parallel shard 2 through fm-test-run.sh" - grep -Fq 'bin/fm-test-run.sh --lane portable-serial' "$CI" \ - || fail "CI must invoke the portable serial remainder through fm-test-run.sh" - grep -Fq 'bin/fm-test-run.sh --check-coverage' "$CI" \ - || fail "CI must prove complete lane coverage through fm-test-run.sh" - # Guard against regression to an uninstrumented inline loop that drops timing. - if grep -Eq 'for test_script in tests/\*\.test\.sh' "$CI"; then - fail "CI Behavior must not re-spell an inline tests/*.test.sh loop; use fm-test-run.sh" - fi - # Preserve other CI lanes this task must not shrink. - grep -Eq 'name:[[:space:]]*Lint shell scripts' "$CI" \ - || fail "CI must retain the lint job" - grep -Eq 'name:[[:space:]]*Stock macOS Bash snapshot compatibility' "$CI" \ - || fail "CI must retain the macOS stock Bash compatibility job" - grep -Eq 'name:[[:space:]]*Repo invariants' "$CI" \ - || fail "CI must retain the repo invariants job" - grep -Fq 'tests-herdr:' "$CI" \ - || fail "CI must retain the required Herdr Behavior job" - pass "CI still owns partitioned broad behavior coverage and companion jobs" -} - -test_nm_yaml_tracked -test_nm_keeps_lint_pin -test_nm_has_no_complete_local_test_command -test_ci_still_runs_broad_behavior_suite diff --git a/tests/fm-no-mistakes-ownership.test.sh b/tests/fm-no-mistakes-ownership.test.sh deleted file mode 100755 index b7e7fc2a6fd..00000000000 --- a/tests/fm-no-mistakes-ownership.test.sh +++ /dev/null @@ -1,39 +0,0 @@ -#!/usr/bin/env bash -# Static contract tests for crew-owned no-mistakes validation runs. -set -u - -# shellcheck source=tests/lib.sh disable=SC1091 -. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" - -validate_contract() { - awk ' - /^### Validate$/ { found = 1; next } - found && /^### / { exit } - found { print } - ' "$ROOT/AGENTS.md" -} - -test_worker_owns_synchronous_driver() { - local contract - contract=$(validate_contract) - - assert_contains "$contract" 'The task worker that starts a no-mistakes run drives the pipeline' \ - "Validate contract does not assign the run to its initiating task worker" - assert_contains "$contract" "owns every \`no-mistakes axi run\` and \`no-mistakes axi respond\` call through the next gate or outcome" \ - "Validate contract does not assign every synchronous driver call to the task worker" - assert_contains "$contract" 'process every synchronous return until completion or a genuinely new escalation' \ - "Validate contract does not require the task worker to process every synchronous return" - pass "Validate contract assigns the complete synchronous driver loop to the initiating task worker" -} - -test_firstmate_never_responds_for_crew_run() { - local contract - contract=$(validate_contract) - - assert_contains "$contract" "Firstmate never invokes \`no-mistakes axi respond\` for a crew-owned run." \ - "Validate contract permits Firstmate to respond directly for a crew-owned run" - pass "Validate contract forbids Firstmate from responding directly for a crew-owned run" -} - -test_worker_owns_synchronous_driver -test_firstmate_never_responds_for_crew_run diff --git a/tests/fm-pending-reply.test.sh b/tests/fm-pending-reply.test.sh index bedf731ba4f..325125eeeb2 100755 --- a/tests/fm-pending-reply.test.sh +++ b/tests/fm-pending-reply.test.sh @@ -59,9 +59,9 @@ case "${1:-}" in fi exit 0 ;; display-message) - for a in "$@"; do case "$a" in *cursor_y*) printf '0\n'; exit 0 ;; esac; done + for a in "$@"; do case "$a" in *cursor_y*) printf '1\n'; exit 0 ;; esac; done printf 'fakepane\n'; exit 0 ;; - capture-pane) printf '\xe2\x94\x82 \xe2\x94\x82\n'; exit 0 ;; + capture-pane) printf '╭────╮\n│ │\n╰────╯\n'; exit 0 ;; list-windows) exit 0 ;; esac exit 0 @@ -677,7 +677,7 @@ test_unknown_backend_state_uses_capture_fallback() { export FM_PENDING_REPLY_NOW=10000 corr=$(fm_pending_reply_create "$home" "$state" "hibit" "$backend fallback") fm_pending_reply_mark_delivered "$state" "$corr" - fm_write_secondmate_meta "$state/hibit.meta" "$sm_home" "session:fm-hibit" + fm_write_secondmate_meta "$state/hibit.meta" "$sm_home" "session:fm-hibit" alpha pi [ "$backend" = tmux ] || printf 'backend=%s\n' "$backend" >> "$state/hibit.meta" fm_backend_busy_state() { printf 'unknown'; } fm_backend_capture() { printf '%s' "$FM_PENDING_TEST_CAPTURE"; } @@ -711,6 +711,37 @@ test_unknown_backend_state_uses_capture_fallback() { pass "tmux and zellij unknown states use bounded capture fallback" } +test_kimi_capture_fallback_uses_recorded_harness() ( + local home state corr rec sm_home + home=$(setup_parent kimi-fallback) + state="$home/state" + sm_home="$home/sm" + mkdir -p "$sm_home/state" + # This fixture clock is intentionally scoped to the isolated subshell. + # shellcheck disable=SC2030,SC2031 + export FM_PENDING_REPLY_NOW=10020 + corr=$(fm_pending_reply_create "$home" "$state" hibit "kimi fallback") + fm_pending_reply_mark_delivered "$state" "$corr" + fm_write_secondmate_meta "$state/hibit.meta" "$sm_home" "session:fm-hibit" alpha kimi + fm_backend_busy_state() { printf 'unknown'; } + fm_backend_capture() { printf '%s' "$FM_PENDING_KIMI_CAPTURE"; } + export FM_PENDING_KIMI_CAPTURE=' 🌑 · Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"' + + [ "$(fm_pending_reply_backend_observation tmux session:fm-hibit fm-hibit codex)" = fallback-idle ] \ + || fail "Kimi spinner leaked into another harness" + export FM_PENDING_KIMI_CAPTURE='Ctrl+c:cancel' + [ "$(fm_pending_reply_backend_observation tmux session:fm-hibit fm-hibit kimi)" = fallback-idle ] \ + || fail "Grok's exact busy token leaked into Kimi pending-reply observation" + export FM_PENDING_KIMI_CAPTURE=' 🌑 · Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"' + fm_pending_reply_tick "$state" + rec=$(fm_pending_reply_path "$state" "$corr") + [ "$(fm_pending_reply_get "$rec" turn_seen_busy)" = 1 ] \ + || fail "recorded Kimi spinner was not observed as busy" + [ "$(phase_of "$state" "$corr")" = awaiting_report ] \ + || fail "working Kimi secondmate entered recovery" + pass "pending replies scope Kimi capture fallback by recorded harness" +) + test_tick_skips_terminal_and_reuses_target_observation() { ( local home state open1 open2 resolved escalated rec probe_log probes scan_log scans snapshot @@ -740,11 +771,15 @@ test_tick_skips_terminal_and_reuses_target_observation() { fm_write_secondmate_meta "$state/hibit.meta" "$home/hibit" "sess:fm-hibit" fm_write_secondmate_meta "$state/resolved.meta" "$home/resolved" "sess:fm-resolved" fm_write_secondmate_meta "$state/escalated.meta" "$home/escalated" "sess:fm-escalated" + # Runtime overrides called indirectly by the pending-reply tick. + # shellcheck disable=SC2329 fm_backend_busy_state() { printf '%s\t%s\n' "$1" "$2" >> "$probe_log" printf 'busy' } + # shellcheck disable=SC2329 fm_backend_capture() { fail "native busy observations should not capture"; } + # shellcheck disable=SC2329 fm_pending_reply_find_resolve_line() { local status_file=$1 corr=$2 line printf '%s\t%s\n' "$status_file" "$corr" >> "$scan_log" @@ -890,6 +925,7 @@ test_document_pointer_resolves test_helper_report_resolves test_busy_idle_observation_via_backend_abstraction test_unknown_backend_state_uses_capture_fallback +test_kimi_capture_fallback_uses_recorded_harness test_tick_skips_terminal_and_reuses_target_observation test_correlations_reuse_only_for_matching_open_task test_tick_end_to_end_missed_then_escalate diff --git a/tests/fm-pi-watch-extension.test.sh b/tests/fm-pi-watch-extension.test.sh index 7347a75ab77..f8883194898 100755 --- a/tests/fm-pi-watch-extension.test.sh +++ b/tests/fm-pi-watch-extension.test.sh @@ -59,60 +59,6 @@ export const Type = { JS } -test_tracked_extension_present_and_self_hashing() { - local text expected_config_source - expected_config_source="config_dir=\\\"\${FM_CONFIG_OVERRIDE:-\$FM_HOME/config}\\\"" - assert_present "$EXT" "tracked Pi primary watcher extension is missing" - text=$(cat "$EXT") - assert_contains "$text" "fm_watch_arm_pi" "tracked extension missing tool name" - assert_contains "$text" "fm-watch-arm-pi" "tracked extension missing command name" - assert_contains "$text" "fm-watch-arm.sh" "tracked extension missing watcher arm" - assert_contains "$text" "sendUserMessage" "tracked extension missing Pi wake API" - assert_contains "$text" 'encodeFirstmateOperationalInput' "tracked extension does not construct typed synthetic user-role wakes" - assert_contains "$text" "deliverAs: \"followUp\"" "tracked extension missing followUp delivery" - assert_contains "$text" ".pi-watch-extension-loaded" "tracked extension missing loaded marker" - assert_contains "$text" 'createHash("sha256").update(readFileSync(extensionFile)).digest("hex")' "tracked extension does not self-hash its own content for extensionVersion" - assert_contains "$text" 'fileURLToPath(import.meta.url)' "tracked extension does not self-locate via import.meta.url" - assert_contains "$text" 'type LockOwnership = "owned" | "missing" | "other"' "tracked extension does not distinguish missing lock from another owner" - assert_contains "$text" "readFileSync(\`\${state}/.lock\`" "tracked extension does not read the effective session lock" - assert_contains "$text" 'return pidAlive(lockPid) ? "other" : "missing"' "tracked extension does not allow a pre-lock load marker" - assert_contains "$text" 'if (lockOwnership() === "other") return' "tracked extension overwrites another live session marker" - assert_contains "$text" 'const ownership = lockOwnership()' "tracked extension arm does not inspect the distinct lock ownership state" - assert_contains "$text" 'if (ownership === "other") return { ok: false' "tracked extension arm does not preserve the live-other read-only refusal" - assert_contains "$text" 'if (ownership === "missing")' "tracked extension arm collapses a stale or absent lock into the live-other refusal" - assert_contains "$text" "no live session holds the lock" "tracked extension arm missing stale-lock recovery guidance" - assert_contains "$text" "run bin/fm-session-start.sh to reclaim it" "tracked extension arm does not direct stale-lock reclamation" - assert_contains "$text" "call fm_watch_arm_pi to re-arm" "tracked extension arm does not direct supervision re-arm" - assert_contains "$text" "writeFileSync(marker, \`\${extensionVersion}\\n\${process.pid}\\n\`)" "tracked extension does not write the content version and process marker" - assert_contains "$text" "const config = process.env.FM_CONFIG_OVERRIDE" "tracked extension missing effective config resolution" - assert_contains "$text" "FM_CONFIG_OVERRIDE: config" "tracked extension does not pass the effective config to the watcher arm" - assert_contains "$text" "FM_WATCH_ARM_SCRIPT: armScript" "tracked extension does not pass the effective watcher arm script" - assert_contains "$text" "$expected_config_source" "tracked extension does not source the effective x-mode config" - assert_contains "$text" "exec \\\"\$FM_WATCH_ARM_SCRIPT\\\" --restart" "tracked extension does not restart into a Pi-owned watcher child" - assert_contains "$text" 'label: "Arm firstmate watcher"' "tracked extension tool is missing its human-readable label" - assert_not_contains "$text" "Always use this tool" "tracked extension kept broad tool-selection guidance" - assert_contains "$text" "only for the first required cycle or after a notification says the cycle is missing, failed, or unhealthy" "tracked extension tool metadata is missing the Pi first-cycle or explicit-repair rule" - assert_contains "$text" "Do not call it after ordinary work, turn completion, or ordinary signal, stale, check, or heartbeat handling" "tracked extension prompt guidance does not prevent redundant ordinary-notification calls" - assert_contains "$text" 'parameters: Type.Object({})' "tracked extension tool is not using Pi's canonical TypeBox schema" - assert_contains "$text" 'content: [{ type: "text", text: result.message }]' "tracked extension tool is missing Pi text content" - assert_contains "$text" 'details: result' "tracked extension tool is missing structured result details" - assert_contains "$text" 'ctx.ui.notify' "tracked extension command does not notify through Pi's UI" - assert_contains "$text" 'process.once("exit", cleanupOnProcessExit)' "tracked extension lacks clean-process-exit cleanup" - assert_not_contains "$text" "[ -f config/x-mode.env ]" "tracked extension kept a repo-relative x-mode config path" - pass "Pi primary watcher extension is tracked, self-hashing, and self-locating" -} - -test_spawn_template_mentions_pi_watch_placeholder() { - local text - text=$(cat "$ROOT/bin/fm-spawn.sh") - assert_contains "$text" "-e __PITURNEND__ -e __PIWATCH__" "Pi secondmate launch template does not include both primary extensions" - assert_contains "$text" "\$PROJ_ABS/.pi/extensions/fm-primary-pi-watch.ts" "fm-spawn does not point the Pi secondmate watch placeholder at the tracked extension" - assert_not_contains "$text" "fm-pi-watch-extension.sh" "fm-spawn should no longer generate the Pi watch extension before launch" - assert_contains "$text" "__PITURNEND__" "fm-spawn does not replace the Pi turn-end guard extension placeholder" - assert_contains "$text" "__PIWATCH__" "fm-spawn does not replace the Pi watch extension placeholder" - pass "Pi secondmate launch wiring includes both tracked primary extensions" -} - test_pi_extension_reports_external_healthy_watcher() { local repo home plugin out status repo="$TMP_ROOT/pi-external-healthy-root" @@ -934,6 +880,187 @@ EOF pass "Pi watcher arm distinguishes all session lock ownership states" } +test_pi_session_transition_generation_owner() { + local repo home plugin child_pid_file arm_log out status + repo="$TMP_ROOT/pi-session-transition-root" + home="$TMP_ROOT/pi-session-transition-home" + child_pid_file="$TMP_ROOT/pi-session-transition-child.pid" + arm_log="$TMP_ROOT/pi-session-transition-arm.log" + mkdir -p "$repo/bin" "$home/state" "$home/config" + install_pi_watch_extension_fixture "$repo" + plugin="$repo/.pi/extensions/fm-primary-pi-watch.ts" + cat > "$repo/bin/fm-watch-arm.sh" <<'SH' +#!/usr/bin/env bash +printf 'watcher: started pid=%s\n' "$$" +printf '%s\n' "$$" > "${FM_CHILD_PID_FILE:?}" +printf 'arm pid=%s\n' "$$" >> "${FM_ARM_LOG:?}" +trap 'exit 0' TERM INT +while :; do sleep 0.2; done +SH + chmod +x "$repo/bin/fm-watch-arm.sh" + out=$(PLUGIN="$plugin" FM_HOME="$home" FM_ROOT_OVERRIDE="$repo" FM_CHILD_PID_FILE="$child_pid_file" FM_ARM_LOG="$arm_log" FM_WATCH_REARM_RETRY_BASE_MS=5 FM_WATCH_REARM_RETRY_MAX_MS=10 FM_WATCH_REARM_RETRY_LIMIT=2 node --input-type=module 2>&1 <<'EOF' +import { existsSync, readFileSync, writeFileSync } from "node:fs"; +import { pathToFileURL } from "node:url"; + +function makePi() { + const handlers = new Map(); + let tool = null; + const pi = { + on(event, handler) { + handlers.set(event, handler); + }, + registerCommand() {}, + registerTool(candidate) { + if (candidate.name === "fm_watch_arm_pi") tool = candidate; + }, + sendUserMessage: async () => {}, + events: { on() {} }, + }; + return { pi, handlers, getTool: () => tool }; +} + +function pidAlive(pid) { + try { + process.kill(Number(pid), 0); + return true; + } catch { + return false; + } +} + +async function waitFor(pred, label, attempts = 250) { + for (let i = 0; i < attempts; i += 1) { + if (pred()) return; + await new Promise((resolve) => setTimeout(resolve, 20)); + } + throw new Error(`timeout waiting for ${label}`); +} + +function liveArmPids() { + if (!existsSync(process.env.FM_ARM_LOG)) return []; + return readFileSync(process.env.FM_ARM_LOG, "utf8") + .trim() + .split(/\n/) + .filter(Boolean) + .map((line) => { + const match = /pid=(\d+)/.exec(line); + return match ? match[1] : ""; + }) + .filter(Boolean) + .filter(pidAlive); +} + +writeFileSync(`${process.env.FM_HOME}/state/.lock`, `${process.pid}\n`); +const mod = await import(pathToFileURL(process.env.PLUGIN).href); + +const startup = makePi(); +mod.default(startup.pi); +await startup.handlers.get("session_start")?.({ type: "session_start", reason: "startup" }, {}); +const first = await startup.getTool().execute("startup", {}, undefined, undefined, {}); +if (!first.details?.ok || !String(first.details.message).includes("started Pi extension arm child")) { + throw new Error(`startup arm failed: ${JSON.stringify(first.details)}`); +} +await waitFor(() => existsSync(process.env.FM_CHILD_PID_FILE), "startup child"); +const startupChild = readFileSync(process.env.FM_CHILD_PID_FILE, "utf8").trim(); +if (!pidAlive(startupChild)) throw new Error("startup child was not alive"); +const staleTool = startup.getTool(); + +async function replaceSession(previous, reason) { + const previousChild = existsSync(process.env.FM_CHILD_PID_FILE) + ? readFileSync(process.env.FM_CHILD_PID_FILE, "utf8").trim() + : ""; + await previous.handlers.get("session_shutdown")?.({ type: "session_shutdown", reason }, {}); + if (previousChild) { + await waitFor(() => !pidAlive(previousChild), `${reason} previous child exit`); + } + const next = makePi(); + mod.default(next.pi); + await next.handlers.get("session_start")?.({ + type: "session_start", + reason, + previousSessionFile: `/tmp/previous-${reason}.jsonl`, + }, {}); + const armed = await next.getTool().execute(`arm-${reason}`, {}, undefined, undefined, {}); + if (!armed.details?.ok) { + throw new Error(`${reason} replacement arm failed: ${JSON.stringify(armed.details)}`); + } + if (String(armed.details.message).includes("shutting down")) { + throw new Error(`${reason} replacement still refused with shutting-down latch`); + } + await waitFor(() => { + if (!existsSync(process.env.FM_CHILD_PID_FILE)) return false; + const child = readFileSync(process.env.FM_CHILD_PID_FILE, "utf8").trim(); + return child && child !== previousChild && pidAlive(child); + }, `${reason} replacement child`); + const live = liveArmPids(); + if (live.length !== 1) { + throw new Error(`${reason} expected exactly one live arm child, got ${live.join(",") || "(none)"}`); + } + return next; +} + +let current = await replaceSession(startup, "new"); +current = await replaceSession(current, "resume"); +current = await replaceSession(current, "fork"); + +// Same bound instance: ordinary shutdown then session_start without a fresh factory. +const sameInstanceChild = readFileSync(process.env.FM_CHILD_PID_FILE, "utf8").trim(); +await current.handlers.get("session_shutdown")?.({ type: "session_shutdown", reason: "new" }, {}); +await current.handlers.get("session_start")?.({ type: "session_start", reason: "new" }, {}); +const sameInstanceArm = await current.getTool().execute("same-instance", {}, undefined, undefined, {}); +if (!sameInstanceArm.details?.ok || String(sameInstanceArm.details.message).includes("shutting down")) { + throw new Error(`same-instance replacement arm failed: ${JSON.stringify(sameInstanceArm.details)}`); +} +await waitFor(() => { + if (!existsSync(process.env.FM_CHILD_PID_FILE)) return false; + const child = readFileSync(process.env.FM_CHILD_PID_FILE, "utf8").trim(); + return child !== sameInstanceChild && pidAlive(child); +}, "same-instance replacement child"); +await waitFor(() => !pidAlive(sameInstanceChild), "same-instance previous child exit"); +if (liveArmPids().length !== 1) { + throw new Error(`same-instance expected one live arm child, got ${liveArmPids().join(",")}`); +} + +// Stale prior-generation callback must not stop, rearm, or clear the active generation. +const activeChild = readFileSync(process.env.FM_CHILD_PID_FILE, "utf8").trim(); +const stale = await staleTool.execute("stale-prior-generation", {}, undefined, undefined, {}); +if (stale.details?.ok !== false || !String(stale.details.message).includes("shutting down")) { + throw new Error(`stale prior generation did not refuse: ${JSON.stringify(stale.details)}`); +} +if (!pidAlive(activeChild)) throw new Error("active generation child died after stale callback"); +if (pidAlive(startupChild)) throw new Error("startup generation child was resurrected"); +if (liveArmPids().length !== 1 || liveArmPids()[0] !== activeChild) { + throw new Error(`stale callback mutated live arm set: ${liveArmPids().join(",")}`); +} +const redundant = await current.getTool().execute("redundant", {}, undefined, undefined, {}); +if (!redundant.details?.ok || !String(redundant.details.message).includes("unchanged")) { + throw new Error(`active generation lost single-flight ownership: ${JSON.stringify(redundant.details)}`); +} + +// Repeated transitions keep exactly one live cycle and never revive the refusal. +for (const reason of ["resume", "fork", "new", "resume"]) { + current = await replaceSession(current, reason); +} + +// Real terminal shutdown still blocks late rearming. +const finalChild = readFileSync(process.env.FM_CHILD_PID_FILE, "utf8").trim(); +await current.handlers.get("session_shutdown")?.({ type: "session_shutdown", reason: "quit" }, {}); +await waitFor(() => !pidAlive(finalChild), "terminal shutdown child exit"); +const quitArm = await current.getTool().execute("after-quit", {}, undefined, undefined, {}); +if (quitArm.details?.ok !== false || quitArm.details.message !== "watcher: not armed - Pi session is shutting down") { + throw new Error(`terminal quit must keep the shutting-down refusal: ${JSON.stringify(quitArm.details)}`); +} +if (liveArmPids().length !== 0) { + throw new Error(`terminal quit left live arm children: ${liveArmPids().join(",")}`); +} +EOF +) + status=$? + expect_code 0 "$status" "Pi session transitions must rearm through an explicit generation owner" + [ -z "$out" ] || fail "Pi session-transition generation owner test printed output: $out" + pass "Pi session transitions use a generation owner across /new /resume /fork, stale callbacks, and quit" +} + test_pi_process_exit_cleanup_listener_lifecycle() { local repo home plugin out status repo="$TMP_ROOT/pi-exit-listener-root" @@ -962,15 +1089,19 @@ if (process.listenerCount("exit") !== before + 1) { throw new Error("Pi extension did not install exactly one process-exit fallback"); } await handlers.get("session_shutdown")?.({ type: "session_shutdown" }, {}); -if (process.listenerCount("exit") !== before) { - throw new Error("session_shutdown did not remove the process-exit fallback"); +if (process.listenerCount("exit") !== before + 1) { + throw new Error("session_shutdown removed the process-lifetime exit fallback"); +} +await handlers.get("session_start")?.({ type: "session_start" }, {}); +if (process.listenerCount("exit") !== before + 1) { + throw new Error("replacement activation duplicated the process-exit fallback"); } EOF ) status=$? - expect_code 0 "$status" "Pi cleanup fallback listener must install once and unregister on session shutdown" + expect_code 0 "$status" "Pi cleanup fallback listener must remain singular across session replacement" [ -z "$out" ] || fail "Pi listener-lifecycle test printed output: $out" - pass "Pi process-exit cleanup listener has a bounded lifecycle" + pass "Pi process-exit cleanup listener remains singular across session replacement" } test_pi_process_exit_cleanup_stops_arm_child() { @@ -984,18 +1115,21 @@ test_pi_process_exit_cleanup_stops_arm_child() { plugin="$repo/.pi/extensions/fm-primary-pi-watch.ts" cat > "$repo/bin/fm-watch-arm.sh" <<'SH' #!/usr/bin/env bash -trap 'printf "cleaned\n" > "$FM_CLEANUP_LOG"; exit 0' TERM +trap 'printf "%s\n" "$$" >> "$FM_CLEANUP_LOG"; exit 0' TERM printf '%s\n' "$$" > "$FM_CHILD_PID_FILE" while :; do sleep 1; done SH chmod +x "$repo/bin/fm-watch-arm.sh" out=$(PLUGIN="$plugin" FM_HOME="$home" FM_ROOT_OVERRIDE="$repo" FM_CLEANUP_LOG="$cleanup_log" FM_CHILD_PID_FILE="$pid_file" node --input-type=module 2>&1 <<'EOF' -import { existsSync, writeFileSync } from "node:fs"; +import { existsSync, readFileSync, writeFileSync } from "node:fs"; import { pathToFileURL } from "node:url"; let tool = null; +const handlers = new Map(); const pi = { - on() {}, + on(event, handler) { + handlers.set(event, handler); + }, registerCommand() {}, registerTool(candidate) { if (candidate.name === "fm_watch_arm_pi") tool = candidate; @@ -1010,19 +1144,31 @@ for (let i = 0; i < 250 && !existsSync(process.env.FM_CHILD_PID_FILE); i += 1) { await new Promise((resolve) => setTimeout(resolve, 20)); } if (!existsSync(process.env.FM_CHILD_PID_FILE)) throw new Error("arm child did not start"); +const firstChild = readFileSync(process.env.FM_CHILD_PID_FILE, "utf8").trim(); +await handlers.get("session_shutdown")?.({ type: "session_shutdown" }, {}); +await handlers.get("session_start")?.({ type: "session_start" }, {}); +await tool.execute("tool-call-replacement", {}, undefined, undefined, {}); +for (let i = 0; i < 250; i += 1) { + const currentChild = readFileSync(process.env.FM_CHILD_PID_FILE, "utf8").trim(); + if (currentChild !== firstChild) break; + await new Promise((resolve) => setTimeout(resolve, 20)); +} +if (readFileSync(process.env.FM_CHILD_PID_FILE, "utf8").trim() === firstChild) { + throw new Error("replacement arm child did not start"); +} process.exit(0); EOF ) status=$? expect_code 0 "$status" "Pi process exit must run the watcher cleanup fallback" [ -z "$out" ] || fail "Pi process-exit cleanup test printed output: $out" + pid=$(cat "$pid_file") i=0 - while [ "$i" -lt 250 ] && [ ! -f "$cleanup_log" ]; do + while [ "$i" -lt 250 ] && ! grep -qx "$pid" "$cleanup_log" 2>/dev/null; do sleep 0.02 i=$((i + 1)) done - [ -f "$cleanup_log" ] || fail "Pi process-exit fallback did not deliver TERM to the arm child" - pid=$(cat "$pid_file") + grep -qx "$pid" "$cleanup_log" 2>/dev/null || fail "Pi process-exit fallback did not deliver TERM to the replacement arm child" if kill -0 "$pid" 2>/dev/null; then kill -TERM "$pid" 2>/dev/null || true fail "Pi arm child $pid survived process-exit cleanup" @@ -1030,26 +1176,6 @@ EOF pass "Pi process-exit cleanup stops the attached arm child" } -test_opencode_primary_watch_plugin_static_wiring() { - local plugin module_boundary text - plugin="$ROOT/.opencode/plugins/fm-primary-watch-arm.js" - module_boundary="$ROOT/.opencode/plugins/package.json" - assert_present "$plugin" "OpenCode primary watch plugin missing" - assert_present "$module_boundary" "OpenCode plugin ESM package boundary missing" - assert_contains "$(cat "$module_boundary")" '"type": "module"' "OpenCode plugin package boundary is not explicitly ESM" - text=$(cat "$plugin") - assert_contains "$text" "session.idle" "OpenCode plugin does not listen for session.idle" - assert_contains "$text" "fm-watch-arm.sh" "OpenCode plugin does not spawn the watcher arm" - assert_contains "$text" "promptAsync" "OpenCode plugin does not wake with promptAsync" - assert_contains "$text" 'encodeFirstmateOperationalInput' "OpenCode plugin does not construct typed synthetic user-role wakes" - assert_contains "$text" ".fm-secondmate-home" "OpenCode plugin does not scope out secondmate homes" - assert_contains "$text" "rev-parse\", \"--git-dir" "OpenCode plugin does not check linked worktree scope" - assert_contains "$text" "sessionOwnsLock" "OpenCode plugin does not gate arm attempts on the session lock" - assert_contains "$text" 'fm-watch-arm.sh" --restart' "OpenCode plugin does not restart into its own watcher child" - assert_contains "$text" 'setArmStatus("external")' "OpenCode plugin still treats an external healthy watcher as armed" - pass "OpenCode primary watcher plugin has the verified TUI wake wiring" -} - test_opencode_plugin_package_boundary_is_explicit_esm() { local fixture plugin out status fixture="$TMP_ROOT/opencode-esm-boundary/.opencode" @@ -1998,8 +2124,6 @@ EOF pass "OpenCode healthy arm output does not suppress the turn-end guard" } -test_tracked_extension_present_and_self_hashing -test_spawn_template_mentions_pi_watch_placeholder test_pi_extension_reports_external_healthy_watcher test_pi_tool_returns_agent_tool_result test_pi_redundant_tool_call_is_owned_noop @@ -2012,9 +2136,9 @@ test_pi_empty_close_retries_instead_of_disappearing test_pi_established_empty_close_honors_retry_limit test_pi_actionable_close_rechecks_session_lock test_pi_arm_distinguishes_session_lock_ownership +test_pi_session_transition_generation_owner test_pi_process_exit_cleanup_listener_lifecycle test_pi_process_exit_cleanup_stops_arm_child -test_opencode_primary_watch_plugin_static_wiring test_opencode_plugin_package_boundary_is_explicit_esm test_opencode_primary_watch_plugin_uses_effective_state_home test_opencode_primary_watch_plugin_sources_effective_config diff --git a/tests/fm-pr-check-security.test.sh b/tests/fm-pr-check-security.test.sh index f30e4cf9642..1b21e5d6a8d 100755 --- a/tests/fm-pr-check-security.test.sh +++ b/tests/fm-pr-check-security.test.sh @@ -98,7 +98,8 @@ SH write_task_meta() { local dir=$1 id=${2:-task-a} fm_write_meta "$dir/home/state/$id.meta" \ - "window=fm-$id" \ + "window=firstmate:fm-$id" \ + "endpoint_task_id=$id" \ "worktree=$dir/wt" \ "project=$dir/project" \ "kind=ship" \ @@ -592,7 +593,8 @@ SH for id in _noncanonical aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa; do dir=$(make_case "legacy-teardown-${id:0:12}") fm_write_meta "$dir/home/state/$id.meta" \ - "window=fm-$id" \ + "window=firstmate:fm-$id" \ + "endpoint_task_id=$id" \ "worktree=$dir/missing-worktree" \ "project=$dir/project" \ 'kind=ship' \ @@ -1676,7 +1678,8 @@ test_complete_single_link_validation() { state="$dir/home/state" fakebin="$dir/fakebin" fm_write_meta "$state/task-a.meta" \ - 'window=fm-task-a' \ + 'window=firstmate:fm-task-a' \ + 'endpoint_task_id=task-a' \ "worktree=$dir/missing-worktree" \ "project=$dir/project" \ 'kind=ship' \ @@ -1836,7 +1839,8 @@ test_obligation_namespace_compatibility() { > "$state/.pr-check-quarantine/_noncanonical.check.abc123" chmod 0600 "$state/.pr-check-quarantine/"* fm_write_meta "$state/_noncanonical.meta" \ - 'window=fm-_noncanonical' \ + 'window=firstmate:fm-_noncanonical' \ + 'endpoint_task_id=_noncanonical' \ "worktree=$dir/missing-worktree" \ "project=$dir/project" \ 'kind=ship' \ @@ -2603,7 +2607,8 @@ test_teardown_removes_poll_artifacts() { dir=$(make_case teardown-cleanup) fakebin="$dir/fakebin" fm_write_meta "$dir/home/state/task-a.meta" \ - 'window=fm-task-a' \ + 'window=firstmate:fm-task-a' \ + 'endpoint_task_id=task-a' \ "worktree=$dir/missing-worktree" \ "project=$dir/project" \ 'kind=ship' \ @@ -2636,7 +2641,8 @@ SH dir=$(make_case teardown-retirement-receipt) fakebin="$dir/fakebin" fm_write_meta "$dir/home/state/task-a.meta" \ - 'window=fm-task-a' \ + 'window=firstmate:fm-task-a' \ + 'endpoint_task_id=task-a' \ "worktree=$dir/missing-worktree" \ "project=$dir/project" \ 'kind=ship' \ @@ -2663,7 +2669,8 @@ SH dir=$(make_case teardown-reserved-quarantine) fakebin="$dir/fakebin" fm_write_meta "$dir/home/state/invalid.meta" \ - 'window=fm-invalid' \ + 'window=firstmate:fm-invalid' \ + 'endpoint_task_id=invalid' \ "worktree=$dir/missing-worktree" \ "project=$dir/project" \ 'kind=ship' \ @@ -2693,7 +2700,8 @@ SH dir=$(make_case "teardown-final-directory-${artifact//./-}") fakebin="$dir/fakebin" fm_write_meta "$dir/home/state/task-a.meta" \ - 'window=fm-task-a' \ + 'window=firstmate:fm-task-a' \ + 'endpoint_task_id=task-a' \ "worktree=$dir/missing-worktree" \ "project=$dir/project" \ 'kind=ship' \ @@ -2733,7 +2741,8 @@ SH dir=$(make_case "teardown-quarantine-link-$kind") fakebin="$dir/fakebin" fm_write_meta "$dir/home/state/task-a.meta" \ - 'window=fm-task-a' \ + 'window=firstmate:fm-task-a' \ + 'endpoint_task_id=task-a' \ "worktree=$dir/missing-worktree" \ "project=$dir/project" \ 'kind=ship' \ @@ -2873,11 +2882,6 @@ EOF [ "$rc" -eq 2 ] || fail "merge wrapper did not refuse a GitLab merge request URL" [ ! -s "$dir/gh-axi.log" ] || fail "merge wrapper reached the GitHub CLI for a GitLab URL" - # The instance is data, never a constant, so self-hosted instances work. - ! grep -qF gitlab.com "$ROOT/bin/fm-pr-lib.sh" \ - || fail "the shared PR library hardcodes a GitLab host" - ! grep -qF gitlab.com "$ROOT/bin/fm-pr-poll.sh" \ - || fail "the static poll hardcodes a GitLab host" pass "GitLab merge requests are followed on any instance and never wake falsely" } diff --git a/tests/fm-secondmate-harness.test.sh b/tests/fm-secondmate-harness.test.sh index 87d60b0eb23..446dcd93a17 100755 --- a/tests/fm-secondmate-harness.test.sh +++ b/tests/fm-secondmate-harness.test.sh @@ -14,11 +14,13 @@ # explicit per-spawn harness arg still wins. # B) Inheritance. The primary pushes a declared, extensible set of LOCAL # (gitignored) config items - config/crew-dispatch.json, config/crew-harness, -# config/backlog-backend, and config/herdr-presentation-spaces - down into -# each secondmate home's config/, so the secondmate's OWN crewmates, -# dispatch profiles, backlog backend, and Herdr presentation opt-in inherit -# the primary's settings. It is primary-authoritative (re-pushed at -# secondmate spawn, on the bootstrap secondmate sweep, and by config push). +# config/backlog-backend, config/backend, config/herdr-presentation-spaces, and +# config/startup-memory-budget - +# down into each secondmate home's config/, so the secondmate's OWN crewmates, +# dispatch profiles, backlog backend, runtime-backend default, and Herdr +# presentation opt-in inherit the primary's settings. It is primary-authoritative +# (re-pushed at secondmate spawn, on the bootstrap secondmate sweep, and by +# config push). # config/secondmate-harness is deliberately NOT inherited (secondmates do # not spawn secondmates). After a successful push that changes allowlisted # config under an already-running home, a literal-content reread instruction @@ -76,6 +78,7 @@ both absent -> own (backward-compat)^-^-^claude^claude crew set, secondmate absent -> crew (backward-compat)^codex^-^codex^codex crew set, secondmate set -> secondmate wins, crew untouched^codex^grok^grok^codex crew absent, secondmate set -> secondmate value, crew own^-^grok^grok^claude +signed Pi wrapper remains a distinct secondmate value^codex^pi-signed^pi-signed^codex secondmate=default defers to crew^codex^default^codex^codex crew=default resolves to own, secondmate follows^default^-^claude^claude secondmate=default with crew absent -> own^-^default^claude^claude @@ -113,6 +116,7 @@ absent file -> own harness, empty model/effort^ABSENT^claude^^ bare harness only -> empty model/effort (backward-compat)^claude^claude^^ harness + model -> model only^claude opus^claude^opus^ harness + model + effort -> both^claude opus high^claude^opus^high +signed Pi wrapper + model + effort preserves every token^pi-signed openai-codex/gpt-5.6-sol max^pi-signed^openai-codex/gpt-5.6-sol^max default harness token -> falls back to crew, empty model/effort^default^claude^^ extra whitespace between tokens is tolerated^grok grok-4 xhigh^grok^grok-4^xhigh leading/trailing blank lines and a comment are skipped^# a comment\n\nclaude opus low\n^claude^opus^low @@ -120,20 +124,138 @@ ROWS pass "C1 fm-harness.sh secondmate-model/secondmate-effort resolve the optional tokens; bare harness stays empty (backward-compat)" } +# =========================================================================== +# A/C) pi-signed process identity and shared Pi marker behavior +# =========================================================================== +test_pi_signed_detection_and_session_lock_identity() { + local dir fakebin got + dir="$TMP_ROOT/pi-signed-identity" + fakebin=$(fm_fakebin "$dir") + cat > "$fakebin/ps" <<'SH' +#!/usr/bin/env bash +set -u +field= pid= +while [ "$#" -gt 0 ]; do + case "$1" in + -o) field=$2; shift 2 ;; + -p) pid=$2; shift 2 ;; + *) shift ;; + esac +done +case "$pid:$field:${FM_TEST_SIGNED_SHAPE:-exact}" in + 100:comm=:*) printf '%s\n' '/test/Pi.app/bin/pi' ;; + 100:args=:*) printf '%s\n' 'Pi' ;; + 100:ppid=:*) printf '%s\n' 200 ;; + 200:comm=:exact) printf '%s\n' '/opt/test/bin/pi-signed' ;; + 200:args=:exact) printf '%s\n' 'pi-signed --model test/model' ;; + 200:comm=:helper) printf '%s\n' '/opt/test/bin/pi-signed-helper' ;; + 200:args=:helper) printf '%s\n' 'pi-signed-helper' ;; + 200:comm=:plain) printf '%s\n' '/bin/zsh' ;; + 200:args=:plain) printf '%s\n' 'zsh' ;; + 200:ppid=:*) printf '%s\n' 1 ;; + *:comm=:*) printf '%s\n' bash ;; + *:args=:*) printf '%s\n' bash ;; + *:ppid=:*) printf '%s\n' 100 ;; +esac +SH + chmod +x "$fakebin/ps" + + got=$(PATH="$fakebin:$BASE_PATH" PI_CODING_AGENT=true "$ROOT/bin/fm-harness.sh") + [ "$got" = pi ] || fail "unmarked shared signed-wrapper ancestry resolved '$got', expected pi" + got=$(PATH="$fakebin:$BASE_PATH" PI_CODING_AGENT=true FM_PI_HARNESS=pi-signed "$ROOT/bin/fm-harness.sh") + [ "$got" = pi-signed ] || fail "selected signed wrapper resolved '$got', expected pi-signed" + got=$(PATH="$fakebin:$BASE_PATH" PI_CODING_AGENT=true FM_PI_HARNESS=pi "$ROOT/bin/fm-harness.sh") + [ "$got" = pi ] || fail "selected plain Pi resolved '$got', expected pi" + got=$(PATH="$fakebin:$BASE_PATH" PI_CODING_AGENT=true FM_PI_HARNESS=pi-signed-helper "$ROOT/bin/fm-harness.sh") + [ "$got" = pi ] || fail "inexact signed selection marker resolved '$got', expected pi" + got=$(env -u PI_CODING_AGENT PATH="$fakebin:$BASE_PATH" FM_PI_HARNESS=pi-signed "$ROOT/bin/fm-harness.sh") + [ "$got" = pi ] || fail "signed selection marker without Pi's family marker resolved '$got', expected pi" + got=$(PATH="$fakebin:$BASE_PATH" PI_CODING_AGENT=true FM_TEST_SIGNED_SHAPE=plain "$ROOT/bin/fm-harness.sh") + [ "$got" = pi ] || fail "plain Pi marker resolved '$got', expected pi" + got=$(PATH="$fakebin:$BASE_PATH" PI_CODING_AGENT=true FM_TEST_SIGNED_SHAPE=helper "$ROOT/bin/fm-harness.sh") + [ "$got" = pi ] || fail "unrelated pi-signed-helper ancestry resolved '$got', expected pi" + + got=$(PATH="$fakebin:$BASE_PATH" bash -c \ + '. "$0/bin/fm-session-lock-lib.sh"; fm_harness_ancestry_pid' "$ROOT") + [ "$got" = 100 ] || fail "session-lock ancestry selected '$got', expected the inner Pi engine pid 100" + PATH="$fakebin:$BASE_PATH" bash -c \ + '. "$0/bin/fm-session-lock-lib.sh"; kill() { return 0; }; fm_harness_pid_alive 200' "$ROOT" \ + || fail "session-lock liveness rejected exact pi-signed holder" + if PATH="$fakebin:$BASE_PATH" FM_TEST_SIGNED_SHAPE=helper bash -c \ + '. "$0/bin/fm-session-lock-lib.sh"; kill() { return 0; }; fm_harness_pid_alive 200' "$ROOT"; then + fail "session-lock liveness accepted unrelated pi-signed-helper" + fi + + pass "pi-signed identity: authoritative launch selection distinguishes shared wrapper ancestry" +} + +test_dash_leading_process_names_are_basename_operands() { + local dir fakebin got err status + dir="$TMP_ROOT/dash-leading-process-names" + fakebin=$(fm_fakebin "$dir") + cat > "$fakebin/ps" <<'SH' +#!/usr/bin/env bash +set -u +field= pid= +while [ "$#" -gt 0 ]; do + case "$1" in + -o) field=$2; shift 2 ;; + -p) pid=$2; shift 2 ;; + *) shift ;; + esac +done +case "$pid:$field" in + 4242:comm=) printf '%s\n' '/opt/test/bin/codex' ;; + 4242:args=) printf '%s\n' 'codex' ;; + 4242:ppid=) printf '%s\n' 1 ;; + 5252:comm=) printf '%s\n' '-codex' ;; + 5252:args=) printf '%s\n' '-codex' ;; + 5252:ppid=) printf '%s\n' 1 ;; + *:comm=) printf '%s\n' '-zsh' ;; + *:args=) printf '%s\n' '-zsh' ;; + *:ppid=) printf '%s\n' 4242 ;; +esac +SH + chmod +x "$fakebin/ps" + + err="$dir/fm-harness.err" + got=$(env -u CLAUDECODE -u PI_CODING_AGENT -u GROK_AGENT \ + PATH="$fakebin:$BASE_PATH" "$ROOT/bin/fm-harness.sh" 2>"$err") + [ "$got" = codex ] || fail "dash-leading shell ancestry resolved '$got', expected codex" + [ ! -s "$err" ] || fail "fm-harness wrote basename option noise for literal -zsh: $(cat "$err")" + + err="$dir/fm-session-lock-ancestry.err" + got=$(PATH="$fakebin:$BASE_PATH" bash -c \ + '. "$0/bin/fm-session-lock-lib.sh"; fm_harness_ancestry_pid' "$ROOT" 2>"$err") + [ "$got" = 4242 ] || fail "session-lock dash-leading ancestry selected '$got', expected pid 4242" + [ ! -s "$err" ] || fail "session-lock ancestry wrote basename option noise for literal -zsh: $(cat "$err")" + + err="$dir/fm-session-lock-alive.err" + PATH="$fakebin:$BASE_PATH" bash -c \ + '. "$0/bin/fm-session-lock-lib.sh"; kill() { return 0; }; fm_harness_pid_alive 5252' \ + "$ROOT" 2>"$err"; status=$? + expect_code 0 "$status" "session-lock liveness should accept literal -codex as a harness process name" + [ ! -s "$err" ] || fail "session-lock liveness wrote basename option noise for literal -codex: $(cat "$err")" + + pass "harness identity: dash-leading ps command names are basename operands, not options" +} + # =========================================================================== # B) propagate_inheritable_config unit behavior # =========================================================================== test_propagate_lib() { - local d src dest m1 m2 outside stdout stderr guard_repo err_text + local d src dest home m1 m2 outside stdout stderr guard_repo err_text d="$TMP_ROOT/prop-lib" src="$d/src" - dest="$d/dest" - mkdir -p "$src" "$dest" + home="$d/home1" + dest="$home/config" + mkdir -p "$src" "$dest" "$home/state" # 1. present source is copied printf '{"default":{"harness":"codex"}}\n' > "$src/crew-dispatch.json" printf 'codex\n' > "$src/crew-harness" printf 'manual\n' > "$src/backlog-backend" + printf 'tmux\n' > "$src/backend" : > "$src/herdr-presentation-spaces" stdout="$d/clean-copy.out" stderr="$d/clean-copy.err" @@ -143,7 +265,11 @@ test_propagate_lib() { [ "$(cat "$dest/crew-dispatch.json")" = '{"default":{"harness":"codex"}}' ] || fail "crew-dispatch.json not propagated" [ "$(cat "$dest/crew-harness")" = codex ] || fail "crew-harness not propagated" [ "$(cat "$dest/backlog-backend")" = manual ] || fail "backlog-backend not propagated" + [ "$(cat "$dest/backend")" = tmux ] || fail "backend not propagated" [ -f "$dest/herdr-presentation-spaces" ] || fail "herdr-presentation-spaces not propagated" + printf 'herdr\n' > "$dest/backend" + propagate_inheritable_config "$src" "$dest" + [ "$(cat "$dest/backend")" = tmux ] || fail "primary backend did not overwrite a divergent destination" # 2. idempotent: an unchanged re-run does not churn the mtime m1=$(date -r "$dest/crew-harness" +%s 2>/dev/null || stat -c %Y "$dest/crew-harness") @@ -160,10 +286,12 @@ test_propagate_lib() { printf '{"default":{"harness":"claude"}}\n' > "$src/crew-dispatch.json" printf 'claude\n' > "$src/crew-harness" printf 'tasks-axi\n' > "$src/backlog-backend" + printf 'zellij\n' > "$src/backend" propagate_inheritable_config "$src" "$dest" [ "$(cat "$dest/crew-dispatch.json")" = '{"default":{"harness":"claude"}}' ] || fail "changed dispatch profile did not converge" [ "$(cat "$dest/crew-harness")" = claude ] || fail "changed value did not converge" [ "$(cat "$dest/backlog-backend")" = tasks-axi ] || fail "changed backlog backend did not converge" + [ "$(cat "$dest/backend")" = zellij ] || fail "changed backend did not converge" outside="$d/outside-target" rm -f "$dest/crew-harness" "$outside" @@ -176,11 +304,14 @@ test_propagate_lib() { [ "$(cat "$outside")" = outside ] || fail "destination symlink target was overwritten" # 4. removing the source mirrors absence downstream (primary-authoritative) - rm -f "$src/crew-dispatch.json" "$src/crew-harness" "$src/backlog-backend" "$src/herdr-presentation-spaces" + printf 'herdr\n' > "$dest/backend" + rm -f "$src/crew-dispatch.json" "$src/crew-harness" "$src/backlog-backend" \ + "$src/backend" "$src/herdr-presentation-spaces" propagate_inheritable_config "$src" "$dest" [ -e "$dest/crew-dispatch.json" ] && fail "dispatch profile absence not mirrored downstream" [ -e "$dest/crew-harness" ] && fail "absence not mirrored downstream" [ -e "$dest/backlog-backend" ] && fail "backlog-backend absence not mirrored downstream" + [ -e "$dest/backend" ] && fail "backend absence not mirrored downstream" [ -e "$dest/herdr-presentation-spaces" ] && fail "herdr-presentation-spaces absence not mirrored downstream" rm -f "$dest/crew-harness" @@ -198,22 +329,25 @@ test_propagate_lib() { [ -d "$dest/crew-harness" ] || fail "failed absence mirror removed the wrong path" rm -rf "$dest/crew-harness" - # 5. secondmate-harness is never inherited + # 5. secondmate-harness is never inherited; backend still is printf 'grok\n' > "$src/secondmate-harness" printf '{"default":{"harness":"codex"}}\n' > "$src/crew-dispatch.json" printf 'codex\n' > "$src/crew-harness" printf 'manual\n' > "$src/backlog-backend" - rm -rf "$d/dest2" - mkdir -p "$d/dest2" - propagate_inheritable_config "$src" "$d/dest2" - [ -e "$d/dest2/secondmate-harness" ] && fail "secondmate-harness was inherited (must not be)" - [ "$(cat "$d/dest2/crew-dispatch.json")" = '{"default":{"harness":"codex"}}' ] || fail "crew-dispatch.json not propagated alongside" - [ "$(cat "$d/dest2/crew-harness")" = codex ] || fail "crew-harness not propagated alongside" - [ "$(cat "$d/dest2/backlog-backend")" = manual ] || fail "backlog-backend not propagated alongside" + printf 'herdr\n' > "$src/backend" + rm -rf "$d/home2" + mkdir -p "$d/home2/config" "$d/home2/state" + propagate_inheritable_config "$src" "$d/home2/config" + [ -e "$d/home2/config/secondmate-harness" ] && fail "secondmate-harness was inherited (must not be)" + [ "$(cat "$d/home2/config/crew-dispatch.json")" = '{"default":{"harness":"codex"}}' ] || fail "crew-dispatch.json not propagated alongside" + [ "$(cat "$d/home2/config/crew-harness")" = codex ] || fail "crew-harness not propagated alongside" + [ "$(cat "$d/home2/config/backlog-backend")" = manual ] || fail "backlog-backend not propagated alongside" + [ "$(cat "$d/home2/config/backend")" = herdr ] || fail "backend not propagated alongside" # 6. nothing to propagate -> destination dir is never created (a true no-op) rm -rf "$d/src3" "$d/dest3" mkdir -p "$d/src3" + # Keep backend out of the empty-source case by clearing it from src3 only. propagate_inheritable_config "$d/src3" "$d/dest3/config" [ -e "$d/dest3/config" ] && fail "empty-source propagation created a destination dir" @@ -304,6 +438,7 @@ test_spawn_split_and_inherit() { printf 'claude\n' > "$w/home/config/crew-harness" printf 'codex\n' > "$w/home/config/secondmate-harness" printf 'manual\n' > "$w/home/config/backlog-backend" + printf 'zellij\n' > "$w/home/config/backend" make_seeded_home "$sm" sm spawn_secondmate "$w" sm "$sm" @@ -318,6 +453,8 @@ test_spawn_split_and_inherit() { || fail "split: home crew-dispatch.json not inherited" [ "$(cat "$sm/config/backlog-backend" 2>/dev/null)" = manual ] \ || fail "split: home backlog-backend not inherited as manual" + [ "$(cat "$sm/config/backend" 2>/dev/null)" = zellij ] \ + || fail "split: home backend not inherited as zellij" [ -e "$sm/config/secondmate-harness" ] \ && fail "split: secondmate-harness leaked into the secondmate home" pass "B2 spawn: secondmate runs the secondmate harness; its home inherits declared config" @@ -470,6 +607,50 @@ spawn_secondmate_capture() { "$ROOT/bin/fm-spawn.sh" "$id" "$home" "$@" --secondmate } +test_spawn_backend_precedence_over_inherited_config() { + local w sm meta launchlog out status + w="$TMP_ROOT/spawn-backend-env-precedence" + sm="$w/sm" + launchlog="$w/launch.log" + mkdir -p "$w/home/config" + printf 'herdr\n' > "$w/home/config/backend" + make_seeded_home "$sm" sm + + out=$(FM_BACKEND=tmux spawn_secondmate_capture \ + "$w" sm "$sm" "$launchlog" 2>&1); status=$? + expect_code 0 "$status" \ + "FM_BACKEND=tmux should beat inherited config/backend=herdr"$'\n'"$out" + + meta="$w/home/state/sm.meta" + [ "$(cat "$sm/config/backend")" = herdr ] \ + || fail "backend precedence fixture did not inherit config/backend=herdr" + assert_no_grep '^backend=' "$meta" \ + "FM_BACKEND=tmux did not beat inherited config/backend=herdr" + pass "B5b spawn: FM_BACKEND wins over inherited config/backend" +} + +test_spawn_explicit_backend_precedence_over_env_and_inherited_config() { + local w sm meta launchlog out status + w="$TMP_ROOT/spawn-backend-flag-precedence" + sm="$w/sm" + launchlog="$w/launch.log" + mkdir -p "$w/home/config" + printf 'herdr\n' > "$w/home/config/backend" + make_seeded_home "$sm" sm + + out=$(FM_BACKEND=zellij spawn_secondmate_capture \ + "$w" sm "$sm" "$launchlog" --backend tmux 2>&1); status=$? + expect_code 0 "$status" \ + "explicit --backend tmux should beat FM_BACKEND=zellij and inherited config/backend=herdr"$'\n'"$out" + + meta="$w/home/state/sm.meta" + [ "$(cat "$sm/config/backend")" = herdr ] \ + || fail "explicit backend precedence fixture did not inherit config/backend=herdr" + assert_no_grep '^backend=' "$meta" \ + "explicit --backend tmux did not beat FM_BACKEND=zellij and inherited config/backend=herdr" + pass "B5c spawn: explicit --backend wins over FM_BACKEND and inherited config/backend" +} + # A bare "<harness>" secondmate-harness file (today's format) must launch with # NO --model/--effort flag at all, and meta must keep recording model=default, # effort=default - the core backward-compat requirement of the new format. @@ -703,6 +884,7 @@ new_world() { printf 'projects/\nstate/\ndata/\n.no-mistakes/\n' [ "$dispatch_ignore" = no ] || printf 'config/crew-dispatch.json\n' printf 'config/crew-harness\nconfig/secondmate-harness\nconfig/backlog-backend\n' + printf 'config/backend\nconfig/herdr-presentation-spaces\nconfig/startup-memory-budget\n' } > "$w/main/.gitignore" printf 'v1\n' > "$w/main/AGENTS.md" printf 'r1\n' > "$w/main/README.md" @@ -898,6 +1080,7 @@ test_bootstrap_sweep_propagates_and_reconverges() { printf '{"default":{"harness":"codex"}}\n' > "$w/home/config/crew-dispatch.json" printf 'codex\n' > "$w/home/config/crew-harness" printf 'manual\n' > "$w/home/config/backlog-backend" + printf 'tmux\n' > "$w/home/config/backend" printf 'grok\n' > "$w/home/config/secondmate-harness" run_bootstrap "$w" >/dev/null [ "$(cat "$w/sm/config/crew-harness" 2>/dev/null)" = codex ] \ @@ -906,6 +1089,8 @@ test_bootstrap_sweep_propagates_and_reconverges() { || fail "sweep: crew-dispatch.json not pushed into the live home" [ "$(cat "$w/sm/config/backlog-backend" 2>/dev/null)" = manual ] \ || fail "sweep: backlog-backend not pushed into the live home" + [ "$(cat "$w/sm/config/backend" 2>/dev/null)" = tmux ] \ + || fail "sweep: backend not pushed into the live home" [ -e "$w/sm/config/secondmate-harness" ] \ && fail "sweep: secondmate-harness was inherited (must not be)" @@ -913,6 +1098,7 @@ test_bootstrap_sweep_propagates_and_reconverges() { printf '{"default":{"harness":"claude"}}\n' > "$w/home/config/crew-dispatch.json" printf 'claude\n' > "$w/home/config/crew-harness" printf 'tasks-axi\n' > "$w/home/config/backlog-backend" + printf 'zellij\n' > "$w/home/config/backend" run_bootstrap "$w" >/dev/null [ "$(cat "$w/sm/config/crew-harness" 2>/dev/null)" = claude ] \ || fail "sweep: home did not re-converge to the primary's new crew-harness" @@ -920,9 +1106,12 @@ test_bootstrap_sweep_propagates_and_reconverges() { || fail "sweep: home did not re-converge to the primary's new crew-dispatch.json" [ "$(cat "$w/sm/config/backlog-backend" 2>/dev/null)" = tasks-axi ] \ || fail "sweep: home did not re-converge to the primary's new backlog-backend" + [ "$(cat "$w/sm/config/backend" 2>/dev/null)" = zellij ] \ + || fail "sweep: home did not re-converge to the primary's new backend" # Mirror absence: primary clears inherited config; the home's copies are removed. - rm -f "$w/home/config/crew-dispatch.json" "$w/home/config/crew-harness" "$w/home/config/backlog-backend" + rm -f "$w/home/config/crew-dispatch.json" "$w/home/config/crew-harness" \ + "$w/home/config/backlog-backend" "$w/home/config/backend" run_bootstrap "$w" >/dev/null [ -e "$w/sm/config/crew-dispatch.json" ] \ && fail "sweep: home crew-dispatch.json not removed after the primary cleared it" @@ -930,6 +1119,8 @@ test_bootstrap_sweep_propagates_and_reconverges() { && fail "sweep: home crew-harness not removed after the primary cleared it" [ -e "$w/sm/config/backlog-backend" ] \ && fail "sweep: home backlog-backend not removed after the primary cleared it" + [ -e "$w/sm/config/backend" ] \ + && fail "sweep: home backend not removed after the primary cleared it" pass "B7 bootstrap sweep pushes, re-converges, and mirrors absence; never inherits secondmate-harness" } @@ -944,6 +1135,7 @@ test_bootstrap_sweep_propagates_when_tracked_current() { printf '{"default":{"harness":"codex"}}\n' > "$w/home/config/crew-dispatch.json" printf 'codex\n' > "$w/home/config/crew-harness" printf 'manual\n' > "$w/home/config/backlog-backend" + printf 'tmux\n' > "$w/home/config/backend" run_bootstrap "$w" >/dev/null [ "$(cat "$w/sm/config/crew-dispatch.json" 2>/dev/null)" = '{"default":{"harness":"codex"}}' ] \ || fail "crew-dispatch.json did not propagate to a tracked-current home" @@ -951,6 +1143,8 @@ test_bootstrap_sweep_propagates_when_tracked_current() { || fail "config did not propagate to a tracked-current home" [ "$(cat "$w/sm/config/backlog-backend" 2>/dev/null)" = manual ] \ || fail "backlog-backend did not propagate to a tracked-current home" + [ "$(cat "$w/sm/config/backend" 2>/dev/null)" = tmux ] \ + || fail "backend did not propagate to a tracked-current home" pass "B8 bootstrap sweep propagates config even when the home's tracked files are already current" } @@ -983,10 +1177,10 @@ test_bootstrap_sweep_defers_dispatch_on_stale_unignored_home() { pass "B9 bootstrap sweep defers new inherited config until the home ignores it" } -# Backward-compat: with no inherited config set, the sweep is a no-op for the -# home's config/ - exactly as before this feature - and ordinary sweep behavior -# (fast-forward) is unaffected. -test_bootstrap_sweep_no_inheritance_is_noop() { +# The primary bootstrap always materializes the startup-memory default, so an +# otherwise empty inherited surface converges that one visible value while +# ordinary tracked-file fast-forward behavior remains unchanged. +test_bootstrap_sweep_materializes_and_inherits_memory_default() { local w c1 w=$(new_world boot-noop) c1=$(git -C "$w/main" rev-parse HEAD) @@ -1000,12 +1194,52 @@ test_bootstrap_sweep_no_inheritance_is_noop() { run_bootstrap "$w" >/dev/null - [ -e "$w/sm/config/crew-dispatch.json" ] && fail "no-inheritance sweep created a home crew-dispatch.json" - [ -e "$w/sm/config/crew-harness" ] && fail "no-inheritance sweep created a home crew-harness" - [ -e "$w/sm/config" ] && fail "no-inheritance sweep created a home config/ dir" + [ -e "$w/sm/config/crew-dispatch.json" ] && fail "default-only sweep created a home crew-dispatch.json" + [ -e "$w/sm/config/crew-harness" ] && fail "default-only sweep created a home crew-harness" + [ -e "$w/sm/config/backend" ] && fail "default-only sweep created a home backend" + [ "$(cat "$w/home/config/startup-memory-budget")" = 7500 ] \ + || fail "primary bootstrap did not materialize the startup-memory default" + [ "$(cat "$w/sm/config/startup-memory-budget")" = 7500 ] \ + || fail "default-only sweep did not converge startup-memory-budget" [ "$(git -C "$w/sm" rev-parse HEAD)" = "$head" ] \ - || fail "no-inheritance sweep did not still fast-forward the tracked files" - pass "B10 bootstrap sweep with no inherited config is a config no-op and still fast-forwards" + || fail "default-only sweep did not still fast-forward the tracked files" + pass "B10 bootstrap sweep materializes and inherits the startup-memory default while fast-forwarding" +} + +# config/backend: present and absent primary state converges exactly. +test_backend_inheritance_present_and_absent() { + local w head out err status instruction + w=$(new_world backend-inherit) + head=$(git -C "$w/main" rev-parse HEAD) + add_sm_worktree "$w" sm "$head" + + printf 'tmux\n' > "$w/home/config/backend" + err="$w/backend-inherit.err" + out=$(run_config_push "$w" 2>"$err"); status=$? + expect_code 0 "$status" "backend present push should succeed" + assert_contains "$out" "backend: pushed" "backend present value should report pushed" + [ "$(cat "$w/sm/config/backend")" = tmux ] || fail "backend present value not pushed" + instruction=$(reread_instruction_path "$w/sm") || fail "backend present reread instruction missing" + assert_contains "$(cat "$instruction")" $'-----BEGIN config/backend-----\ntmux\n-----END config/backend-----' \ + "backend present reread must include exact bytes" + + printf 'herdr\n' > "$w/sm/config/backend" + printf 'zellij\n' > "$w/home/config/backend" + out=$(run_config_push "$w" 2>"$err"); status=$? + expect_code 0 "$status" "backend changed push should succeed" + assert_contains "$out" "backend: pushed" "backend changed value should report pushed" + [ "$(cat "$w/sm/config/backend")" = zellij ] \ + || fail "primary backend did not overwrite the divergent destination" + + rm -f "$w/home/config/backend" + out=$(run_config_push "$w" 2>"$err"); status=$? + expect_code 0 "$status" "backend absence push should succeed" + assert_contains "$out" "backend: pushed - mirrored primary absence" "backend should mirror primary absence" + [ -e "$w/sm/config/backend" ] && fail "backend not removed on primary absence" + instruction=$(reread_instruction_path "$w/sm") || fail "backend absence reread instruction missing" + assert_contains "$(cat "$instruction")" $'-----BEGIN config/backend-----\nABSENT\n-----END config/backend-----' \ + "backend absence reread must use ABSENT token" + pass "B12b backend inheritance: present values and primary absence converge exactly" } test_bootstrap_sweep_surfaces_config_propagation_failure() { @@ -1046,7 +1280,7 @@ test_bootstrap_rereads_after_partial_propagation() { } test_config_push_propagates_reports_without_ff_or_nudge() { - local w c1 sm_real old_head out err status out2 tmp log + local w c1 sm_real old_head out err status out2 tmp log instruction w=$(new_world config-push-basic) c1=$(git -C "$w/main" rev-parse HEAD) add_sm_worktree "$w" sm "$c1" @@ -1064,6 +1298,7 @@ test_config_push_propagates_reports_without_ff_or_nudge() { printf '{"default":{"harness":"codex"}}\n' > "$w/home/config/crew-dispatch.json" printf 'codex\n' > "$w/home/config/crew-harness" printf 'manual\n' > "$w/home/config/backlog-backend" + printf 'tmux\n' > "$w/home/config/backend" err="$w/config-push-basic.err" log="$w/config-push-basic.tmux.log" out=$(run_config_push "$w" "$log" 2>"$err"); status=$? @@ -1079,12 +1314,18 @@ test_config_push_propagates_reports_without_ff_or_nudge() { "config push did not report crew-harness as pushed" assert_contains "$out" "backlog-backend: pushed" \ "config push did not report backlog-backend as pushed" + assert_contains "$out" "backend: pushed" \ + "config push did not report backend as pushed" assert_contains "$out" "config-reread: sent" \ "config push with changed config must send a literal reread instruction" assert_not_contains "$out" "NUDGE_SECONDMATES" \ "config push must not use the AGENTS.md instruction-surface nudge channel" [ "$(git -C "$w/sm" rev-parse HEAD)" = "$old_head" ] \ || fail "config push fast-forwarded tracked files" + [ "$(cat "$w/sm/config/backend")" = tmux ] || fail "config push did not write backend" + instruction=$(reread_instruction_path "$w/sm") || fail "config-push reread instruction missing" + assert_contains "$(cat "$instruction")" $'-----BEGIN config/backend-----\ntmux\n-----END config/backend-----' \ + "config-push reread must include exact backend bytes" [ ! -s "$err" ] || fail "clean config push wrote unexpected stderr: $(cat "$err")" assert_contains "$(cat "$log")" "[fm-from-firstmate]" \ "config reread must use the marked routed secondmate path" @@ -1098,6 +1339,8 @@ test_config_push_propagates_reports_without_ff_or_nudge() { "idempotent config push did not report crew-harness as unchanged" assert_contains "$out2" "backlog-backend: unchanged" \ "idempotent config push did not report backlog-backend as unchanged" + assert_contains "$out2" "backend: unchanged" \ + "idempotent config push did not report backend as unchanged" assert_not_contains "$out2" "config-reread: sent" \ "unchanged config must not send a reread message" [ ! -s "$log" ] || fail "unchanged config push still invoked tmux send: $(cat "$log")" @@ -1234,6 +1477,7 @@ test_config_reread_per_home_changed_sets_and_exact_bytes() { printf '%s' "$multiline_json" > "$w/home/config/crew-dispatch.json" printf 'codex\n' > "$w/home/config/crew-harness" printf 'manual\n' > "$w/home/config/backlog-backend" + printf 'tmux\n' > "$w/home/config/backend" { shared_captain_header_for_tests printf '%s\n' "shared secret preference body that must never appear in a config reread" @@ -1252,6 +1496,7 @@ test_config_reread_per_home_changed_sets_and_exact_bytes() { || fail "beta did not receive multiline dispatch" [ "$(cat "$w/alpha/config/crew-harness")" = codex ] || fail "alpha harness not updated" [ "$(cat "$w/alpha/config/backlog-backend")" = manual ] || fail "alpha backlog-backend not updated" + [ "$(cat "$w/alpha/config/backend")" = tmux ] || fail "alpha backend not updated" instr_a=$(reread_instruction_path "$w/alpha") || fail "alpha instruction missing after config push" instr_b=$(reread_instruction_path "$w/beta") || fail "beta instruction missing after config push" @@ -1261,19 +1506,21 @@ test_config_reread_per_home_changed_sets_and_exact_bytes() { [ "$(reread_mode "$instr_b")" = 600 ] || fail "beta instruction is not private" # Deterministic allowlist path order and exact destination bytes for alpha - # (all three config items were missing/stale and therefore pushed). + # (allowlisted config items were missing/stale and therefore pushed). assert_grep "These inherited config files changed" "$instr_a" "alpha framing missing" assert_grep "defaults/rules" "$instr_a" "alpha must preserve agent judgment framing" assert_contains "$(cat "$instr_a")" "config/crew-dispatch.json" "alpha missing dispatch path" assert_contains "$(cat "$instr_a")" "config/crew-harness" "alpha missing harness path" assert_contains "$(cat "$instr_a")" "config/backlog-backend" "alpha missing backlog path" + assert_contains "$(cat "$instr_a")" "config/backend" "alpha missing backend path" # Path order follows FM_INHERITABLE_CONFIG. awk ' /config\/crew-dispatch\.json/ { d=NR } /config\/crew-harness/ { h=NR } /config\/backlog-backend/ { b=NR } + /config\/backend/ && !/backlog-backend/ { k=NR } END { - if (!(d && h && b && d < h && h < b)) exit 1 + if (!(d && h && b && k && d < h && h < b && b < k)) exit 1 } ' "$instr_a" || fail "alpha instruction path order is not deterministic allowlist order" @@ -1284,6 +1531,8 @@ test_config_reread_per_home_changed_sets_and_exact_bytes() { "alpha instruction must include exact harness scalar bytes" assert_contains "$(cat "$instr_a")" $'-----BEGIN config/backlog-backend-----\nmanual\n-----END config/backlog-backend-----' \ "alpha instruction must include exact backlog-backend scalar bytes" + assert_contains "$(cat "$instr_a")" $'-----BEGIN config/backend-----\ntmux\n-----END config/backend-----' \ + "alpha instruction must include exact backend scalar bytes" # No parsed/effective summary, no SHA, no captain-shared dump. assert_not_contains "$(cat "$instr_a")" "Default worker" "must not emit parsed worker summary" @@ -1367,6 +1616,7 @@ test_config_reread_isolation_and_absent_and_send_failure() { printf '%s\n' $'crew-dispatch.json\tpushed\tmirrored primary absence' printf '%s\n' $'crew-harness\tunchanged\t' printf '%s\n' $'backlog-backend\tunchanged\t' + printf '%s\n' $'backend\tunchanged\t' printf '%s\n' $'data/captain-shared.md\tpushed\t' } > "$report" rm -f "$w/beta/config/crew-dispatch.json" @@ -1947,6 +2197,7 @@ cat > "$w/main/bin/fm-spawn.sh" <<SH . '$w/main/bin/fm-config-inherit-lib.sh' printf '%s' spawn >> '$log' printf '%s' codex > '$w/sm/config/crew-harness' +printf '%s\n' 7500 > '$w/sm/config/startup-memory-budget' SH chmod +x "$w/main/bin/fm-spawn.sh" fakebin=$(make_fake_toolchain "$w") @@ -2050,12 +2301,16 @@ SH test_harness_resolution test_secondmate_model_effort_tokens +test_pi_signed_detection_and_session_lock_identity +test_dash_leading_process_names_are_basename_operands test_propagate_lib test_spawn_split_and_inherit test_spawn_backward_compat_crew_fallback test_spawn_bare_backward_compat test_spawn_explicit_harness_wins test_spawn_unverified_secondmate_harness_refused +test_spawn_backend_precedence_over_inherited_config +test_spawn_explicit_backend_precedence_over_env_and_inherited_config test_spawn_bare_harness_no_model_effort_flag test_spawn_secondmate_harness_model_token test_spawn_secondmate_harness_model_and_effort_tokens @@ -2067,7 +2322,8 @@ test_spawn_fallback_chain_and_crew_scout_unaffected test_bootstrap_sweep_propagates_and_reconverges test_bootstrap_sweep_propagates_when_tracked_current test_bootstrap_sweep_defers_dispatch_on_stale_unignored_home -test_bootstrap_sweep_no_inheritance_is_noop +test_bootstrap_sweep_materializes_and_inherits_memory_default +test_backend_inheritance_present_and_absent test_bootstrap_sweep_surfaces_config_propagation_failure test_bootstrap_rereads_after_partial_propagation test_config_push_propagates_reports_without_ff_or_nudge diff --git a/tests/fm-secondmate-lifecycle-e2e.test.sh b/tests/fm-secondmate-lifecycle-e2e.test.sh index fd71fdb359a..31af58c1276 100755 --- a/tests/fm-secondmate-lifecycle-e2e.test.sh +++ b/tests/fm-secondmate-lifecycle-e2e.test.sh @@ -135,6 +135,7 @@ phase_spawn() { phase_send() { : > "$LOG" + : > "$PANE" # The meta window (firstmate:fm-design) must win over a foreign same-named # window returned by list-windows. PATH="$FAKEBIN:$PATH" FM_HOME="$HOME_DIR" FM_FAKE_TMUX_WINDOW="other-session:fm-design" \ diff --git a/tests/fm-secondmate-liveness.test.sh b/tests/fm-secondmate-liveness.test.sh index 2f42a3a6a88..ed356638962 100755 --- a/tests/fm-secondmate-liveness.test.sh +++ b/tests/fm-secondmate-liveness.test.sh @@ -97,7 +97,7 @@ SH test_tmux_agent_state_classifies() { local fb out - for harness in claude codex opencode grok; do + for harness in claude codex opencode grok kimi pi pi-signed pi-launcher Pi; do fb=$(make_probe_tmux "$TMP_ROOT/tmux-$harness" "$harness") out=$(PATH="$fb:$BASE_PATH" bash -c '. "$0/bin/fm-backend.sh"; fm_backend_agent_state tmux sess:win' "$ROOT") [ "$out" = alive ] || fail "a live $harness foreground process should classify as alive, got '$out'" @@ -206,7 +206,7 @@ test_agent_state_dispatcher_and_compatibility() { make_toolchain() { local dir=$1 fakebin fakebin=$(fm_fakebin "$dir") - fm_fake_exit0 "$fakebin" node gh-axi chrome-devtools-axi lavish-axi + fm_fake_exit0 "$fakebin" node gh-axi chrome-devtools-axi lavish-axi pi-signed cat > "$fakebin/gh" <<'SH' #!/usr/bin/env bash exit 0 @@ -349,7 +349,7 @@ test_sweep_respawns_confirmed_dead_secondmate() { assert_not_contains "$out" "SECONDMATE_LIVENESS: secondmate sm1: respawned" \ "a successfully respawned secondmate should be handled silently" - assert_contains "$(cat "$log")" "kill-window -t firstmate:fm-sm1" \ + assert_contains "$(cat "$log")" "kill-window -t =firstmate:=fm-sm1" \ "the stale endpoint must be killed before respawn (tmux refuses a same-named window over a live one)" assert_contains "$(cat "$log")" "new-window" \ "a confirmed-dead secondmate should actually be relaunched" @@ -391,6 +391,25 @@ test_sweep_respawns_authoritatively_missing_pi_secondmate() { pass "sweep: an authoritatively missing Pi secondmate window is relaunched" } +test_sweep_respawns_authoritatively_missing_pi_signed_secondmate() { + local w fb tmuxfb log out + w=$(new_world sweep-missing-pi-signed) + printf '%s\n' pi-signed > "$w/home/config/secondmate-harness" + add_sm_home "$w" sm1 firstmate:fm-sm1 pi-signed + fb=$(make_toolchain "$w"); tmuxfb=$(make_liveness_tmux "$w") + log="$w/calls.log"; : > "$log" + + out=$(run_bootstrap "$tmuxfb:$fb" "$w/home" missing "$log") + + assert_not_contains "$out" "unverified for recovery" \ + "a recorded pi-signed secondmate should be verified for recovery" + assert_contains "$(cat "$log")" "new-window" \ + "an authoritatively missing pi-signed secondmate should be relaunched" + assert_not_contains "$(cat "$log")" "kill-window" \ + "an absent pi-signed window should not need a destructive pre-kill" + pass "sweep: an authoritatively missing pi-signed secondmate window is relaunched" +} + test_sweep_never_acts_on_ambiguous_existing_process() { local w fb tmuxfb log out w=$(new_world sweep-ambiguous) @@ -514,6 +533,7 @@ test_agent_state_dispatcher_and_compatibility test_sweep_respawns_confirmed_dead_secondmate test_sweep_leaves_alive_secondmate_untouched test_sweep_respawns_authoritatively_missing_pi_secondmate +test_sweep_respawns_authoritatively_missing_pi_signed_secondmate test_sweep_never_acts_on_ambiguous_existing_process test_sweep_never_acts_on_transient_unreadability test_sweep_reports_missing_endpoint_relaunch_failure diff --git a/tests/fm-secondmate-safety.test.sh b/tests/fm-secondmate-safety.test.sh index bf65516053b..31331621c82 100755 --- a/tests/fm-secondmate-safety.test.sh +++ b/tests/fm-secondmate-safety.test.sh @@ -1435,8 +1435,8 @@ EOF [ ! -d "$childwt" ] || fail "force teardown did not remove child worktree" [ ! -e "$home/state/domain.meta" ] || fail "teardown did not clear parent meta" grep -F -- '- domain ' "$home/data/secondmates.md" >/dev/null && fail "force teardown did not remove secondmate registry route" - grep -F 'kill-window -t firstmate:fm-child' "$log" >/dev/null || fail "force teardown did not kill child window" - grep -F 'kill-window -t firstmate:fm-domain' "$log" >/dev/null || fail "force teardown did not kill parent window" + grep -F 'kill-window -t =firstmate:=fm-child' "$log" >/dev/null || fail "force teardown did not kill child window" + grep -F 'kill-window -t =firstmate:=fm-domain' "$log" >/dev/null || fail "force teardown did not kill parent window" pass "secondmate force teardown discards child work" } @@ -1614,7 +1614,7 @@ EOF || fail "force teardown refused $opdir symlinked inside the secondmate home" [ ! -e "$subhome" ] || fail "force teardown did not remove subhome with inside $opdir symlink" [ ! -e "$home/state/domain.meta" ] || fail "force teardown did not clear parent meta for inside $opdir symlink" - grep -F 'kill-window -t firstmate:fm-domain' "$log" >/dev/null || fail "force teardown did not kill parent window for inside $opdir symlink" + grep -F 'kill-window -t =firstmate:=fm-domain' "$log" >/dev/null || fail "force teardown did not kill parent window for inside $opdir symlink" done pass "force teardown allows operational directory symlinks inside the subhome" } diff --git a/tests/fm-secondmate-sync.test.sh b/tests/fm-secondmate-sync.test.sh index 79781e28add..999aebc2df3 100755 --- a/tests/fm-secondmate-sync.test.sh +++ b/tests/fm-secondmate-sync.test.sh @@ -837,15 +837,6 @@ test_seed_marker_does_not_mask_real_dirt() { pass "T14 marker tolerance does not mask a genuinely dirty home" } -# --- T15: the shipped firstmate repo gitignores the seed marker ----------------- -# Pins the actual fix so it cannot silently regress: without this .gitignore entry -# every seeded home would read dirty again the moment it lands on this repo's HEAD. -test_repo_gitignores_seed_marker() { - grep -qxF '.fm-secondmate-home' "$ROOT/.gitignore" \ - || fail "the firstmate repo .gitignore must ignore the seed marker (.fm-secondmate-home)" - pass "T15 the firstmate repo gitignores the secondmate seed marker" -} - test_ff_updated test_ff_current test_ff_dirty @@ -866,6 +857,5 @@ test_spawn_warns_when_sync_skipped_before_launch test_seed_marker_clean_when_gitignored test_seed_marker_converges_existing_home test_seed_marker_does_not_mask_real_dirt -test_repo_gitignores_seed_marker echo "# all fm-secondmate-sync tests passed" diff --git a/tests/fm-secrets-check.test.sh b/tests/fm-secrets-check.test.sh index 0b132e6ab29..508f4003181 100755 --- a/tests/fm-secrets-check.test.sh +++ b/tests/fm-secrets-check.test.sh @@ -365,15 +365,26 @@ PY } test_inventory_rejects_unsafe_doppler_workflow_targets() { - local case_root manifest workflow out rc - for case_root in owned hosted indeterminate quoted fork unknown-trigger override; do + local case_root manifest workflow out rc command + for case_root in owned hosted indeterminate quoted fork unknown-trigger override cli-owned cli-hosted; do case_root="$TMP_ROOT/workflow-$case_root" mkdir -p "$case_root/docs" "$case_root/.github/workflows" cp "$EXAMPLE" "$case_root/docs/secrets-policy.json" manifest="$case_root/docs/secrets-policy.json" workflow="$case_root/.github/workflows/deploy.yml" workflow_trigger=$'on:\n push:\n workflow_dispatch:' + command='' case "$case_root" in + *cli-owned) + runner='[self-hosted, Linux, X64, fleet-ci]' + job_key='deploy' + command='doppler run -- ./scripts/deploy.sh' + ;; + *cli-hosted) + runner='ubuntu-latest' + job_key='deploy' + command='doppler run -- ./scripts/deploy.sh' + ;; *owned) runner='[self-hosted, Linux, X64, fleet-ci]' job_key='deploy' @@ -439,7 +450,17 @@ with open(sys.argv[1], "w", encoding="utf-8") as fh: PY ;; esac - cat >"$workflow" <<EOF + if [ -n "$command" ]; then + cat >"$workflow" <<EOF +${workflow_trigger:-} +jobs: + $job_key: + runs-on: $runner + steps: + - run: $command +EOF + else + cat >"$workflow" <<EOF ${workflow_trigger:-} jobs: $job_key: @@ -450,12 +471,13 @@ jobs: doppler-token: \${{ secrets.DOPPLER_TOKEN }} inject-env-vars: true EOF + fi fm_git_init_commit "$case_root" git -C "$case_root" add docs/secrets-policy.json .github/workflows/deploy.yml git -C "$case_root" -c user.name='Firstmate Tests' -c user.email='tests@example.invalid' commit -qm workflow rc=0 out=$("$CHECK" inventory "$case_root" 2>&1) || rc=$? - if [ "$case_root" = "$TMP_ROOT/workflow-owned" ]; then + if [ "$case_root" = "$TMP_ROOT/workflow-owned" ] || [ "$case_root" = "$TMP_ROOT/workflow-cli-owned" ]; then [ "$rc" -eq 0 ] || fail "inventory rejected the positively proven owned workflow"$'\n'"$out" assert_contains "$out" "doppler_refs=true" "inventory did not inspect the owned workflow" else diff --git a/tests/fm-send-popup-settle.test.sh b/tests/fm-send-popup-settle.test.sh index 685ce627abb..f9d55528070 100755 --- a/tests/fm-send-popup-settle.test.sh +++ b/tests/fm-send-popup-settle.test.sh @@ -50,9 +50,9 @@ set -u case "${1:-}" in send-keys) exit 0 ;; display-message) - for a in "$@"; do case "$a" in *cursor_y*) printf '0\n'; exit 0 ;; esac; done + for a in "$@"; do case "$a" in *cursor_y*) printf '1\n'; exit 0 ;; esac; done printf 'fakepane\n'; exit 0 ;; - capture-pane) printf '\xe2\x94\x82 \xe2\x94\x82\n'; exit 0 ;; + capture-pane) printf '╭────╮\n│ │\n╰────╯\n'; exit 0 ;; list-windows) exit 0 ;; esac exit 0 diff --git a/tests/fm-send-secondmate-marker-herdr-e2e.test.sh b/tests/fm-send-secondmate-marker-herdr-e2e.test.sh index 528ea49f223..5e28b9aaf8d 100755 --- a/tests/fm-send-secondmate-marker-herdr-e2e.test.sh +++ b/tests/fm-send-secondmate-marker-herdr-e2e.test.sh @@ -27,7 +27,7 @@ if [ "${FM_SEND_MARKER_HERDR_E2E:-0}" != 1 ]; then exit 0 fi -for tool in git herdr jq pi python3; do +for tool in git herdr jq pi; do command -v "$tool" >/dev/null 2>&1 || { echo "skip: $tool not found"; exit 0; } done @@ -39,6 +39,7 @@ SECOND_HOME="$TMP_ROOT/secondmate-home" CAPTURE="$TMP_ROOT/pi-before-agent.jsonl" FAKEBIN="$TMP_ROOT/fakebin" ORIGINAL_PATH=$PATH +REAL_PI=$(command -v pi) ID='marker-pi-sm' REQUEST='FM_MARKER_HERDR_E2E exact-id request' DIRECT='FM_MARKER_HERDR_DIRECT captain input' @@ -93,38 +94,25 @@ You are a task-local secondmate used only for the marker transport regression. Stay idle and do not initiate work. EOF -# The extension is already an explicit Pi -e resource in the real secondmate -# launch template, so its project_trust hook can grant session-only trust before -# project resources load. before_agent_start records the exact prompt bytes and -# aborts before any provider request, keeping this transport regression local. +# A separate explicit Pi extension grants session-only project trust, records +# before_agent_start prompt bytes, and aborts before any provider request. +# The PATH wrapper adds only that test resource while preserving the production +# secondmate launch and its own extension arguments unchanged. CAPTURE_JSON=$(printf '%s' "$CAPTURE" | jq -Rs .) -python3 - "$SECOND_HOME/.pi/extensions/fm-primary-turnend-guard.ts" "$CAPTURE_JSON" <<'PY' -from pathlib import Path -import sys - -path = Path(sys.argv[1]) -capture_json = sys.argv[2] -source = path.read_text() -import_anchor = 'import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";\n' -source = source.replace( - import_anchor, - import_anchor - + 'import { appendFileSync as fmAppendFileSync } from "node:fs";\n' - + f'const fmCapturePath = {capture_json};\n', - 1, -) -factory_anchor = 'export default function (pi: ExtensionAPI) {\n' -replacement = '''export default function (pi: ExtensionAPI) { +CAPTURE_EXTENSION="$TMP_ROOT/fm-send-marker-capture.ts" +cat > "$CAPTURE_EXTENSION" <<EOF +import { appendFileSync } from "node:fs"; +const capturePath = $CAPTURE_JSON; +export default function (pi: any) { pi.on("project_trust", () => ({ trusted: "yes", remember: false })); pi.on("before_agent_start", (event, ctx) => { - fmAppendFileSync(fmCapturePath, `${JSON.stringify({ prompt: event.prompt, hex: Buffer.from(event.prompt, "utf8").toString("hex") })}\\n`); + appendFileSync(capturePath, \`\${JSON.stringify({ prompt: event.prompt, hex: Buffer.from(event.prompt, "utf8").toString("hex") })}\\n\`); ctx.abort(); }); -''' -if import_anchor not in source or factory_anchor not in source: - raise SystemExit("Pi extension insertion point missing") -path.write_text(source.replace(factory_anchor, replacement, 1)) -PY +} +EOF +printf '#!/usr/bin/env bash\nexec %q -e %q "$@"\n' "$REAL_PI" "$CAPTURE_EXTENSION" > "$FAKEBIN/pi" +chmod +x "$FAKEBIN/pi" "$LAB_HELPER" provision "$SESSION" PATH="$FAKEBIN:$ORIGINAL_PATH" FM_GATE_REFUSE_BYPASS=1 FM_HOME="$SENDER_HOME" HERDR_SESSION="$SESSION" \ diff --git a/tests/fm-send-secondmate-marker.test.sh b/tests/fm-send-secondmate-marker.test.sh index 3790dfc8509..d50c5802055 100755 --- a/tests/fm-send-secondmate-marker.test.sh +++ b/tests/fm-send-secondmate-marker.test.sh @@ -53,9 +53,9 @@ case "${1:-}" in fi exit 0 ;; display-message) - for a in "$@"; do case "$a" in *cursor_y*) printf '0\n'; exit 0 ;; esac; done + for a in "$@"; do case "$a" in *cursor_y*) printf '1\n'; exit 0 ;; esac; done printf 'fakepane\n'; exit 0 ;; - capture-pane) printf '\xe2\x94\x82 \xe2\x94\x82\n'; exit 0 ;; + capture-pane) printf '╭────╮\n│ │\n╰────╯\n'; exit 0 ;; list-windows) exit 0 ;; esac exit 0 diff --git a/tests/fm-send-settle.test.sh b/tests/fm-send-settle.test.sh index 93a9c351b24..01d2d427e7b 100755 --- a/tests/fm-send-settle.test.sh +++ b/tests/fm-send-settle.test.sh @@ -36,9 +36,9 @@ set -u case "${1:-}" in send-keys) exit 0 ;; display-message) - for a in "$@"; do case "$a" in *cursor_y*) printf '0\n'; exit 0 ;; esac; done + for a in "$@"; do case "$a" in *cursor_y*) printf '1\n'; exit 0 ;; esac; done printf 'fakepane\n'; exit 0 ;; - capture-pane) printf '\xe2\x94\x82 \xe2\x94\x82\n'; exit 0 ;; + capture-pane) printf '╭────╮\n│ │\n╰────╯\n'; exit 0 ;; list-windows) exit 0 ;; esac exit 0 diff --git a/tests/fm-send-strict.test.sh b/tests/fm-send-strict.test.sh index 21305d4a233..1faf98a0ce2 100755 --- a/tests/fm-send-strict.test.sh +++ b/tests/fm-send-strict.test.sh @@ -35,19 +35,22 @@ case "${1:-}" in exit 0 ;; display-message) target= + cursor=0 while [ $# -gt 0 ]; do case "$1" in -t) target=$2; shift 2 ;; + *cursor_y*) cursor=1; shift ;; *) shift ;; esac done if [ -n "${FM_FAKE_TMUX_DEAD_TARGET:-}" ] && [ "$target" = "$FM_FAKE_TMUX_DEAD_TARGET" ]; then exit 1 fi + [ "$cursor" = 1 ] && { printf '1\n'; exit 0; } printf '%%1\n' exit 0 ;; capture-pane) - printf '\xe2\x94\x82 \xe2\x94\x82\n' + printf '╭────╮\n│ │\n╰────╯\n' exit 0 ;; list-windows) printf 'foreign:%s\n' "${FM_FAKE_TMUX_WINDOW:-fm-lost}" diff --git a/tests/fm-session-start.test.sh b/tests/fm-session-start.test.sh index d8bf405b041..9bcf04eddba 100755 --- a/tests/fm-session-start.test.sh +++ b/tests/fm-session-start.test.sh @@ -405,15 +405,22 @@ SH # run_session_start <home> <root> <path> # Drop every harness env marker from bin/fm-harness.sh detect_own so the # surrounding interactive shell cannot leak past the suite's fake ps harness. -# Markers today: CLAUDECODE (claude), PI_CODING_AGENT (pi), GROK_AGENT (grok). +# Markers today: CLAUDECODE (claude), PI_CODING_AGENT plus FM_PI_HARNESS +# (Pi family), GROK_AGENT (grok). # codex and opencode have no env markers (ancestry only). Without this, a local # claude/pi/grok session fails cases that pin a different fake harness while CI # (no ambient markers) still passes. run_session_start() { - local home=$1 root=$2 path=$3 - env -u CLAUDECODE -u PI_CODING_AGENT -u GROK_AGENT \ - FM_HOME="$home" FM_ROOT_OVERRIDE="$root" PATH="$path" \ - "$SESSION_START" + local home=$1 root=$2 path=$3 pi_harness=${4:-} + if [ -n "$pi_harness" ]; then + env -u CLAUDECODE -u GROK_AGENT PI_CODING_AGENT=true FM_PI_HARNESS="$pi_harness" \ + FM_HOME="$home" FM_ROOT_OVERRIDE="$root" PATH="$path" \ + "$SESSION_START" + else + env -u CLAUDECODE -u PI_CODING_AGENT -u FM_PI_HARNESS -u GROK_AGENT \ + FM_HOME="$home" FM_ROOT_OVERRIDE="$root" PATH="$path" \ + "$SESSION_START" + fi } # prepare_session_start_secondmate <name>: a throwaway main home and Pi @@ -452,7 +459,7 @@ EOF run_session_start_secondmate() { local root=$1 home=$2 fakebin=$3 mate=$4 log=$5 spawned=$6 mode=$7 - FM_BACKEND=tmux FM_FAKE_TMUX_MODE="$mode" FM_FAKE_TMUX_LOG="$log" \ + TMUX='' FM_BACKEND=tmux FM_FAKE_TMUX_MODE="$mode" FM_FAKE_TMUX_LOG="$log" \ FM_FAKE_TMUX_SPAWNED="$spawned" FM_FAKE_SECOND_MATE_HOME="$mate" \ FM_FAKE_SECOND_MATE_ID="$SESSION_START_SECOND_MATE_ID" \ run_session_start "$home" "$root" "$fakebin:$BASE_PATH" @@ -618,7 +625,7 @@ EOF assert_contains "$out" "Skipping every mutating step" "read-only banner did not explain what was skipped" assert_contains "$out" "skipped (read-only session)" "wake-queue section did not report itself skipped" assert_contains "$out" "WATCHER DOWN - SUPERVISION IS OFF" "read-only guard did not surface watcher-liveness alarm" - assert_contains "$out" "queued wakes pending - left untouched for the session holding the fleet lock" "read-only guard did not leave queued wakes to the lock holder" + assert_contains "$out" "queued wakes pending - left untouched because this session lacks verified fleet-lock ownership" "read-only guard did not leave queued wakes untouched without verified lock ownership" assert_contains "$out" "TANGLE: primary checkout on feature branch 'fm/read-only-tangle'" "read-only bootstrap did not surface the tangle diagnostic" assert_contains "$out" "read-only session must leave restore work" "read-only tangle diagnostic did not explain restore ownership" assert_contains "$out" "Stay read-only: do not arm" "read-only next step did not block direct watcher repair" @@ -645,6 +652,105 @@ EOF pass "a lock refusal prints a loud read-only banner, skips every mutating step, and still completes the digest" } +test_lock_write_failure_read_only_path() { + local rec root home fakebin out status + rec=$(new_world lock-write-failure) + IFS='|' read -r root home fakebin <<EOF +$rec +EOF + make_fake_toolchain "$fakebin" + make_fake_ps_claude "$fakebin" + append_wake "$home/state" signal task-a "done: must remain queued" || fail "seed wake failed" + chmod 0500 "$home/state" + + status=0 + out=$(run_session_start "$home" "$root" "$fakebin:$BASE_PATH") || status=$? + chmod 0700 "$home/state" + + expect_code 0 "$status" "fm-session-start.sh must exit 0 when lock publication fails" + assert_contains "$out" "cannot write session lock" "lock publication failure was not surfaced" + assert_contains "$out" "READ-ONLY SESSION" "lock publication failure did not force a read-only session" + assert_contains "$out" "FLEET LOCK OWNERSHIP WAS NOT VERIFIED" "lock publication failure was misreported as a live holder" + assert_contains "$out" "lacks verified fleet-lock ownership" "lock publication failure did not explain why queued wakes remain untouched" + assert_not_contains "$out" "ANOTHER LIVE FIRSTMATE SESSION HOLDS THE FLEET LOCK" "lock publication failure falsely claimed a live lock holder" + [ -s "$home/state/.wake-queue" ] || fail "lock publication failure allowed the wake queue to mutate" + + pass "session start stays read-only when lock ownership cannot be published" +} + +test_session_lock_concurrent_single_winner() { + local rec root home fakebin ready completed winners pids i pid count + rec=$(new_world lock-concurrency) + IFS='|' read -r root home fakebin <<EOF +$rec +EOF + ready="$home/ready" + completed="$home/done" + winners="$home/winners" + mkdir -p "$ready" "$completed" + : > "$winners" + cat > "$fakebin/ps" <<'SH' +#!/usr/bin/env bash +set -u +pid= +previous= +for argument in "$@"; do + [ "$previous" = -p ] && pid=$argument + previous=$argument +done +case "$*" in + *"comm="*) + if [ -f "$FM_FAKE_LOCK_STATE/harness-$pid" ]; then + printf '%s\n' /usr/local/bin/claude + else + printf '%s\n' /bin/bash + fi + ;; + *"args="*) + if [ -f "$FM_FAKE_LOCK_STATE/harness-$pid" ]; then + printf '%s\n' claude + else + printf '%s\n' bash + fi + ;; + *"ppid="*) printf '%s\n' "$FM_FAKE_HARNESS_PID" ;; + *) exit 1 ;; +esac +SH + chmod +x "$fakebin/ps" + + pids= + i=1 + while [ "$i" -le 40 ]; do + ( + harness_pid=$BASHPID + : > "$home/state/harness-$harness_pid" + : > "$ready/$i" + while [ "$(find "$ready" -type f | wc -l | tr -d ' ')" -lt 40 ]; do + sleep 0.01 + done + if FM_HOME="$home" FM_FAKE_LOCK_STATE="$home/state" \ + FM_FAKE_HARNESS_PID="$harness_pid" PATH="$fakebin:$BASE_PATH" \ + "$ROOT/bin/fm-lock.sh" >/dev/null 2>&1; then + printf '%s\n' "$harness_pid" >> "$winners" + fi + : > "$completed/$i" + while [ "$(find "$completed" -type f | wc -l | tr -d ' ')" -lt 40 ]; do + sleep 0.01 + done + ) & + pids="$pids $!" + i=$((i + 1)) + done + for pid in $pids; do + wait "$pid" 2>/dev/null || true + done + count=$(awk 'NF { count++ } END { print count + 0 }' "$winners") + [ "$count" -eq 1 ] || fail "concurrent session-lock acquisition produced $count winners" + + pass "concurrent session-lock acquisition admits exactly one live harness" +} + # --- output ordering ---------------------------------------------------------- test_output_ordering_diagnostics_lead() { @@ -859,7 +965,7 @@ EOF out=$(run_session_start_secondmate "$root" "$home" "$fakebin" "$mate" "$log" "$spawned" shell) assert_not_contains "$out" "SECONDMATE_LIVENESS:" "successful bare-shell recovery should stay non-actionable" - assert_contains "$(cat "$log")" "kill-window -t firstmate:fm-$SESSION_START_SECOND_MATE_ID" \ + assert_contains "$(cat "$log")" "kill-window -t =firstmate:=fm-$SESSION_START_SECOND_MATE_ID" \ "the proven bare-shell path did not remove its existing dead endpoint" assert_contains "$(cat "$log")" "new-window" "the proven bare-shell path did not relaunch" assert_contains "$out" "endpoint: alive (backend=tmux window=firstmate:fm-$SESSION_START_SECOND_MATE_ID)" \ @@ -1152,6 +1258,29 @@ EOF pass "session start emits exactly one detected harness block and reports Pi extension load state" } +test_pi_signed_primary_uses_pi_extensions_without_identity_normalization() { + local rec root home fakebin out + rec=$(new_world pi-signed-supervision-block) + IFS='|' read -r root home fakebin <<EOF +$rec +EOF + make_fake_toolchain "$fakebin" + make_fake_ps_harness "$fakebin" pi-signed + + out=$(FM_FAKE_HARNESS=pi-signed run_session_start "$home" "$root" "$fakebin:$BASE_PATH" pi-signed) + + assert_contains "$out" "SUPERVISION OPERATING INSTRUCTIONS - primary harness: pi-signed" \ + "session start normalized a pi-signed primary to pi" + assert_contains "$out" "Mode: Pi extension background wake." \ + "pi-signed primary did not reuse Pi's supervision protocol" + assert_contains "$out" "PI_WATCH_EXTENSION: not loaded" \ + "pi-signed primary skipped Pi extension validation" + assert_contains "$out" "restart pi-signed so $root/.pi/extensions/fm-primary-turnend-guard.ts and $root/.pi/extensions/fm-primary-pi-watch.ts auto-load" \ + "pi-signed extension diagnostic did not preserve the executable identity" + + pass "session start preserves pi-signed primary identity while applying Pi extension guarantees" +} + test_pi_diagnostic_rejects_stale_loaded_marker() { local rec root home fakebin out marker holder_pid rec=$(new_world pi-stale-loaded-marker) @@ -1258,6 +1387,8 @@ EOF test_context_digest_absent_empty_present test_lock_refusal_read_only_path +test_lock_write_failure_read_only_path +test_session_lock_concurrent_single_winner test_output_ordering_diagnostics_lead test_herdr_backend_diagnostics_follow_real_session_start test_session_start_relaunches_missing_pi_secondmate @@ -1277,6 +1408,7 @@ test_fleet_digest_empty_fleet test_next_step_sources_x_mode_cadence test_next_step_afk_delegates_to_daemon test_supervision_block_exactly_one_and_pi_diagnostic +test_pi_signed_primary_uses_pi_extensions_without_identity_normalization test_pi_diagnostic_rejects_stale_loaded_marker test_pi_diagnostic_accepts_prelock_loaded_marker test_pi_diagnostic_rejects_missing_turnend_guard_marker diff --git a/tests/fm-sessionstart-nudge.test.sh b/tests/fm-sessionstart-nudge.test.sh index 28bb3d18b55..878295cba5a 100755 --- a/tests/fm-sessionstart-nudge.test.sh +++ b/tests/fm-sessionstart-nudge.test.sh @@ -148,44 +148,6 @@ EOF pass "OpenCode session.created delivers the exact wrapper nudge once per session" } -test_tracked_harness_registration() { - local command pi_plugin opencode_plugin - jq -e '.hooks.SessionStart | length == 1' "$ROOT/.claude/settings.json" >/dev/null \ - || fail "Claude SessionStart hook is not registered exactly once" - jq -e '.hooks.SessionStart[0].matcher == "startup|resume|clear"' "$ROOT/.claude/settings.json" >/dev/null \ - || fail "Claude SessionStart matcher must include startup/resume/clear and exclude compact" - jq -e 'any(.hooks.SessionStart[]?.hooks[]?.command?; contains("fm-sessionstart-nudge.sh"))' \ - "$ROOT/.claude/settings.json" >/dev/null || fail "Claude SessionStart hook does not invoke the wrapper" - - command=$(jq -r '.hooks.SessionStart[0].hooks[0].command' "$ROOT/.codex/hooks.json") - # shellcheck disable=SC2016 - assert_contains "$command" 'payload=$(cat' "Codex SessionStart hook does not read its payload" - # shellcheck disable=SC2016 - assert_contains "$command" 'root=$(pwd -P)' "Codex SessionStart hook is not pwd-anchored" - assert_contains "$command" 'fm-sessionstart-nudge.sh' "Codex SessionStart hook does not invoke the wrapper" - - command=$(jq -r '.hooks.SessionStart[0].hooks[0].command' "$ROOT/.grok/hooks/fm-primary-sessionstart-nudge.json") - # shellcheck disable=SC2016 - assert_contains "$command" '${GROK_WORKSPACE_ROOT:-}' "Grok SessionStart hook lacks an inline-default workspace root" - # shellcheck disable=SC2016 - assert_not_contains "$command" '${GROK_WORKSPACE_ROOT}' "Grok SessionStart hook contains a bare variable expansion" - assert_contains "$command" 'fm-sessionstart-nudge.sh' "Grok SessionStart hook does not invoke the wrapper" - - pi_plugin=$(cat "$ROOT/.pi/extensions/fm-primary-turnend-guard.ts") - assert_contains "$pi_plugin" '["startup", "new", "resume"]' "Pi SessionStart handler has the wrong reason allowlist" - assert_contains "$pi_plugin" 'fm-sessionstart-nudge.sh' "Pi SessionStart handler does not invoke the wrapper" - assert_contains "$pi_plugin" 'firstmate-sessionstart-nudge' "Pi SessionStart handler does not inject a custom context message" - assert_contains "$pi_plugin" 'details: { kind: "session-start" }' "Pi SessionStart context does not retain its exact structured kind" - assert_contains "$pi_plugin" 'pi.sendMessage' "Pi SessionStart handler does not use the context-safe message API" - - opencode_plugin=$(cat "$ROOT/.opencode/plugins/fm-primary-sessionstart-nudge.js") - assert_contains "$opencode_plugin" 'session.created' "OpenCode plugin does not listen for session.created" - assert_contains "$opencode_plugin" 'fm-sessionstart-nudge.sh' "OpenCode plugin does not invoke the wrapper" - assert_contains "$opencode_plugin" 'promptAsync' "OpenCode plugin does not prompt the nudge turn" - - pass "all five verified harnesses register the shared session-start nudge" -} - test_genuine_primary_nudges test_gate_env_is_silent test_gate_common_dir_is_silent @@ -194,4 +156,3 @@ test_linked_secondmate_primary_nudges test_missing_state_is_silent test_owned_lock_is_silent test_opencode_plugin_delivers_exact_nudge_once -test_tracked_harness_registration diff --git a/tests/fm-spawn-dispatch-profile.test.sh b/tests/fm-spawn-dispatch-profile.test.sh index db8da9520e6..e5f017608dc 100755 --- a/tests/fm-spawn-dispatch-profile.test.sh +++ b/tests/fm-spawn-dispatch-profile.test.sh @@ -42,7 +42,7 @@ esac exit 0 SH chmod +x "$fakebin/tmux" - fm_fake_exit0 "$fakebin" treehouse + fm_fake_exit0 "$fakebin" treehouse pi-signed printf '%s\n' "$fakebin" } @@ -84,10 +84,15 @@ run_spawn() { local home=$1 wt=$2 fakebin=$3 launchlog=$4 shift 4 : > "$launchlog" + # CLAUDE_CONFIG_DIR is forwarded onto claude launches by fm-spawn, so pin it + # explicitly (empty by default) instead of leaking the invoking shell's value, + # which would make launch assertions depend on the developer's environment. + # A test opts in to the set case via FM_TEST_CLAUDE_CONFIG_DIR. FM_ROOT_OVERRIDE='' FM_HOME="$home" \ FM_STATE_OVERRIDE="$home/state" FM_DATA_OVERRIDE="$home/data" \ FM_PROJECTS_OVERRIDE="$home/projects" FM_CONFIG_OVERRIDE="$home/config" \ FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$wt" TMUX="fake,1,0" \ + CLAUDE_CONFIG_DIR="${FM_TEST_CLAUDE_CONFIG_DIR:-}" \ FM_FAKE_LAUNCH_LOG="$launchlog" GROK_HOME="$home/grok-home" PATH="$fakebin:$PATH" \ "$SPAWN" "$@" 2>&1 } @@ -123,6 +128,153 @@ test_no_profile_keeps_claude_profile_defaults() { pass "no --model/--effort records defaults and types the claude launch instructions" } +test_relative_home_overrides_launch_with_absolute_cross_process_paths() { + local rec id out status launch home_real + id=profile-relative-paths-z1b + rec=$(make_spawn_case profile-relative-paths pi "$id") + read_case_record "$rec" + home_real=$(cd "$HOME_DIR" && pwd -P) + mkdir -p "$CASE_DIR/cdpath/home/state" "$CASE_DIR/cdpath/home/data" + : > "$LAUNCH_LOG" + + out=$( + cd "$CASE_DIR" || exit 1 + CDPATH="$CASE_DIR/cdpath" FM_ROOT_OVERRIDE='' FM_HOME=home \ + FM_STATE_OVERRIDE=home/state FM_DATA_OVERRIDE=home/data \ + FM_PROJECTS_OVERRIDE=home/projects FM_CONFIG_OVERRIDE=home/config \ + FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$WT_DIR" TMUX="fake,1,0" \ + CLAUDE_CONFIG_DIR='' FM_FAKE_LAUNCH_LOG="$LAUNCH_LOG" \ + GROK_HOME=home/grok-home PATH="$FAKEBIN_DIR:$PATH" \ + "$SPAWN" "$id" "$PROJ_DIR" 2>&1 + ) + status=$? + expect_code 0 "$status" "spawn with relative home overrides should succeed" + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "-e '$home_real/state/$id.pi-ext.ts'" \ + "relative FM_STATE_OVERRIDE leaked into Pi's cross-process extension path" + assert_contains "$launch" "< '$home_real/data/$id/brief.md'" \ + "relative FM_DATA_OVERRIDE leaked into the cross-process brief path" + pass "relative home overrides ignore CDPATH and become absolute before spawn launch construction" +} + +test_home_defaults_preserve_absolute_or_resolve_relative_paths() { + local rec relative_id absolute_id out status launch home_real linked_home + relative_id=profile-relative-home-defaults-z1c + absolute_id=profile-absolute-home-defaults-z1d + rec=$(make_spawn_case profile-home-defaults pi "$relative_id" "$absolute_id") + read_case_record "$rec" + home_real=$(cd "$HOME_DIR" && pwd -P) + + : > "$LAUNCH_LOG" + out=$( + cd "$CASE_DIR" || exit 1 + FM_ROOT_OVERRIDE='' FM_HOME=home \ + FM_STATE_OVERRIDE='' FM_DATA_OVERRIDE='' \ + FM_PROJECTS_OVERRIDE=home/projects FM_CONFIG_OVERRIDE=home/config \ + FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$WT_DIR" TMUX="fake,1,0" \ + CLAUDE_CONFIG_DIR='' FM_FAKE_LAUNCH_LOG="$LAUNCH_LOG" \ + GROK_HOME=home/grok-home PATH="$FAKEBIN_DIR:$PATH" \ + "$SPAWN" "$relative_id" "$PROJ_DIR" 2>&1 + ) + status=$? + expect_code 0 "$status" "spawn with relative FM_HOME defaults should succeed" + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "-e '$home_real/state/$relative_id.pi-ext.ts'" \ + "relative FM_HOME leaked into Pi's default cross-process extension path" + assert_contains "$launch" "< '$home_real/data/$relative_id/brief.md'" \ + "relative FM_HOME leaked into the default cross-process brief path" + + linked_home="$CASE_DIR/home-link" + ln -s "$HOME_DIR" "$linked_home" + : > "$LAUNCH_LOG" + out=$( + FM_ROOT_OVERRIDE='' FM_HOME="$linked_home" \ + FM_STATE_OVERRIDE='' FM_DATA_OVERRIDE='' \ + FM_PROJECTS_OVERRIDE="$linked_home/projects" FM_CONFIG_OVERRIDE="$linked_home/config" \ + FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$WT_DIR" TMUX="fake,1,0" \ + CLAUDE_CONFIG_DIR='' FM_FAKE_LAUNCH_LOG="$LAUNCH_LOG" \ + GROK_HOME="$linked_home/grok-home" PATH="$FAKEBIN_DIR:$PATH" \ + "$SPAWN" "$absolute_id" "$PROJ_DIR" 2>&1 + ) + status=$? + expect_code 0 "$status" "spawn with absolute symlink-spelled FM_HOME defaults should succeed" + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "-e '$linked_home/state/$absolute_id.pi-ext.ts'" \ + "absolute FM_HOME spelling changed in Pi's default cross-process extension path" + assert_contains "$launch" "< '$linked_home/data/$absolute_id/brief.md'" \ + "absolute FM_HOME spelling changed in the default cross-process brief path" + pass "FM_HOME defaults resolve relative paths and preserve absolute spellings" +} + +test_absolute_override_spelling_is_preserved_in_launch_paths() { + local rec id out status launch linked_home + id=profile-absolute-paths-z1c + rec=$(make_spawn_case profile-absolute-paths pi "$id") + read_case_record "$rec" + linked_home="$CASE_DIR/home-link" + ln -s "$HOME_DIR" "$linked_home" + : > "$LAUNCH_LOG" + + out=$( + FM_ROOT_OVERRIDE='' FM_HOME="$linked_home" \ + FM_STATE_OVERRIDE="$linked_home/state" FM_DATA_OVERRIDE="$linked_home/data" \ + FM_PROJECTS_OVERRIDE="$linked_home/projects" FM_CONFIG_OVERRIDE="$linked_home/config" \ + FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$WT_DIR" TMUX="fake,1,0" \ + CLAUDE_CONFIG_DIR='' FM_FAKE_LAUNCH_LOG="$LAUNCH_LOG" \ + GROK_HOME="$linked_home/grok-home" PATH="$FAKEBIN_DIR:$PATH" \ + "$SPAWN" "$id" "$PROJ_DIR" 2>&1 + ) + status=$? + expect_code 0 "$status" "spawn with absolute symlink-spelled overrides should succeed" + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "-e '$linked_home/state/$id.pi-ext.ts'" \ + "absolute FM_STATE_OVERRIDE spelling changed in Pi's cross-process extension path" + assert_contains "$launch" "< '$linked_home/data/$id/brief.md'" \ + "absolute FM_DATA_OVERRIDE spelling changed in the cross-process brief path" + pass "absolute override spellings are preserved in spawn launch paths" +} + +test_unresolvable_relative_overrides_fail_loudly() { + local rec id out status + id=profile-unresolvable-paths-z1d + rec=$(make_spawn_case profile-unresolvable-paths pi "$id") + read_case_record "$rec" + + out=$( + cd "$CASE_DIR" || exit 1 + FM_ROOT_OVERRIDE='' FM_HOME=missing-home \ + FM_STATE_OVERRIDE='' FM_DATA_OVERRIDE='' \ + "$SPAWN" "$id" "$PROJ_DIR" 2>&1 + ) + status=$? + expect_code 1 "$status" "spawn with an unresolvable relative home should fail" + assert_contains "$out" "FM_HOME directory cannot be resolved: missing-home" \ + "spawn did not name the unresolvable FM_HOME" + + out=$( + cd "$CASE_DIR" || exit 1 + FM_ROOT_OVERRIDE='' FM_HOME=home \ + FM_STATE_OVERRIDE=missing-state FM_DATA_OVERRIDE=home/data \ + "$SPAWN" "$id" "$PROJ_DIR" 2>&1 + ) + status=$? + expect_code 1 "$status" "spawn with an unresolvable relative state override should fail" + assert_contains "$out" "FM_STATE_OVERRIDE directory cannot be resolved: missing-state" \ + "spawn did not name the unresolvable FM_STATE_OVERRIDE" + + out=$( + cd "$CASE_DIR" || exit 1 + FM_ROOT_OVERRIDE='' FM_HOME=home \ + FM_STATE_OVERRIDE=home/state FM_DATA_OVERRIDE=missing-data \ + "$SPAWN" "$id" "$PROJ_DIR" 2>&1 + ) + status=$? + expect_code 1 "$status" "spawn with an unresolvable relative data override should fail" + assert_contains "$out" "FM_DATA_OVERRIDE directory cannot be resolved: missing-data" \ + "spawn did not name the unresolvable FM_DATA_OVERRIDE" + pass "unresolvable relative spawn overrides fail with named diagnostics" +} + test_active_dispatch_profile_requires_explicit_harness_for_ship() { local rec id out status id=profile-required-ship-z11 @@ -342,7 +494,7 @@ test_pi_threads_model_and_max_effort() { expect_code 0 "$status" "pi spawn with max effort should succeed" assert_meta_profile "$HOME_DIR/state/$id.meta" pi openai-codex/gpt-5.6-sol max launch=$(cat "$LAUNCH_LOG") - assert_contains "$launch" "pi --model 'openai-codex/gpt-5.6-sol' --thinking 'max' -e" \ + assert_contains "$launch" "FM_PI_HARNESS=pi pi --model 'openai-codex/gpt-5.6-sol' --thinking 'max' -e" \ "pi launch did not thread the requested model and max thinking level" assert_not_contains "$launch" "FM_FIRSTMATE_PI_LAUNCH_BRIEF=" \ "pi launch still exports the removed Calm input-reroute binding" @@ -351,38 +503,70 @@ test_pi_threads_model_and_max_effort() { pass "pi receives --model and --thinking max profile flags" } -test_quota_selected_default_array_reaches_spawn() { - local rec id quota random selected diagnostic harness model effort out status launch - id=profile-selected-default-z17 - rec=$(make_spawn_case profile-selected-default claude "$id") +test_pi_signed_threads_shared_pi_profile_and_preserves_identity() { + local rec id out status launch + id=profile-pi-signed-z8b + rec=$(make_spawn_case profile-pi-signed pi-signed "$id") read_case_record "$rec" - cat > "$HOME_DIR/config/crew-dispatch.json" <<'JSON' -{"default":[{"harness":"claude","model":"claude-sonnet-5","effort":"low"},{"harness":"codex","model":"gpt-5.5","effort":"high"}]} -JSON - quota="$CASE_DIR/quota.json" - random="$CASE_DIR/random" - printf '\000\000\000\000' > "$random" - cat > "$quota" <<'JSON' -{"schemaVersion":2,"providers":[{"provider":"claude","state":{"status":"fresh"},"windows":[{"id":"five_hour","kind":"session","percentRemaining":10}]},{"provider":"codex","state":{"status":"fresh"},"windows":[{"id":"five_hour","kind":"session","percentRemaining":90}]}]} -JSON - - selected=$(FM_DISPATCH_RANDOM_SOURCE="$random" "$ROOT/bin/fm-dispatch-select.sh" --quota-json "$quota" \ - "$(jq -c .default "$HOME_DIR/config/crew-dispatch.json")" 2>"$CASE_DIR/selection.err") - diagnostic=$(cat "$CASE_DIR/selection.err") - harness=$(printf '%s\n' "$selected" | jq -r .harness) - model=$(printf '%s\n' "$selected" | jq -r .model) - effort=$(printf '%s\n' "$selected" | jq -r .effort) - out=$(run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" \ - "$id" "$PROJ_DIR" --harness "$harness" --model "$model" --effort "$effort") + + out=$(run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR" \ + --model openai-codex/gpt-5.6-sol --effort max) + status=$? + expect_code 0 "$status" "pi-signed spawn with max effort should succeed" + assert_contains "$out" "spawned $id harness=pi-signed" "pi-signed spawn did not preserve its visible identity" + assert_meta_profile "$HOME_DIR/state/$id.meta" pi-signed openai-codex/gpt-5.6-sol max + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "FM_PI_HARNESS=pi-signed pi-signed --model 'openai-codex/gpt-5.6-sol' --thinking 'max' -e" \ + "pi-signed launch did not share Pi's model, thinking, and extension semantics" + assert_contains "$launch" "fm-operational-input.sh' encode launch-brief" \ + "pi-signed launch lost the canonical typed launch-brief envelope" + assert_present "$HOME_DIR/state/$id.pi-ext.ts" "pi-signed launch did not install Pi's turn-end extension" + pass "pi-signed shares Pi launch semantics while preserving its configured and recorded identity" +} + +test_pi_signed_missing_binary_refuses_before_endpoint_or_metadata() { + local rec id out status + id=profile-pi-signed-missing-z8c + rec=$(make_spawn_case profile-pi-signed-missing pi-signed "$id") + read_case_record "$rec" + rm -f "$FAKEBIN_DIR/pi-signed" + : > "$LAUNCH_LOG" + + out=$(FM_ROOT_OVERRIDE='' FM_HOME="$HOME_DIR" \ + FM_STATE_OVERRIDE="$HOME_DIR/state" FM_DATA_OVERRIDE="$HOME_DIR/data" \ + FM_PROJECTS_OVERRIDE="$HOME_DIR/projects" FM_CONFIG_OVERRIDE="$HOME_DIR/config" \ + FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$WT_DIR" TMUX="fake,1,0" \ + FM_FAKE_LAUNCH_LOG="$LAUNCH_LOG" PATH="$FAKEBIN_DIR:/usr/bin:/bin:/usr/sbin:/sbin" \ + "$SPAWN" "$id" "$PROJ_DIR" 2>&1) status=$? + expect_code 1 "$status" "a missing pi-signed executable should refuse the spawn" + assert_contains "$out" "pi-signed executable not found on PATH" \ + "missing pi-signed refusal did not name the actionable requirement" + assert_absent "$HOME_DIR/state/$id.meta" "missing pi-signed refusal wrote task metadata" + [ ! -s "$LAUNCH_LOG" ] || fail "missing pi-signed refusal typed a launch command" + pass "pi-signed refuses safely and actionably when the selected executable is unavailable" +} + +test_pi_signed_persistent_secondmate_uses_pi_extensions_and_identity() { + local rec id sm out status launch + id=profile-pi-signed-secondmate-z8d + rec=$(make_spawn_case profile-pi-signed-secondmate codex "$id") + read_case_record "$rec" + printf '%s\n' pi-signed > "$HOME_DIR/config/secondmate-harness" + sm="$CASE_DIR/secondmate-home" + make_seeded_secondmate_home "$sm" "$id" + sm=$(cd "$sm" && pwd -P) - expect_code 0 "$status" "quota-selected default-array profile should reach spawn" - assert_contains "$diagnostic" "selection basis: quota-selected" "selection did not expose its quota basis" - assert_meta_profile "$HOME_DIR/state/$id.meta" codex gpt-5.5 high + out=$(run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$sm" --secondmate) + status=$? + expect_code 0 "$status" "pi-signed persistent secondmate spawn should succeed" + assert_contains "$out" "spawned $id harness=pi-signed kind=secondmate" \ + "pi-signed secondmate spawn did not preserve its runtime identity" + assert_meta_profile "$HOME_DIR/state/$id.meta" pi-signed default default launch=$(cat "$LAUNCH_LOG") - assert_contains "$launch" "codex --model 'gpt-5.5' -c 'model_reasoning_effort=\"high\"'" \ - "quota-selected default profile did not reach the concrete launch" - pass "top-level default array resolves through quota selection into the real spawn path" + assert_contains "$launch" "FM_PI_HARNESS=pi-signed pi-signed -e '$sm/.pi/extensions/fm-primary-turnend-guard.ts' -e '$sm/.pi/extensions/fm-primary-pi-watch.ts'" \ + "pi-signed secondmate did not share Pi's primary extension launch shape" + pass "pi-signed is a distinct persistent secondmate runtime with shared Pi supervision semantics" } test_batch_forwards_shared_profile_flags() { @@ -404,6 +588,55 @@ test_batch_forwards_shared_profile_flags() { pass "batch dispatch forwards shared --harness, --model, and --effort to every pair" } +test_claude_forwards_firstmate_config_dir_when_set() { + local rec id out status launch + id=profile-claude-cfgdir-z17 + rec=$(make_spawn_case profile-claude-cfgdir claude "$id") + read_case_record "$rec" + + out=$(FM_TEST_CLAUDE_CONFIG_DIR="/opt/test/claude-work" \ + run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR") + status=$? + expect_code 0 "$status" "claude spawn with CLAUDE_CONFIG_DIR set should succeed" + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "CLAUDE_CONFIG_DIR='/opt/test/claude-work' CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false claude" \ + "claude launch did not forward firstmate's CLAUDE_CONFIG_DIR to the crewmate pane" + pass "claude forwards firstmate's CLAUDE_CONFIG_DIR so the crewmate uses the same credential store" +} + +test_claude_omits_config_dir_prefix_when_unset() { + local rec id out status launch + id=profile-claude-nocfgdir-z18 + rec=$(make_spawn_case profile-claude-nocfgdir claude "$id") + read_case_record "$rec" + + # run_spawn pins CLAUDE_CONFIG_DIR empty by default, exercising the single-store + # default path where fm-spawn adds no prefix. + out=$(run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR") + status=$? + expect_code 0 "$status" "claude spawn without CLAUDE_CONFIG_DIR should succeed" + launch=$(cat "$LAUNCH_LOG") + assert_not_contains "$launch" "CLAUDE_CONFIG_DIR=" \ + "claude launch must not add a config-dir prefix when firstmate has no CLAUDE_CONFIG_DIR set" + pass "claude omits the config-dir prefix when firstmate runs with the single-store default" +} + +test_non_claude_harness_ignores_config_dir() { + local rec id out status launch + id=profile-codex-nocfgdir-z19 + rec=$(make_spawn_case profile-codex-nocfgdir codex "$id") + read_case_record "$rec" + + out=$(FM_TEST_CLAUDE_CONFIG_DIR="/opt/test/claude-work" \ + run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR") + status=$? + expect_code 0 "$status" "codex spawn with CLAUDE_CONFIG_DIR set should succeed" + launch=$(cat "$LAUNCH_LOG") + assert_not_contains "$launch" "CLAUDE_CONFIG_DIR=" \ + "non-claude harness launch must not receive the claude-specific config-dir prefix" + pass "non-claude harnesses do not receive the claude CLAUDE_CONFIG_DIR prefix" +} + test_active_dispatch_profile_does_not_block_secondmate_launch() { local rec id sm out status id=profile-secondmate-z16 @@ -423,6 +656,10 @@ test_active_dispatch_profile_does_not_block_secondmate_launch() { } test_no_profile_keeps_claude_profile_defaults +test_relative_home_overrides_launch_with_absolute_cross_process_paths +test_home_defaults_preserve_absolute_or_resolve_relative_paths +test_absolute_override_spelling_is_preserved_in_launch_paths +test_unresolvable_relative_overrides_fail_loudly test_active_dispatch_profile_requires_explicit_harness_for_ship test_active_dispatch_profile_requires_explicit_harness_for_scout test_active_dispatch_profile_allows_explicit_harness @@ -436,8 +673,13 @@ test_grok_omits_invalid_max_reasoning_effort test_grok_omits_invalid_xhigh_reasoning_effort test_opencode_threads_model_and_ignores_effort_axis test_pi_threads_model_and_max_effort -test_quota_selected_default_array_reaches_spawn +test_pi_signed_threads_shared_pi_profile_and_preserves_identity +test_pi_signed_missing_binary_refuses_before_endpoint_or_metadata +test_pi_signed_persistent_secondmate_uses_pi_extensions_and_identity test_batch_forwards_shared_profile_flags +test_claude_forwards_firstmate_config_dir_when_set +test_claude_omits_config_dir_prefix_when_unset +test_non_claude_harness_ignores_config_dir test_active_dispatch_profile_does_not_block_secondmate_launch echo "# all fm-spawn-dispatch-profile tests passed" diff --git a/tests/fm-startup-memory-budget.test.sh b/tests/fm-startup-memory-budget.test.sh new file mode 100755 index 00000000000..813a273ba6e --- /dev/null +++ b/tests/fm-startup-memory-budget.test.sh @@ -0,0 +1,310 @@ +#!/usr/bin/env bash +# Behavioral coverage for the visible startup-memory budget, its safe parser, +# accounting command, primary-to-secondmate convergence, and exact reread bytes. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +BASE_PATH=${FM_TEST_BASE_PATH:-/usr/bin:/bin:/usr/sbin:/sbin} +TMP_ROOT=$(fm_test_tmproot fm-startup-memory-budget) +BUDGET="$ROOT/bin/fm-startup-memory-budget.sh" +BOOTSTRAP="$ROOT/bin/fm-bootstrap.sh" +CONFIG_PUSH="$ROOT/bin/fm-config-push.sh" + +make_fake_toolchain() { + local dir=$1 fakebin + fakebin=$(fm_fakebin "$dir") + fm_fake_exit0 "$fakebin" node gh-axi chrome-devtools-axi lavish-axi quota-axi + cat > "$fakebin/gh" <<'SH' +#!/usr/bin/env bash +exit 0 +SH + cat > "$fakebin/treehouse" <<'SH' +#!/usr/bin/env bash +if [ "${1:-}" = get ] && [ "${2:-}" = --help ]; then + printf '%s\n' 'Usage: treehouse get [--lease]' +fi +SH + cat > "$fakebin/no-mistakes" <<'SH' +#!/usr/bin/env bash +if [ "${1:-}" = --version ]; then + printf '%s\n' 'no-mistakes version v1.31.2 (fake)' +fi +SH + cat > "$fakebin/tasks-axi" <<'SH' +#!/usr/bin/env bash +case "${1:-}:${2:-}" in + --version:*) printf '%s\n' '0.2.3' ;; + update:--help) printf '%s\n' '--archive-body' ;; + mv:--help) printf '%s\n' 'usage: tasks-axi mv <id> [<id>...]' ;; +esac +SH + cat > "$fakebin/tmux" <<'SH' +#!/usr/bin/env bash +[ -z "${FM_FAKE_TMUX_LOG:-}" ] || printf '%s\n' "$*" >> "$FM_FAKE_TMUX_LOG" +case "$*" in + *display-message*'#{pane_current_command}'*) printf '%s\n' codex ;; + *display-message*'#{pane_id}'*) printf '%s\n' '%1' ;; + *display-message*'#{cursor_y}'*) printf '%s\n' 0 ;; + *capture-pane*) printf '\n' ;; +esac +exit 0 +SH + chmod +x "$fakebin"/* + printf '%s\n' "$fakebin" +} + +new_bootstrap_world() { + local name=$1 world root home + world="$TMP_ROOT/$name" + root="$world/root" + home="$world/home" + mkdir -p "$home/config" "$home/data" "$home/state" "$root/bin" + git init -q -b main "$root" + printf '%s\n' 'config/' > "$root/.gitignore" + printf '%s\n' '# Firstmate test root' > "$root/AGENTS.md" + printf '%s\n' '#!/usr/bin/env bash' 'exit 0' > "$root/bin/placeholder.sh" + chmod +x "$root/bin/placeholder.sh" + git -C "$root" add -A + git -C "$root" -c user.name=fmtest -c user.email=fmtest@example.invalid commit -qm initial + printf '%s|%s\n' "$root" "$home" +} + +run_bootstrap() { + local root=$1 home=$2 fakebin=$3 + PATH="$fakebin:$BASE_PATH" FM_BACKEND=tmux FM_HOME="$home" FM_ROOT_OVERRIDE="$root" \ + "$BOOTSTRAP" +} + +test_primary_bootstrap_materializes_visible_default() { + local rec root home fakebin out second + rec=$(new_bootstrap_world materialize) + root=${rec%%|*} + home=${rec#*|} + fakebin=$(make_fake_toolchain "$TMP_ROOT/materialize") + + out=$(run_bootstrap "$root" "$home" "$fakebin") + [ -z "$out" ] || fail "default materialization should stay quiet, got: $out" + [ "$(<"$home/config/startup-memory-budget")" = 7500 ] \ + || fail "bootstrap did not materialize the visible 7500 default" + [ "$(FM_HOME="$home" "$BUDGET" read)" = 7500 ] \ + || fail "read command did not expose the generated default" + + printf '321\n' > "$home/config/startup-memory-budget" + run_bootstrap "$root" "$home" "$fakebin" >/dev/null + [ "$(<"$home/config/startup-memory-budget")" = 321 ] \ + || fail "bootstrap replaced a valid captain-selected budget" + + second="$TMP_ROOT/materialize/secondmate" + mkdir -p "$second/config" "$second/data" "$second/state" + printf '%s\n' sm > "$second/.fm-secondmate-home" + run_bootstrap "$root" "$second" "$fakebin" >/dev/null + [ ! -e "$second/config/startup-memory-budget" ] \ + || fail "secondmate bootstrap created an independent budget instead of awaiting inheritance" + pass "primary bootstrap materializes only the visible default and preserves valid captain choices" +} + +expect_rejected_read() { + local home=$1 expected=$2 out rc + set +e + out=$(FM_HOME="$home" "$BUDGET" read 2>&1) + rc=$? + set -e + [ "$rc" -ne 0 ] || fail "unsafe budget unexpectedly parsed: $expected" + assert_contains "$out" "$expected" "unsafe budget rejection was not specific" +} + +test_safe_parser_rejects_ambiguous_and_unsafe_values() { + local home outside + home="$TMP_ROOT/parser-home" + mkdir -p "$home/config" "$home/data" + printf '42\n' > "$home/config/startup-memory-budget" + [ "$(FM_HOME="$home" "$BUDGET" read)" = 42 ] || fail "valid positive decimal budget was rejected" + + printf '0\n' > "$home/config/startup-memory-budget" + expect_rejected_read "$home" 'value must be one positive decimal integer' + printf '42\nextra\n' > "$home/config/startup-memory-budget" + expect_rejected_read "$home" 'value must be one positive decimal integer' + printf '+42\n' > "$home/config/startup-memory-budget" + expect_rejected_read "$home" 'value must be one positive decimal integer' + + outside="$TMP_ROOT/parser-outside" + printf '77\n' > "$outside" + rm -f "$home/config/startup-memory-budget" + ln -s "$outside" "$home/config/startup-memory-budget" + expect_rejected_read "$home" 'file is symlinked' + [ "$(<"$outside")" = 77 ] || fail "symlink rejection changed its external target" + + rm -f "$home/config/startup-memory-budget" + ln "$outside" "$home/config/startup-memory-budget" + expect_rejected_read "$home" 'file is hardlinked' + [ "$(<"$outside")" = 77 ] || fail "hardlink rejection changed its external source" + + rm -f "$home/config/startup-memory-budget" + rm -rf "$home/config" + ln -s "$TMP_ROOT/parser-config-target" "$home/config" + mkdir -p "$TMP_ROOT/parser-config-target" + printf '88\n' > "$TMP_ROOT/parser-config-target/startup-memory-budget" + expect_rejected_read "$home" 'config directory is symlinked' + pass "budget parser accepts one exact positive value and rejects malformed or unsafe inputs" +} + +test_budget_accounting_reports_all_three_files_and_safe_failure() { + local home out rc outside + home="$TMP_ROOT/accounting-home" + mkdir -p "$home/config" "$home/data" + printf '10\n' > "$home/config/startup-memory-budget" + printf 'abc\n' > "$home/data/captain.md" + printf 'abcdef\n' > "$home/data/captain-shared.md" + + out=$(FM_HOME="$home" "$BUDGET" report) + assert_contains "$out" 'estimator=ceil(UTF-8 bytes / 3) conservative-local-estimate' \ + "report did not name the stable estimator" + assert_contains "$out" 'file=data/captain.md bytes=4 estimated_tokens=2 status=present' \ + "report did not account for captain memory" + assert_contains "$out" 'file=data/captain-shared.md bytes=7 estimated_tokens=3 status=present' \ + "report did not account for shared memory" + assert_contains "$out" 'file=data/learnings.md bytes=0 estimated_tokens=0 status=absent' \ + "report did not account for absent learnings" + assert_contains "$out" 'total_estimated_tokens=5' "report total was not the sum of all three files" + assert_contains "$out" 'budget_status=within-budget' "report did not classify the initial total" + + printf 'abcdefabcdefabcdefabcdef\n' > "$home/data/learnings.md" + out=$(FM_HOME="$home" "$BUDGET" report) + assert_contains "$out" 'budget_status=over-budget' "report did not surface an over-budget total" + + outside="$TMP_ROOT/accounting-outside" + printf 'outside\n' > "$outside" + rm -f "$home/data/captain.md" + ln -s "$outside" "$home/data/captain.md" + set +e + out=$(FM_HOME="$home" "$BUDGET" report 2>&1) + rc=$? + set -e + expect_code 2 "$rc" "unsafe memory input should fail the accounting command" + assert_contains "$out" 'memory file is not an ordinary regular file' \ + "accounting failure did not identify the unsafe memory file" + [ "$(<"$outside")" = outside ] || fail "accounting failure changed a symlink target" + pass "budget accounting sums the three startup files and reports safe failures" +} + +new_propagation_world() { + local world=$1 root="$1/root" home="$1/home" sm="$1/sm" head + mkdir -p "$home/config" "$home/data" "$home/state" "$root/bin" + touch "$home/state/.last-watcher-beat" + git init -q -b main "$root" + printf '%s\n' 'config/' > "$root/.gitignore" + printf '%s\n' '# Firstmate test root' > "$root/AGENTS.md" + printf '%s\n' '#!/usr/bin/env bash' 'exit 0' > "$root/bin/placeholder.sh" + chmod +x "$root/bin/placeholder.sh" + git -C "$root" add -A + git -C "$root" -c user.name=fmtest -c user.email=fmtest@example.invalid commit -qm initial + head=$(git -C "$root" rev-parse HEAD) + git -C "$root" worktree add -q --detach "$sm" "$head" + printf '%s\n' sm > "$sm/.fm-secondmate-home" + mkdir -p "$sm/config" "$sm/data" "$sm/state" "$sm/projects" + { + printf 'window=firstmate:fm-sm\n' + printf 'kind=secondmate\n' + printf 'harness=codex\n' + printf 'home=%s\n' "$sm" + } > "$home/state/sm.meta" + printf '%s|%s|%s\n' "$root" "$home" "$sm" +} + +latest_reread_instruction() { + local home=$1 state path latest= + state=$(cd "$home/state" && pwd -P) || return 1 + for path in "$state"/.fm-inherited-config-reread.*; do + case "$path" in *.pending) continue ;; esac + [ -f "$path" ] && [ ! -L "$path" ] || continue + latest=$path + done + [ -n "$latest" ] || return 1 + printf '%s\n' "$latest" +} + +run_config_push() { + local root=$1 home=$2 fakebin=$3 log=$4 + PATH="$fakebin:$BASE_PATH" FM_HOME="$home" FM_ROOT_OVERRIDE="$root" FM_SEND_SETTLE=0 \ + FM_FAKE_TMUX_LOG="$log" "$CONFIG_PUSH" +} + +test_primary_budget_converges_with_exact_reread_and_safe_failures() { + local world="$TMP_ROOT/propagation" rec root home sm fakebin log out rc instruction expected outside + mkdir -p "$world" + rec=$(new_propagation_world "$world") + root=${rec%%|*} + rec=${rec#*|} + home=${rec%%|*} + sm=${rec#*|} + fakebin=$(make_fake_toolchain "$world") + log="$world/tmux.log" + + printf '321\n' > "$home/config/startup-memory-budget" + out=$(run_config_push "$root" "$home" "$fakebin" "$log") + assert_contains "$out" 'startup-memory-budget: pushed' \ + "config push did not report the new budget as inherited" + [ "$(<"$sm/config/startup-memory-budget")" = 321 ] \ + || fail "secondmate did not receive the primary budget bytes" + instruction=$(latest_reread_instruction "$sm") || fail "budget propagation did not publish a reread instruction" + expected=$(printf '%s\n\n%s\n%s\n321\n%s' \ + 'These inherited config files changed. Re-read and apply their exact contents at every future intake. They are defaults/rules and do not remove your judgment to choose differently when warranted.' \ + 'config/startup-memory-budget' \ + '-----BEGIN config/startup-memory-budget-----' \ + '-----END config/startup-memory-budget-----') + [ "$(<"$instruction")" = "$expected" ] \ + || fail "budget reread payload was not the exact destination bytes" + assert_contains "$(<"$log")" "CONFIG_REREAD: $instruction" \ + "budget propagation did not send the pointer to its exact reread generation" + + outside="$world/unsafe-budget" + printf '555\n' > "$outside" + rm -f "$sm/config/startup-memory-budget" + ln "$outside" "$sm/config/startup-memory-budget" + set +e + out=$(run_config_push "$root" "$home" "$fakebin" "$log" 2>&1) + rc=$? + set -e + expect_code 1 "$rc" "unsafe inherited destination should stop propagation" + assert_contains "$out" 'startup-memory-budget: error - unsafe or invalid destination: file is hardlinked' \ + "unsafe inherited destination did not produce a concrete propagation error" + [ "$(<"$outside")" = 555 ] || fail "unsafe destination handling changed its hardlinked source" + rm -f "$sm/config/startup-memory-budget" + run_config_push "$root" "$home" "$fakebin" "$log" >/dev/null + [ "$(<"$sm/config/startup-memory-budget")" = 321 ] \ + || fail "safe retry did not restore the converged primary budget" + + rm -f "$home/config/startup-memory-budget" + out=$(run_config_push "$root" "$home" "$fakebin" "$log") + assert_contains "$out" 'startup-memory-budget: pushed - mirrored primary absence' \ + "primary absence was not reported as a converging removal" + [ ! -e "$sm/config/startup-memory-budget" ] \ + || fail "primary absence did not remove the inherited budget" + instruction=$(latest_reread_instruction "$sm") || fail "budget absence did not publish a reread instruction" + assert_contains "$(<"$instruction")" $'-----BEGIN config/startup-memory-budget-----\nABSENT\n-----END config/startup-memory-budget-----' \ + "budget absence reread did not use the explicit ABSENT payload" + + rm -f "$sm/config/startup-memory-budget" + printf '555\n' > "$outside" + ln -s "$outside" "$home/config/startup-memory-budget" + set +e + out=$(run_config_push "$root" "$home" "$fakebin" "$log" 2>&1) + rc=$? + set -e + expect_code 1 "$rc" "unsafe primary budget should stop propagation" + assert_contains "$out" 'startup-memory-budget: error - unsafe or invalid primary source: file is symlinked' \ + "unsafe primary budget did not produce a concrete propagation error" + [ ! -e "$sm/config/startup-memory-budget" ] \ + || fail "unsafe primary budget changed the converged secondmate copy" + [ "$(<"$outside")" = 555 ] || fail "unsafe primary budget handling changed its symlink target" + pass "budget propagation converges through config push with exact rereads, absence, and safe rejection" +} + +test_primary_bootstrap_materializes_visible_default +test_safe_parser_rejects_ambiguous_and_unsafe_values +test_budget_accounting_reports_all_three_files_and_safe_failure +test_primary_budget_converges_with_exact_reread_and_safe_failures + +echo '# all fm-startup-memory-budget tests passed' diff --git a/tests/fm-stow-contract.test.sh b/tests/fm-stow-contract.test.sh deleted file mode 100755 index c43f47c4dbb..00000000000 --- a/tests/fm-stow-contract.test.sh +++ /dev/null @@ -1,37 +0,0 @@ -#!/usr/bin/env bash -# Behavior tests for /stow's inspect-then-update memory contract. -set -u - -# shellcheck source=tests/lib.sh disable=SC1091 -. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" - -test_stow_skill_task_note_contract() { - local stow="$ROOT/.agents/skills/stow/SKILL.md" - - assert_grep 'tasks-axi show <id> --full' "$stow" "stow skill does not require inspecting task notes first" - assert_grep 'tasks-axi update <id> --body-file <path>' "$stow" "stow skill does not require task body replacement" - assert_grep '--archive-body' "$stow" "stow skill does not document recoverable task body archival" - assert_grep 'Never append.' "$stow" "stow skill does not forbid append-first task notes" - assert_no_grep 'carry that context into the replacement body' "$stow" "stow skill still preserves archive-only context in the replacement body" - pass "stow skill task-note contract includes recoverable body archival" -} - -test_agents_backlog_task_note_contract() { - local agents="$ROOT/AGENTS.md" - - # shellcheck disable=SC2016 # Literal backticks must remain unexpanded. - assert_grep 'current `tasks-axi --help` own the backlog schema' "$agents" \ - "AGENTS.md does not point exact task-note mechanics to the command owner" - assert_grep 'Inspect the current task note before replacing its considered body' "$agents" \ - "AGENTS.md does not require inspecting task notes before replacement" - assert_grep 'archive the superseded body when recoverability matters rather than appending by default' "$agents" \ - "AGENTS.md lost recoverable replacement and no-append semantics" - assert_no_grep 'tasks-axi show <id> --full' "$agents" \ - "AGENTS.md duplicates exact task-note read syntax from its conditional owner" - assert_no_grep 'tasks-axi update <id> --body-file <path>' "$agents" \ - "AGENTS.md duplicates exact task-note update syntax from its conditional owner" - pass "AGENTS.md keeps task-note hygiene inline and points exact mechanics to their owner" -} - -test_stow_skill_task_note_contract -test_agents_backlog_task_note_contract diff --git a/tests/fm-subagent-pretool-check.test.sh b/tests/fm-subagent-pretool-check.test.sh index 0584d226179..c1a2115897a 100755 --- a/tests/fm-subagent-pretool-check.test.sh +++ b/tests/fm-subagent-pretool-check.test.sh @@ -7,7 +7,6 @@ set -u . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" CHECK="$ROOT/bin/fm-subagent-pretool-check.sh" -SETTINGS="$ROOT/.claude/settings.json" TMP_ROOT=$(fm_test_tmproot fm-subagent-pretool-tests) PRIMARY="$TMP_ROOT/primary" STATE="$PRIMARY/state" @@ -31,6 +30,17 @@ DELEGATION_TOOLS='Task Agent Workflow RemoteTrigger Monitor ScheduleWakeup SendM # Tools that must stay available: denying these would break ordinary work. PRESERVED_TOOLS='Bash Edit Read Write Skill ToolSearch WebFetch WebSearch NotebookEdit ReportFindings DesignSync PushNotification' +# Session-local todo-list tools. They match a delegation stem but create no +# runnable work, so the guard's plan-only exclusion must allow them. +PLAN_ONLY_TOOLS='TaskCreate TaskUpdate' + +# Names the plan-only exclusion must NOT release. Five of them contain a +# plan-only name as a substring and would be let through by a substring rather +# than exact-name match; bare Task is what a shortened entry of "task" would +# release. Together they make the exact-name contract testable instead of +# assumed. +PLAN_ONLY_NEAR_MISSES='TaskCreateAgent TaskCreateWorktree TaskUpdateAgent RemoteTaskCreate Task TaskCreator' + run_tool() { local tool=$1 rc=0 shift @@ -62,20 +72,15 @@ expect_deny() { } # --------------------------------------------------------------------------- -# Tracked settings boundary and delegation-shape PreToolUse guard. +# Delegation-shape PreToolUse guard. # --------------------------------------------------------------------------- -test_tracked_settings_do_not_ship_permissions_deny() { - jq -e 'keys == ["hooks"] and (has("permissions") | not)' "$SETTINGS" >/dev/null \ - || fail "tracked Claude settings must contain only hooks and no permissions key" - pass "tracked Claude settings do not ship permissions.deny" -} - test_guard_denies_every_currently_known_delegation_tool() { local tool for tool in $DELEGATION_TOOLS; do case "$tool" in TaskOutput|TaskStop|TaskGet|TaskList|CronList) continue ;; + TaskCreate|TaskUpdate) continue ;; esac expect_deny "known delegation tool" "$tool" done @@ -107,6 +112,28 @@ test_guard_allows_ordinary_and_observe_only_tools() { pass "the guard leaves ordinary tools and observe-or-stop operations alone" } +test_guard_allows_session_local_todo_tools() { + # These write, so they are not observe-or-stop, but what they write is the + # harness's session-local todo list: no executor, no agent, no worktree, no + # schedule, nothing that outlives the session. Denying them stops the primary + # tracking its own plan and grants no delegation power in exchange. + local tool + for tool in $PLAN_ONLY_TOOLS; do + expect_allow "session-local todo tool" "$tool" + done + pass "the guard leaves the session-local todo list alone" +} + +test_plan_only_exclusion_is_exact_name() { + # The plan-only exclusion must never widen by substring or by a shorter stem. + # Every name here would be released by such a widening and must stay denied. + local tool + for tool in $PLAN_ONLY_NEAR_MISSES; do + expect_deny "plan-only near miss" "$tool" + done + pass "the plan-only exclusion releases exactly two names and nothing that merely contains them" +} + test_guard_never_classifies_mcp_tools() { # An MCP server names its own tools; a task or agent noun there is common and # has nothing to do with fleet dispatch. @@ -249,33 +276,11 @@ test_missing_jq_stdin_transport_fails_open() { pass "missing jq for stdin transport fails open rather than denying every tool call" } -test_claude_hook_registration_preserves_bash_seatbelts() { - jq -e ' - [.hooks.PreToolUse[] | .hooks[].command] - | any(contains("fm-subagent-pretool-check.sh --claude")) - ' "$SETTINGS" >/dev/null || fail "Claude settings omit the delegation-shape PreToolUse guard" - # A stem-enumerating matcher repeats the fail-open-by-enumeration defect the - # script exists to remove. Match all tools and let the script be the single - # owner of classification. - jq -e ' - [.hooks.PreToolUse[] | select(.hooks[].command | contains("fm-subagent-pretool-check.sh")) | .matcher] | .[0] - | . == ".*" - ' "$SETTINGS" >/dev/null || fail "the guard matcher must match all tools" - jq -e ' - [.hooks.PreToolUse[] | select(.matcher == "Bash") | .hooks[].command] as $bash - | ($bash | any(contains("fm-arm-pretool-check.sh"))) - and ($bash | any(contains("fm-cd-pretool-check.sh"))) - and ($bash | any(contains("fm-continuity-pretool-check.sh"))) - ' "$SETTINGS" >/dev/null || fail "the existing Bash PreToolUse seatbelts changed" - jq -e '.hooks.Stop[0].hooks[0].command | contains("fm-turnend-guard.sh")' "$SETTINGS" >/dev/null \ - || fail "the Stop turn-end guard changed" - pass "Claude wires the guard while preserving the Bash seatbelts and the Stop guard" -} - -test_tracked_settings_do_not_ship_permissions_deny test_guard_denies_every_currently_known_delegation_tool test_guard_denies_hypothetical_future_tools test_guard_allows_ordinary_and_observe_only_tools +test_guard_allows_session_local_todo_tools +test_plan_only_exclusion_is_exact_name test_guard_never_classifies_mcp_tools test_deny_message_defers_to_intake_classification test_escape_hatch_allows_deliberate_use @@ -284,4 +289,3 @@ test_secondmate_home_is_in_scope test_stdin_transports_and_output_shapes test_malformed_transport_fails_open test_missing_jq_stdin_transport_fails_open -test_claude_hook_registration_preserves_bash_seatbelts diff --git a/tests/fm-supervision-instructions.test.sh b/tests/fm-supervision-instructions.test.sh index 95858d3c931..e8e5f4f919b 100755 --- a/tests/fm-supervision-instructions.test.sh +++ b/tests/fm-supervision-instructions.test.sh @@ -14,7 +14,7 @@ test_selected_harness_block_only() { assert_contains "$out" "SUPERVISION OPERATING INSTRUCTIONS - primary harness: codex" "codex heading missing" assert_contains "$out" "Mode: Codex foreground checkpoint." "codex snippet missing" assert_contains "$out" "bin/fm-watch-checkpoint.sh" "codex checkpoint helper missing" - assert_not_contains "$out" "Mode: Claude background-notify supervision." "renderer printed the claude snippet too" + assert_not_contains "$out" "Mode: Claude Stop-hook-owned supervision." "renderer printed the claude snippet too" assert_not_contains "$out" "Mode: Pi extension background wake." "renderer printed the pi snippet too" pass "renderer prints exactly the selected harness block" } @@ -86,9 +86,10 @@ test_cross_harness_ordinary_continuation_and_repair_matrix() { out=$("$RENDER" --harness claude) ordinary=$(printf '%s\n' "$out" | grep -F -- '- Ordinary wake:') - assert_contains "$ordinary" "re-arm" "claude ordinary-wake line does not tell the model to re-arm" - assert_contains "$ordinary" "Claude Code background task" "claude ordinary-wake line lost tracked background ownership" - assert_contains "$ordinary" "bin/fm-watch-arm.sh" "claude ordinary-wake line lost the background arm command" + assert_contains "$ordinary" "Stop-owned auto-arm" "claude ordinary-wake line does not leave continuity to the Stop hook" + assert_contains "$ordinary" "bin/fm-claude-stop-autoarm.sh" "claude ordinary-wake line lost the auto-arm script name" + assert_contains "$ordinary" "do not arm another cycle" "claude ordinary-wake line does not forbid a model re-arm" + assert_not_contains "$ordinary" "bin/fm-watch-arm.sh" "claude ordinary-wake line incorrectly calls the manual arm" out=$("$RENDER" --harness claude --repair-line) assert_contains "$out" "Claude Code background task" "claude recovery line lost its tracked background repair" assert_contains "$out" "bin/fm-watch-arm.sh" "claude recovery line lost the arm command" @@ -114,6 +115,22 @@ test_cross_harness_ordinary_continuation_and_repair_matrix() { pass "renderer preserves every harness ordinary-continuation and missing-cycle repair path" } +test_pi_signed_preserves_identity_with_pi_supervision_protocol() { + local out ordinary + out=$("$RENDER" --harness pi-signed) + assert_contains "$out" "primary harness: pi-signed" \ + "pi-signed supervision normalized the visible runtime identity to pi" + assert_contains "$out" "Mode: Pi extension background wake." \ + "pi-signed did not reuse Pi's authoritative supervision protocol" + ordinary=$(printf '%s\n' "$out" | grep -F -- '- Ordinary wake:') + assert_contains "$ordinary" "Pi extension already owns watcher continuity" \ + "pi-signed ordinary-wake semantics diverged from Pi" + out=$("$RENDER" --harness pi-signed --repair-line) + assert_contains "$out" "Pi tool fm_watch_arm_pi" \ + "pi-signed repair semantics diverged from Pi" + pass "pi-signed keeps its identity while sharing Pi's supervision protocol" +} + test_grok_is_background_notify() { local out out=$("$RENDER" --harness grok) @@ -159,6 +176,7 @@ test_unknown_fallback test_conditional_stanzas test_repair_lines test_cross_harness_ordinary_continuation_and_repair_matrix +test_pi_signed_preserves_identity_with_pi_supervision_protocol test_grok_is_background_notify test_grok_command_sources_effective_config test_pi_snippet_uses_effective_extension_path diff --git a/tests/fm-teardown-endpoint-safety.test.sh b/tests/fm-teardown-endpoint-safety.test.sh new file mode 100755 index 00000000000..5786102cd3e --- /dev/null +++ b/tests/fm-teardown-endpoint-safety.test.sh @@ -0,0 +1,275 @@ +#!/usr/bin/env bash +# Regression tests for cleanup endpoint identity validation. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +TEARDOWN="$ROOT/bin/fm-teardown.sh" +TMP_ROOT=$(fm_test_tmproot fm-teardown-endpoint-safety) +REAL_TMUX=$(command -v tmux || true) + +make_case() { # <name> + local dir=$1 + mkdir -p "$TMP_ROOT/$dir/home/state" "$TMP_ROOT/$dir/home/data" \ + "$TMP_ROOT/$dir/home/config" "$TMP_ROOT/$dir/fakebin" \ + "$TMP_ROOT/$dir/worktree" "$TMP_ROOT/$dir/project" + : > "$TMP_ROOT/$dir/worktree/sentinel" + : > "$TMP_ROOT/$dir/runtime.log" + cat > "$TMP_ROOT/$dir/fakebin/tmux" <<'SH' +#!/usr/bin/env bash +printf 'tmux' >> "${FM_RUNTIME_LOG:?}" +printf ' <%s>' "$@" >> "${FM_RUNTIME_LOG:?}" +printf '\n' >> "${FM_RUNTIME_LOG:?}" +exit 0 +SH + cat > "$TMP_ROOT/$dir/fakebin/treehouse" <<'SH' +#!/usr/bin/env bash +printf 'treehouse' >> "${FM_RUNTIME_LOG:?}" +printf ' <%s>' "$@" >> "${FM_RUNTIME_LOG:?}" +printf '\n' >> "${FM_RUNTIME_LOG:?}" +exit 0 +SH + chmod +x "$TMP_ROOT/$dir/fakebin/tmux" "$TMP_ROOT/$dir/fakebin/treehouse" + printf '%s\n' "$TMP_ROOT/$dir" +} + +run_case() { # <case> <id> + local dir=$1 id=$2 + FM_HOME="$dir/home" FM_ROOT_OVERRIDE="$ROOT" \ + FM_RUNTIME_LOG="$dir/runtime.log" PATH="$dir/fakebin:$PATH" \ + "$TEARDOWN" "$id" --force +} + +assert_refused_without_mutation() { # <case> <id> <description> + local dir=$1 id=$2 description=$3 rc + set +e + run_case "$dir" "$id" > "$dir/stdout" 2> "$dir/stderr" + rc=$? + set -e + [ "$rc" -ne 0 ] || fail "$description: teardown unexpectedly succeeded" + assert_present "$dir/home/state/$id.meta" "$description: metadata changed before refusal" + assert_present "$dir/worktree/sentinel" "$description: worktree changed before refusal" + [ ! -s "$dir/runtime.log" ] || fail "$description: runtime command ran before refusal: $(cat "$dir/runtime.log")" +} + +test_invalid_endpoint_records_refuse_before_mutation() { + local dir id=endpoint-a + + dir=$(make_case missing) + fm_write_meta "$dir/home/state/$id.meta" \ + "worktree=$dir/worktree" "project=$dir/project" "kind=scout" + assert_refused_without_mutation "$dir" "$id" "missing endpoint" + + dir=$(make_case empty) + fm_write_meta "$dir/home/state/$id.meta" \ + "window=" "worktree=$dir/worktree" "project=$dir/project" "kind=scout" + assert_refused_without_mutation "$dir" "$id" "empty endpoint" + + dir=$(make_case malformed) + fm_write_meta "$dir/home/state/$id.meta" \ + "window=ambient-current-window" "worktree=$dir/worktree" \ + "project=$dir/project" "kind=scout" + assert_refused_without_mutation "$dir" "$id" "malformed endpoint" + + dir=$(make_case mismatched) + fm_write_meta "$dir/home/state/$id.meta" \ + "window=isolated:fm-other-task" "endpoint_task_id=other-task" \ + "worktree=$dir/worktree" "project=$dir/project" "kind=scout" + assert_refused_without_mutation "$dir" "$id" "task-mismatched endpoint" + + dir=$(make_case empty-binding) + fm_write_meta "$dir/home/state/$id.meta" \ + "window=isolated:fm-$id" "endpoint_task_id=" \ + "worktree=$dir/worktree" "project=$dir/project" "kind=scout" + assert_refused_without_mutation "$dir" "$id" "empty task binding" + + dir=$(make_case duplicate-binding) + fm_write_meta "$dir/home/state/$id.meta" \ + "window=isolated:fm-$id" "endpoint_task_id=$id" "endpoint_task_id=$id" \ + "worktree=$dir/worktree" "project=$dir/project" "kind=scout" + assert_refused_without_mutation "$dir" "$id" "duplicate task binding" + + pass "fm-teardown: missing, empty, malformed, ambiguous, and task-mismatched endpoints refuse before every mutation or runtime call" +} + +test_supported_backend_endpoint_records_validate() { + local dir id backend target + dir=$(make_case valid-backends) + # shellcheck source=/dev/null + . "$ROOT/bin/fm-backend.sh" + + id=tmux-task + fm_write_meta "$dir/home/state/$id.meta" \ + "window=firstmate:fm-$id" "worktree=$dir/worktree" "project=$dir/project" + fm_backend_validate_task_endpoint "$dir/home/state/$id.meta" "$id" || fail "valid tmux endpoint refused" + [ "$FM_BACKEND_VALIDATED_BACKEND:$FM_BACKEND_VALIDATED_TARGET" = "tmux:firstmate:fm-$id" ] || fail "tmux endpoint validation returned wrong identity" + + id=tmux-spaced-session + fm_write_meta "$dir/home/state/$id.meta" \ + "window=team work:fm-$id" "worktree=$dir/worktree" "project=$dir/project" + fm_backend_validate_task_endpoint "$dir/home/state/$id.meta" "$id" || fail "valid tmux endpoint with a spaced session name refused" + [ "$FM_BACKEND_VALIDATED_TARGET" = "team work:fm-$id" ] || fail "tmux validation changed the spaced session identity" + + id=herdr-task + fm_write_meta "$dir/home/state/$id.meta" \ + "window=lab:w1:p2" "endpoint_task_id=$id" "worktree=$dir/worktree" "project=$dir/project" \ + "backend=herdr" "herdr_session=lab" "herdr_workspace_id=w1" "herdr_tab_id=w1:t2" "herdr_pane_id=w1:p2" + fm_backend_validate_task_endpoint "$dir/home/state/$id.meta" "$id" || fail "valid Herdr endpoint refused" + + id=zellij-task + fm_write_meta "$dir/home/state/$id.meta" \ + "window=lab:7" "endpoint_task_id=$id" "worktree=$dir/worktree" "project=$dir/project" \ + "backend=zellij" "zellij_session=lab" "zellij_tab_id=3" "zellij_pane_id=7" + fm_backend_validate_task_endpoint "$dir/home/state/$id.meta" "$id" || fail "valid Zellij endpoint refused" + + id=orca-task + fm_write_meta "$dir/home/state/$id.meta" \ + "window=fm-$id" "endpoint_task_id=$id" "terminal=term-7" \ + "worktree=$dir/worktree" "project=$dir/project" "backend=orca" "orca_worktree_id=worktree-9" + fm_backend_validate_task_endpoint "$dir/home/state/$id.meta" "$id" || fail "valid Orca endpoint refused" + [ "$FM_BACKEND_VALIDATED_TARGET" = term-7 ] || fail "Orca validation did not select its terminal" + + id=cmux-task + fm_write_meta "$dir/home/state/$id.meta" \ + "window=workspace-1:surface-2" "endpoint_task_id=$id" "worktree=$dir/worktree" "project=$dir/project" \ + "backend=cmux" "cmux_workspace_id=workspace-1" "cmux_surface_id=surface-2" + fm_backend_validate_task_endpoint "$dir/home/state/$id.meta" "$id" || fail "valid cmux endpoint refused" + + for backend in tmux herdr zellij orca cmux; do + set +e + fm_backend_kill "$backend" "" >/dev/null 2>&1 + target=$? + set -e + [ "$target" -ne 0 ] || fail "$backend generic kill accepted an empty target" + done + pass "cleanup identity: valid tmux, Herdr, Zellij, Orca, and cmux records validate while every empty backend target refuses" +} + +test_tmux_empty_target_refuses_without_invocation() { + local dir rc + dir=$(make_case direct-empty) + set +e + FM_RUNTIME_LOG="$dir/runtime.log" PATH="$dir/fakebin:$PATH" \ + bash -c '. "$1/bin/fm-backend.sh"; fm_backend_source tmux; fm_backend_tmux_kill ""' _ "$ROOT" \ + > "$dir/stdout" 2> "$dir/stderr" + rc=$? + set -e + [ "$rc" -ne 0 ] || fail "direct empty tmux target unexpectedly succeeded" + [ ! -s "$dir/runtime.log" ] || fail "direct empty tmux target invoked tmux" + pass "tmux backend: direct empty target returns nonzero without invoking tmux" +} + +test_recorded_process_identity_cleanup_is_exact() { + local dir target_pid control_pid target_record control_record live_command + dir=$(make_case recorded-process) + sleep 30 & + control_pid=$! + sleep 30 & + target_pid=$! + printf '%s\n' "$control_pid" > "$dir/control.pid" + printf '%s\n' "$target_pid" > "$dir/target.pid" + target_record=$(cat "$dir/target.pid") + control_record=$(cat "$dir/control.pid") + [ "$target_record" = "$target_pid" ] && [ "$control_record" = "$control_pid" ] \ + || fail "recorded process identity changed before cleanup" + live_command=$(ps -p "$target_record" -o comm= 2>/dev/null | tr -d '[:space:]') + case "$live_command" in sleep) ;; *) fail "recorded target pid no longer belongs to the expected child" ;; esac + kill -TERM "$target_record" + wait "$target_record" 2>/dev/null || true + kill -0 "$target_record" 2>/dev/null && fail "exact target pid survived cleanup" + kill -0 "$control_record" 2>/dev/null || fail "independent control process was disturbed" + kill -TERM "$control_record" + wait "$control_record" 2>/dev/null || true + pass "process cleanup: creation-time PID identity removes only the exact child and preserves the control child" +} + +isolated_tmux_window_exists() { # <dir> <socket> <session> <window> + ( cd "$1" && "$REAL_TMUX" -S "$2" list-windows -t "$3" -F '#{window_name}' 2>/dev/null ) \ + | grep -Fqx "$4" +} + +test_isolated_tmux_invalid_and_valid_cleanup() { + local dir socket socket_id session='endpoint safety' target_id=target control=control target=fm-target + local prefix_target=fm-prefix prefix_survivor=fm-prefix2 rc + [ -n "$REAL_TMUX" ] || { echo "skip - tmux not installed"; return 0; } + dir=$(make_case isolated-real) + socket=dedicated.sock + socket_id="$dir/$socket" + ( cd "$dir" && env -u TMUX -u TMUX_PANE "$REAL_TMUX" -S "$socket" new-session -d -s "$session" -n "$control" ) + ( cd "$dir" && env -u TMUX -u TMUX_PANE "$REAL_TMUX" -S "$socket" new-window -d -t "$session:" -n "$target" ) + printf '%s\n' "$socket_id" > "$dir/socket.identity" + cat > "$dir/fakebin/tmux" <<SH +#!/usr/bin/env bash +set -eu +[ -z "\${TMUX:-}" ] && [ -z "\${TMUX_PANE:-}" ] || exit 91 +[ "\${FM_TEST_TMUX_SOCKET:-}" = '$socket_id' ] || exit 92 +[ "\$(cat '$dir/socket.identity')" = '$socket_id' ] || exit 93 +printf 'tmux' >> "\${FM_RUNTIME_LOG:?}" +printf ' <%s>' "\$@" >> "\${FM_RUNTIME_LOG:?}" +printf '\n' >> "\${FM_RUNTIME_LOG:?}" +cd '$dir' +exec '$REAL_TMUX' -S '$socket' "\$@" +SH + chmod +x "$dir/fakebin/tmux" + + fm_write_meta "$dir/home/state/invalid.meta" \ + "window=" "worktree=$dir/worktree" "project=$dir/project" "kind=scout" + set +e + env -u TMUX -u TMUX_PANE FM_TEST_TMUX_SOCKET="$socket_id" \ + FM_HOME="$dir/home" FM_ROOT_OVERRIDE="$ROOT" FM_RUNTIME_LOG="$dir/runtime.log" \ + PATH="$dir/fakebin:$PATH" "$TEARDOWN" invalid --force \ + > "$dir/invalid.out" 2> "$dir/invalid.err" + rc=$? + set -e + [ "$rc" -ne 0 ] || fail "isolated invalid endpoint unexpectedly succeeded" + [ ! -s "$dir/runtime.log" ] || fail "isolated invalid endpoint reached tmux" + isolated_tmux_window_exists "$dir" "$socket" "$session" "$control" || fail "invalid cleanup removed control window" + isolated_tmux_window_exists "$dir" "$socket" "$session" "$target" || fail "invalid cleanup removed target window" + + set +e + # shellcheck disable=SC2016 # $1 expands inside the isolated child shell. + env -u TMUX -u TMUX_PANE FM_TEST_TMUX_SOCKET="$socket_id" FM_RUNTIME_LOG="$dir/runtime.log" \ + PATH="$dir/fakebin:$PATH" bash -c \ + '. "$1/bin/fm-backend.sh"; fm_backend_source tmux; fm_backend_tmux_kill ""' _ "$ROOT" \ + > "$dir/empty.out" 2> "$dir/empty.err" + rc=$? + set -e + [ "$rc" -ne 0 ] || fail "isolated direct empty target unexpectedly succeeded" + [ ! -s "$dir/runtime.log" ] || fail "isolated direct empty target reached tmux" + isolated_tmux_window_exists "$dir" "$socket" "$session" "$control" || fail "direct empty cleanup removed control window" + isolated_tmux_window_exists "$dir" "$socket" "$session" "$target" || fail "direct empty cleanup removed target window" + + ( cd "$dir" && env -u TMUX -u TMUX_PANE "$REAL_TMUX" -S "$socket" new-window -d -t "=$session:" -n "$prefix_survivor" ) + # shellcheck disable=SC2016 # $1 and $2 expand inside the isolated child shell. + env -u TMUX -u TMUX_PANE FM_TEST_TMUX_SOCKET="$socket_id" FM_RUNTIME_LOG="$dir/runtime.log" \ + PATH="$dir/fakebin:$PATH" bash -c \ + '. "$1/bin/fm-backend.sh"; fm_backend_source tmux; fm_backend_tmux_kill "$2"' _ "$ROOT" "$session:$prefix_target" + isolated_tmux_window_exists "$dir" "$socket" "$session" "$prefix_survivor" \ + || fail "missing exact target cleanup removed its prefix-matched neighbor" + + fm_write_meta "$dir/home/state/$target_id.meta" \ + "window=$session:$target" "endpoint_task_id=$target_id" \ + "worktree=$dir/nonexistent-worktree" "project=$dir/nonexistent-project" \ + "kind=scout" "mode=no-mistakes" + env -u TMUX -u TMUX_PANE FM_TEST_TMUX_SOCKET="$socket_id" \ + FM_HOME="$dir/home" FM_ROOT_OVERRIDE="$ROOT" FM_RUNTIME_LOG="$dir/runtime.log" \ + PATH="$dir/fakebin:$PATH" "$TEARDOWN" "$target_id" --force \ + > "$dir/valid.out" 2> "$dir/valid.err" \ + || fail "isolated valid endpoint teardown failed: $(cat "$dir/valid.err")" + isolated_tmux_window_exists "$dir" "$socket" "$session" "$target" \ + && fail "valid cleanup did not remove the exact target window" + isolated_tmux_window_exists "$dir" "$socket" "$session" "$control" \ + || fail "valid cleanup removed the independent control window" + grep -Fqx "tmux <kill-window> <-t> <=$session:=$target>" "$dir/runtime.log" \ + || fail "valid cleanup did not invoke exactly the recorded target: $(cat "$dir/runtime.log")" + + ( cd "$dir" && env -u TMUX -u TMUX_PANE "$REAL_TMUX" -S "$socket" kill-server 2>/dev/null ) || true + pass "fm-teardown: exact tmux cleanup preserves invalid and prefix-matched neighbors while removing only the recorded target" +} + +test_invalid_endpoint_records_refuse_before_mutation +test_supported_backend_endpoint_records_validate +test_tmux_empty_target_refuses_without_invocation +test_recorded_process_identity_cleanup_is_exact +test_isolated_tmux_invalid_and_valid_cleanup diff --git a/tests/fm-teardown.test.sh b/tests/fm-teardown.test.sh index c3fa616ac8d..a57a08f6b35 100755 --- a/tests/fm-teardown.test.sh +++ b/tests/fm-teardown.test.sh @@ -155,7 +155,8 @@ SH write_meta() { local case_dir=$1 mode=$2 kind=$3 fm_write_meta "$case_dir/state/task-x1.meta" \ - "window=fm-task-x1" \ + "window=firstmate:fm-task-x1" \ + "endpoint_task_id=task-x1" \ "worktree=$case_dir/wt" \ "project=$case_dir/project" \ "kind=$kind" \ @@ -1248,7 +1249,12 @@ test_herdr_teardown_clears_escalation_marker() { write_meta "$case_dir" local-only ship sed -i.bak 's/^window=.*/window=default:wG:pQ/' "$case_dir/state/task-x1.meta" rm -f "$case_dir/state/task-x1.meta.bak" - printf '%s\n' 'backend=herdr' >> "$case_dir/state/task-x1.meta" + printf '%s\n' \ + 'backend=herdr' \ + 'herdr_session=default' \ + 'herdr_workspace_id=wG' \ + 'herdr_tab_id=wG:tQ' \ + 'herdr_pane_id=wG:pQ' >> "$case_dir/state/task-x1.meta" cat > "$case_dir/fakebin/herdr" <<'SH' #!/usr/bin/env bash exit 0 diff --git a/tests/fm-test-isolation-proof.test.sh b/tests/fm-test-isolation-proof.test.sh index ff9e388da1e..1847338e8cd 100755 --- a/tests/fm-test-isolation-proof.test.sh +++ b/tests/fm-test-isolation-proof.test.sh @@ -1,24 +1,12 @@ #!/usr/bin/env bash -# Contract tests for bin/fm-test-isolation-proof.sh - the Phase 2 pre-shard -# isolation proof harness. -# -# These tests assert the candidate-set contract, serial exclusions, aggregate -# failure reporting, and that Phase 4 production shards consume this exact set. -# They deliberately do NOT re-run the full concurrent candidate matrix on every -# invocation (that matrix is owned by the harness itself and archived under -# docs/fm-test-isolation-proof.md after a deliberate proof run). +# Behavioral tests for the isolation-proof and test-run public interfaces. set -u -# shellcheck disable=SC1091 # shellcheck source=tests/lib.sh . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" PROOF="$ROOT/bin/fm-test-isolation-proof.sh" RUNNER="$ROOT/bin/fm-test-run.sh" -CI="$ROOT/.github/workflows/ci.yml" -CONTRIB="$ROOT/CONTRIBUTING.md" -PROOF_DOC="$ROOT/docs/fm-test-isolation-proof.md" -PROOF_JSON="$ROOT/docs/fm-test-isolation-proof.json" assert_present "$PROOF" "bin/fm-test-isolation-proof.sh is missing" [ -x "$PROOF" ] || fail "bin/fm-test-isolation-proof.sh must be executable" @@ -31,7 +19,6 @@ test_list_candidates_nonempty_and_stable() { [ "$count" -ge 10 ] || fail "expected a bounded non-trivial candidate set, got $count" sorted=$(printf '%s\n' "$listed" | LC_ALL=C sort) [ "$listed" = "$sorted" ] || fail "--list must be sorted for a stable matrix" - # No duplicates. [ "$(printf '%s\n' "$listed" | uniq | wc -l | tr -d ' ')" = "$count" ] \ || fail "--list must not duplicate candidates" while IFS= read -r line; do @@ -47,14 +34,8 @@ test_list_candidates_nonempty_and_stable() { test_candidates_exclude_serial_classes() { local listed listed=$("$PROOF" --list) - # Self must never re-enter the concurrent matrix. - printf '%s\n' "$listed" | grep -Fq 'tests/fm-test-isolation-proof.test.sh' \ - && fail "isolation-proof test must not be a parallel candidate" - # Continuity fixture starts a background sleep holder. - printf '%s\n' "$listed" | grep -Fq 'tests/fm-continuity-pretool-check.test.sh' \ - && fail "continuity pretool check must stay serial (process holder)" - # Real tmux smoke, watcher lock, real herdr, AFK, live harnesses stay serial. for banned in \ + tests/fm-test-isolation-proof.test.sh \ tests/fm-backend-tmux-smoke.test.sh \ tests/fm-watcher-lock.test.sh \ tests/fm-wake-queue.test.sh \ @@ -69,16 +50,6 @@ test_candidates_exclude_serial_classes() { pass "serial classes remain excluded from the parallel candidate set" } -test_candidates_match_archived_proof() { - local listed archived - assert_present "$PROOF_JSON" "docs/fm-test-isolation-proof.json missing" - listed=$("$PROOF" --list) - archived=$(jq -r '.scripts[].path' "$PROOF_JSON" | LC_ALL=C sort) - [ "$listed" = "$archived" ] \ - || fail "candidate set must exactly match the archived isolation proof" - pass "candidate set exactly matches the archived isolation proof" -} - test_extra_hermetic_candidates_present() { local listed listed=$("$PROOF" --list) @@ -92,15 +63,13 @@ test_extra_hermetic_candidates_present() { printf '%s\n' "$listed" | grep -Fxq "$want" \ || fail "extra hermetic candidate missing: $want" done - pass "audited fake-backend / stub-network extras are candidates" + pass "audited fake-backend and stub-network extras are candidates" } test_list_exclusions_documents_reasons() { local out out=$("$PROOF" --list-exclusions) [ -n "$out" ] || fail "--list-exclusions printed nothing" - printf '%s\n' "$out" | grep -Fq 'fm-continuity-pretool-check.test.sh' \ - || fail "exclusions must document continuity process-holder reason" printf '%s\n' "$out" | grep -Fq 'fm-watcher-lock.test.sh' \ || fail "exclusions must document watcher-lock serial reason" printf '%s\n' "$out" | grep -Fq 'fm-backend-herdr-smoke.test.sh' \ @@ -116,82 +85,7 @@ test_family_map_labels_this_contract() { pass "isolation-proof contract test is family-mapped" } -test_aggregate_failure_under_concurrency() { - local tmp pass_f fail_f harness rc out - tmp=$(mktemp -d "${TMPDIR:-/tmp}/fm-isolation-agg.XXXXXX") - pass_f="$tmp/pass.test.sh" - fail_f="$tmp/fail.test.sh" - cat >"$pass_f" <<'SH' -#!/usr/bin/env bash -echo "ok - pass" -exit 0 -SH - cat >"$fail_f" <<'SH' -#!/usr/bin/env bash -echo "not ok - fail" -exit 1 -SH - chmod +x "$pass_f" "$fail_f" - # Minimal fixture harness mirroring aggregate + concurrent wait semantics. - harness="$tmp/harness.sh" - cat >"$harness" <<'SH' -#!/usr/bin/env bash -set -eu -jobs=$1 -shift -pids=() -rcs=() -paths=() -idx=0 -for s in "$@"; do - idx=$((idx + 1)) - ( - bash "$s" - echo $? >"${TMPDIR:-/tmp}/iso-rc-$idx" - ) & - pids+=("$!") - paths+=("$s") - while [ "${#pids[@]}" -ge "$jobs" ]; do - wait "${pids[0]}" || true - pids=("${pids[@]:1}") - done -done -while [ "${#pids[@]}" -gt 0 ]; do - wait "${pids[0]}" || true - pids=("${pids[@]:1}") -done -failed=0 -for i in $(seq 1 "$idx"); do - rc=$(cat "${TMPDIR:-/tmp}/iso-rc-$i" 2>/dev/null || echo 1) - [ "$rc" -eq 0 ] || failed=$((failed + 1)) - rm -f "${TMPDIR:-/tmp}/iso-rc-$i" -done -echo "FM_ISOLATION_SUMMARY total=$idx failed=$failed" -[ "$failed" -eq 0 ] -SH - chmod +x "$harness" - set +e - out=$(TMPDIR="$tmp" bash "$harness" 2 "$pass_f" "$fail_f" 2>&1) - rc=$? - set -e - [ "$rc" -ne 0 ] || fail "concurrent aggregate must fail when any candidate fails" - printf '%s\n' "$out" | grep -Fq 'FM_ISOLATION_SUMMARY total=2 failed=1' \ - || fail "aggregate summary must report total=2 failed=1: $out" - rm -rf "$tmp" - pass "aggregate failure reporting survives concurrency" -} - -test_phase4_consumes_proven_set_only() { - assert_present "$CI" "ci.yml missing" - assert_present "$RUNNER" "fm-test-run.sh missing" - # Phase 4 portable parallel lanes must exist and use lane selection, not --all. - grep -Fq 'bin/fm-test-run.sh --lane portable-parallel-1' "$CI" \ - || fail "CI portable parallel 1 must use --lane portable-parallel-1" - grep -Fq 'bin/fm-test-run.sh --lane portable-parallel-2' "$CI" \ - || fail "CI portable parallel 2 must use --lane portable-parallel-2" - grep -Fq 'bin/fm-test-run.sh --lane portable-serial' "$CI" \ - || fail "CI portable serial must use --lane portable-serial" - # Shard union must equal this harness's proven list. +test_parallel_shards_consume_the_proven_set() { local proven shards proven=$("$PROOF" --list | LC_ALL=C sort -u) shards=$( @@ -202,33 +96,12 @@ test_phase4_consumes_proven_set_only() { ) [ "$proven" = "$shards" ] \ || fail "portable parallel shards must equal isolation-proof --list exactly" - # Local --jobs is bounded to this proven set (refuse is contract-tested in - # fm-test-run.test.sh); the option must exist. - grep -E '^[[:space:]]*--jobs\)' "$RUNNER" >/dev/null 2>&1 \ - || fail "fm-test-run.sh must expose bounded --jobs after Phase 4" - pass "Phase 4 portable shards consume the proven-isolated set only" -} - -test_docs_record_proof_owner() { - assert_present "$PROOF_DOC" "docs/fm-test-isolation-proof.md missing" - grep -Fq 'bin/fm-test-isolation-proof.sh' "$PROOF_DOC" \ - || fail "proof doc must name the harness owner" - grep -Fq 'production_sharding_enabled' "$PROOF_DOC" \ - || fail "proof doc must record the archived proof-time sharding flag" - grep -Fq 'concurrency' "$PROOF_DOC" \ - || fail "proof doc must record concurrency" - assert_present "$CONTRIB" "CONTRIBUTING.md missing" - grep -Fq 'fm-test-isolation-proof' "$CONTRIB" \ - || fail "CONTRIBUTING must document the isolation-proof entry point" - pass "docs archive the isolation-proof owner and posture" + pass "parallel shards consume the proven-isolated set only" } test_list_candidates_nonempty_and_stable test_candidates_exclude_serial_classes -test_candidates_match_archived_proof test_extra_hermetic_candidates_present test_list_exclusions_documents_reasons test_family_map_labels_this_contract -test_aggregate_failure_under_concurrency -test_phase4_consumes_proven_set_only -test_docs_record_proof_owner +test_parallel_shards_consume_the_proven_set diff --git a/tests/fm-test-run.test.sh b/tests/fm-test-run.test.sh index 76859092c8a..d07bf310c2b 100755 --- a/tests/fm-test-run.test.sh +++ b/tests/fm-test-run.test.sh @@ -11,8 +11,6 @@ set -u . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" RUNNER="$ROOT/bin/fm-test-run.sh" -CI="$ROOT/.github/workflows/ci.yml" -CONTRIB="$ROOT/CONTRIBUTING.md" assert_present "$RUNNER" "bin/fm-test-run.sh is missing" [ -x "$RUNNER" ] || fail "bin/fm-test-run.sh must be executable" @@ -97,7 +95,7 @@ init_changed_fixture_repo() { chmod +x "$repo/bin/fm-test-run.sh" for script in \ fm-brief.test.sh \ - fm-captain-translation-contract.test.sh \ + fm-ask-user-authority.test.sh \ fm-cd-pretool-check.test.sh \ fm-daemon.test.sh \ fm-backend-herdr-smoke.test.sh \ @@ -119,7 +117,6 @@ init_changed_fixture_repo() { : >"$repo/tests/fm-backend-herdr-eventwait.test.py" : >"$repo/bin/fm-supervisor-target-lib.sh" : >"$repo/bin/unmapped-source.sh" - printf '# .agents/skills/example/SKILL.md\n' >>"$repo/tests/fm-captain-translation-contract.test.sh" printf '# .claude/settings.json\n# .pi/extensions/fm-primary-turnend-guard.ts\n' \ >>"$repo/tests/fm-cd-pretool-check.test.sh" printf '# .pi/extensions/fm-primary-pi-watch.ts\n' >>"$repo/tests/fm-pi-watch-extension.test.sh" @@ -167,7 +164,7 @@ test_changed_dependency_selection_and_unmapped_failure() { printf '\n' >>"$repo/.pi/extensions/fm-primary-pi-watch.ts" printf '\n' >>"$repo/.pi/extensions/fm-primary-turnend-guard.ts" listed=$(cd "$repo" && bin/fm-test-run.sh --list --changed --base HEAD) - assert_contains "$listed" "tests/fm-captain-translation-contract.test.sh" "skill source selects contract coverage" + assert_contains "$listed" "tests/fm-ask-user-authority.test.sh" "skill source selects pure contract coverage" assert_contains "$listed" "tests/fm-cd-pretool-check.test.sh" "Claude and Pi source selects hook coverage" assert_contains "$listed" "tests/fm-pi-watch-extension.test.sh" "Pi source selects watcher coverage" git -C "$repo" add .agents .claude .pi @@ -353,84 +350,6 @@ test_exclude_family() { pass "exclude-family drops the named primary family after selection" } -test_ci_and_docs_call_the_owner() { - assert_present "$CI" "ci.yml missing" - assert_present "$CONTRIB" "CONTRIBUTING.md missing" - grep -Fq 'tests-portable-parallel-1:' "$CI" \ - || fail "CI must define portable parallel shard 1" - grep -Fq 'tests-portable-parallel-2:' "$CI" \ - || fail "CI must define portable parallel shard 2" - grep -Fq 'tests-portable-serial:' "$CI" \ - || fail "CI must define the portable serial lane" - grep -Fq 'bin/fm-test-run.sh --lane portable-parallel-1' "$CI" \ - || fail "CI shard 1 must invoke --lane portable-parallel-1" - grep -Fq 'bin/fm-test-run.sh --lane portable-parallel-2' "$CI" \ - || fail "CI shard 2 must invoke --lane portable-parallel-2" - local shard job_body - for shard in 1 2; do - job_body=$(awk -v job=" tests-portable-parallel-$shard:" ' - $0 == job { in_job=1; next } - in_job && /^ [a-zA-Z0-9_-]+:/ { exit } - in_job { print } - ' "$CI") - printf '%s\n' "$job_body" | grep -Fq 'npm install -g tasks-axi' \ - || fail "CI portable parallel shard $shard must install tasks-axi" - printf '%s\n' "$job_body" | grep -Fq 'tasks-axi --version' \ - || fail "CI portable parallel shard $shard must verify tasks-axi" - done - grep -Fq 'bin/fm-test-run.sh --lane portable-serial' "$CI" \ - || fail "CI portable serial must invoke --lane portable-serial" - grep -Fq 'bin/fm-test-run.sh --check-coverage' "$CI" \ - || fail "CI must run the coverage guard" - grep -Fq 'tests-herdr:' "$CI" \ - || fail "CI must define the required tests-herdr job" - grep -Fq 'bin/fm-test-run.sh --family real-herdr-gated' "$CI" \ - || fail "Herdr CI job must run the real-herdr-gated family via fm-test-run" - grep -Fq -- "--fail-on-gate-skip 'herdr not found'" "$CI" \ - || fail "Herdr CI job must fail on herdr-not-found skips" - grep -Fq 'bin/fm-install-herdr.sh' "$CI" \ - || fail "Herdr CI job must install via bin/fm-install-herdr.sh" - grep -Fq 'bin/fm-install-treehouse.sh' "$CI" \ - || fail "Herdr CI job must install via bin/fm-install-treehouse.sh" - grep -Fq 'bin/fm-herdr-ci-cleanup.sh' "$CI" \ - || fail "Herdr CI job must use bounded lab cleanup" - grep -Fq 'tests-timing-aggregate:' "$CI" \ - || fail "CI must aggregate per-lane timing artifacts" - grep -Fq 'timeout-minutes: 20' "$CI" \ - || fail "portable serial hang tripwire must be timeout-minutes: 20" - grep -Fq 'timeout-minutes: 10' "$CI" \ - || fail "portable parallel shards must keep a hang tripwire (10m)" - # Interim full-suite 25m portable timeout must not remain after sharding. - if grep -Eq 'timeout-minutes: 25' "$CI"; then - fail "CI still has interim timeout-minutes: 25 after portable sharding" - fi - # Stale "~2-3 minutes" claim must not remain. - if grep -Eq '2-3 minutes' "$CI"; then - fail "CI workflow still claims the suite finishes in ~2-3 minutes" - fi - # No retry-green strategy on Behavior lanes. - if grep -Eqi 'retry:|max-attempts:|continue-on-error:\s*true' "$CI"; then - fail "CI must not use retries or continue-on-error as a green strategy" - fi - grep -Fq 'fm-test-timing' "$CI" \ - || fail "CI must upload timing artifacts" - grep -Fq 'bin/fm-test-run.sh --all' "$CONTRIB" \ - || fail "CONTRIBUTING must document bin/fm-test-run.sh --all" - grep -Fq 'bin/fm-test-run.sh --family' "$CONTRIB" \ - || fail "CONTRIBUTING must document family selection" - grep -Fq 'bin/fm-test-run.sh --changed' "$CONTRIB" \ - || fail "CONTRIBUTING must document changed-file selection" - grep -Fq 'bin/fm-test-run.sh --proven-isolated --jobs' "$CONTRIB" \ - || fail "CONTRIBUTING must document proven-isolated --jobs" - grep -Fq 'intent-targeted' "$CONTRIB" \ - || fail "CONTRIBUTING must document intent-targeted no-mistakes Test" - # Do not restore a complete-suite commands.test. - if grep -E '^[[:space:]]*test:[[:space:]].*tests/\*\.test\.sh' "$ROOT/.no-mistakes.yaml" >/dev/null 2>&1; then - fail ".no-mistakes.yaml must not set a full-suite commands.test" - fi - pass "CI and CONTRIBUTING call the one-owner runner; no full-suite local Test" -} - test_portable_shard_union_and_coverage_guard() { local s1 s2 proven serial herdr all_count union_count overlap out first s1=$("$RUNNER" --list --lane portable-parallel-1) @@ -462,8 +381,8 @@ test_portable_shard_union_and_coverage_guard() { || fail "lanes must not duplicate scripts" # LPT order: first script of shard 1 is the longest proven script. first=$(printf '%s\n' "$s1" | head -n 1) - [ "$first" = "tests/fm-arm-pretool-check.test.sh" ] \ - || fail "shard 1 must start with longest proven script, got $first" + [ "$first" = "tests/fm-x-mode.test.sh" ] \ + || fail "shard 1 must start with the longest proven script, got $first" pass "portable shard union, disjointness, and coverage guard hold" } @@ -493,8 +412,8 @@ test_jobs_parallel_scheduler_and_failure_propagation() { runner="$repo/bin/fm-test-run.sh" evidence="$tmp/evidence" fake_bin="$tmp/fake-bin" - a=tests/fm-no-mistakes-ownership.test.sh - b=tests/fm-stow-contract.test.sh + a=tests/fm-brief.test.sh + b=tests/fm-composer-lib.test.sh c=tests/fm-lint.test.sh d=tests/fm-supervision-instructions.test.sh mkdir -p "$repo/bin" "$repo/tests" "$evidence" "$fake_bin" @@ -683,7 +602,6 @@ test_aggregate_exit_behavior test_gate_skip_accounting test_fail_on_gate_skip_token test_exclude_family -test_ci_and_docs_call_the_owner test_portable_shard_union_and_coverage_guard test_jobs_requires_proven_isolated test_jobs_parallel_scheduler_and_failure_propagation diff --git a/tests/fm-tmux-submit-busy.test.sh b/tests/fm-tmux-submit-busy.test.sh index 76233aaa250..f3eb49a7eb7 100755 --- a/tests/fm-tmux-submit-busy.test.sh +++ b/tests/fm-tmux-submit-busy.test.sh @@ -27,7 +27,7 @@ COMPOSER="${FM_FAKE_COMPOSER:?}" case "${1:-}" in display-message) for a in "$@"; do - case "$a" in *cursor_y*) printf '0\n'; exit 0 ;; esac + case "$a" in *cursor_y*) printf '1\n'; exit 0 ;; esac done exit 0 ;; capture-pane) cat "$COMPOSER" 2>/dev/null; exit 0 ;; @@ -37,10 +37,11 @@ case "${1:-}" in case "$1" in -t) shift ;; -l) ;; Enter) is_enter=1 ;; esac; shift done if [ "$is_enter" = 1 ]; then + [ -z "${FM_FAKE_SENT:-}" ] || printf 'Enter\n' >> "$FM_FAKE_SENT" if [ -n "${FM_FAKE_SWALLOW:-}" ] && [ -f "$FM_FAKE_SWALLOW" ]; then [ "${FM_FAKE_PERSIST_SWALLOW:-0}" = 1 ] || rm -f "$FM_FAKE_SWALLOW" else - printf '│ > │\n' > "$COMPOSER" + printf '╭─────╮\n│ > │\n╰─────╯\n' > "$COMPOSER" fi fi exit 0 ;; @@ -59,7 +60,7 @@ test_busy_pane_pending_returns_empty() { composer="$dir/composer" sent="$dir/sent.log" vfile="$dir/verdict" - printf '│ > fix findings 1 and 3 │\n' > "$composer" + printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$composer" : > "$sent" touch "$dir/.swallow" # Pre-check: composer state should be pending (via function, not $()). @@ -70,8 +71,8 @@ test_busy_pane_pending_returns_empty() { FM_FAKE_SWALLOW="$dir/.swallow" FM_FAKE_PERSIST_SWALLOW=1 FM_FAKE_PANE_BUSY=1 \ fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null [ "$(cat "$vfile")" = empty ] || fail "busy-pane pending should return empty, got '$(cat "$vfile")'" - [ "$(grep -c 'fix findings' "$sent" 2>/dev/null || true)" -eq 0 ] \ - || fail "busy-pane should not retype text" + [ "$(grep -c '^Enter$' "$sent" 2>/dev/null || true)" -eq 3 ] \ + || fail "proven pending should consume the configured Enter retry budget" pass "fm_tmux_submit_enter_core: busy pane + pending composer returns empty (message queued)" } @@ -82,7 +83,7 @@ test_idle_pane_pending_returns_pending() { composer="$dir/composer" sent="$dir/sent.log" vfile="$dir/verdict" - printf '│ > fix findings 1 and 3 │\n' > "$composer" + printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$composer" : > "$sent" touch "$dir/.swallow" PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" FM_FAKE_SENT="$sent" \ @@ -99,7 +100,7 @@ test_busy_pane_composer_clears_first_try() { composer="$dir/composer" sent="$dir/sent.log" vfile="$dir/verdict" - printf '│ > fix findings 1 and 3 │\n' > "$composer" + printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$composer" : > "$sent" PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" FM_FAKE_SENT="$sent" FM_FAKE_PANE_BUSY=1 \ fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null @@ -114,7 +115,7 @@ test_idle_pane_composer_clears_first_try() { composer="$dir/composer" sent="$dir/sent.log" vfile="$dir/verdict" - printf '│ > fix findings 1 and 3 │\n' > "$composer" + printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$composer" : > "$sent" PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" FM_FAKE_SENT="$sent" FM_FAKE_PANE_BUSY=0 \ fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null @@ -122,7 +123,145 @@ test_idle_pane_composer_clears_first_try() { pass "fm_tmux_submit_enter_core: idle pane clears composer on first Enter - returns empty as before" } +test_busy_pane_unknown_stays_unknown() { + local dir fakebin composer vfile + dir="$TMP_ROOT/busy-unknown" + fakebin=$(make_submit_mock "$dir") + composer="$dir/composer" + vfile="$dir/verdict" + printf '│ > unbounded\n' > "$composer" + touch "$dir/.swallow" + PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" FM_FAKE_PANE_BUSY=1 \ + FM_FAKE_SWALLOW="$dir/.swallow" FM_FAKE_PERSIST_SWALLOW=1 \ + fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null + [ "$(cat "$vfile")" = unknown ] \ + || fail "a busy pane must not convert an unsafe composer to empty, got '$(cat "$vfile")'" + pass "fm_tmux_submit_enter_core: busy conversion is limited to proven pending input" +} + +test_busy_pane_ambiguous_pending_retries_without_conversion() { + local dir fakebin composer sent vfile + dir="$TMP_ROOT/busy-ambiguous-pending" + fakebin=$(make_submit_mock "$dir") + composer="$dir/composer" + sent="$dir/sent.log" + vfile="$dir/verdict" + : > "$sent" + printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$composer" + touch "$dir/.swallow" + PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" fm_tmux_composer_state "win" > "$vfile" 2>/dev/null + [ "$(cat "$vfile")" = pending-unproven ] \ + || fail "ambiguous composer text should be pending-unproven, got '$(cat "$vfile")'" + PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" FM_FAKE_SENT="$sent" FM_FAKE_PANE_BUSY=1 \ + FM_FAKE_SWALLOW="$dir/.swallow" FM_FAKE_PERSIST_SWALLOW=1 \ + fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null + [ "$(cat "$vfile")" = pending-unproven ] \ + || fail "a busy pane must not convert pending-unproven to empty, got '$(cat "$vfile")'" + [ "$(grep -c '^Enter$' "$sent" 2>/dev/null || true)" -eq 3 ] \ + || fail "pending-unproven should consume the configured Enter retry budget" + pass "fm_tmux_submit_enter_core: pending-unproven retries without busy conversion" +} + +test_unrecognized_state_skips_busy_conversion() { + local dir fakebin composer busy_called vfile + dir="$TMP_ROOT/unrecognized-state" + fakebin=$(make_submit_mock "$dir") + composer="$dir/composer" + busy_called="$dir/busy-called" + vfile="$dir/verdict" + printf '╭─────╮\n│ > │\n╰─────╯\n' > "$composer" + ( + # shellcheck disable=SC2329 + fm_tmux_composer_state() { printf 'future-state'; } + # shellcheck disable=SC2329 + fm_pane_is_busy() { touch "$busy_called"; return 0; } + PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" \ + fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null + ) || fail "unrecognized-state submit check failed" + [ "$(cat "$vfile")" = future-state ] \ + || fail "unrecognized state should be preserved, got '$(cat "$vfile")'" + [ ! -e "$busy_called" ] \ + || fail "unrecognized state must not trigger busy conversion" + pass "fm_tmux_submit_enter_core: unrecognized states skip busy conversion" +} + +test_claude_busy_signature_uses_real_capture_shapes() { + local dir fakebin composer + dir="$TMP_ROOT/claude-signature" + fakebin=$(make_submit_mock "$dir") + composer="$dir/composer" + pane_busy() { + PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" \ + bash -c '. "$1/bin/fm-tmux-lib.sh"; fm_pane_is_busy "$2" "$3"' \ + _ "$ROOT" "$1" "${2:-}" + } + + # Live Claude 2.1.220 capture 1: spinner glyph and word from one turn. + printf '✢ Pollinating… (16s · ↓ 1.1k tokens · thought for 1s)\n' > "$composer" + pane_busy live claude || fail "Claude capture 1 should be busy" + + # Live Claude 2.1.220 capture 2: a later turn with a changed glyph and word. + printf '✽ Proofing… (5s · thinking with high effort)\n' > "$composer" + pane_busy live claude || fail "Claude capture 2 should be busy" + + # Real idle Claude capture shape from the verified pane sample. + printf '✻ Worked for 31s\n' > "$composer" + pane_busy idle claude && fail "Claude Worked-for capture must be idle" + + # The new signature is Claude-scoped and must not widen the shared default. + printf '✢ Pollinating… (16s · ↓ 1.1k tokens)\n' > "$composer" + pane_busy live && fail "Claude signature must not match without the Claude harness" + + # Each verified harness must use only its own signature. + printf 'Ctrl+c:cancel\n' > "$composer" + pane_busy cross claude && fail "Claude must ignore Grok's cancel footer" + printf 'esc interrupt\n' > "$composer" + pane_busy cross claude && fail "Claude must ignore OpenCode's interrupt footer" + printf 'Working...\n' > "$composer" + pane_busy cross codex && fail "Codex must ignore Pi's Working footer" + printf 'esc interrupt\n' > "$composer" + pane_busy cross codex && fail "Codex must ignore OpenCode's interrupt footer" + printf 'Ctrl+c:cancel\n' > "$composer" + pane_busy cross opencode && fail "OpenCode must ignore Grok's cancel footer" + printf 'esc interrupt\n' > "$composer" + pane_busy cross pi && fail "Pi must ignore OpenCode's interrupt footer" + printf 'esc to interrupt\n' > "$composer" + pane_busy cross grok && fail "Grok must ignore Claude's legacy interrupt footer" + printf 'esc to interrupt\n' > "$composer" + pane_busy own codex || fail "Codex's escape footer should be busy" + printf 'esc interrupt\n' > "$composer" + pane_busy own opencode || fail "OpenCode's interrupt footer should be busy" + + # No harness keeps the historical combined-pattern compatibility fallback. + printf 'Working...\n' > "$composer" + pane_busy fallback || fail "no-harness fallback should retain Pi's shared signature" + printf 'Ctrl+c:cancel\n' > "$composer" + pane_busy fallback || fail "no-harness fallback should retain Grok's shared signature" + + # A supplied harness must never use another harness's signature. This is + # particularly important for Kimi: its idle key-tip rotation can include the + # same cancel token Grok uses to mean busy. + printf 'Working...\n' > "$composer" + pane_busy unknown kimi && fail "Kimi must ignore Pi's Working footer" + printf 'Ctrl+c:cancel\n' > "$composer" + pane_busy unknown kimi && fail "idle Kimi must ignore Grok's cancel footer" + + # Older Claude Code and the existing Pi and Grok signatures remain unchanged. + printf 'esc to interrupt\n' > "$composer" + pane_busy old-claude claude || fail "older Claude escape footer should be busy" + printf 'Working...\n' > "$composer" + pane_busy pi pi || fail "Pi Working footer should be busy" + pane_busy pi-signed pi-signed || fail "pi-signed should share Pi's exact Working footer" + printf 'Ctrl+c:cancel\n' > "$composer" + pane_busy grok grok || fail "Grok cancel footer should be busy" + pass "fm_pane_is_busy: Claude spinner is scoped, multi-frame, and backward-compatible" +} + test_busy_pane_pending_returns_empty test_idle_pane_pending_returns_pending test_busy_pane_composer_clears_first_try test_idle_pane_composer_clears_first_try +test_busy_pane_unknown_stays_unknown +test_busy_pane_ambiguous_pending_retries_without_conversion +test_unrecognized_state_skips_busy_conversion +test_claude_busy_signature_uses_real_capture_shapes diff --git a/tests/fm-turnend-guard.test.sh b/tests/fm-turnend-guard.test.sh index 3a15a7c21da..242407c1a32 100755 --- a/tests/fm-turnend-guard.test.sh +++ b/tests/fm-turnend-guard.test.sh @@ -77,6 +77,19 @@ test_predicate_queue_pending_flag() { pass "fm_supervision_status: FM_SUP_QUEUE_PENDING tracks state/.wake-queue" } +test_predicate_x_mode_needs_supervision() { + local state="$TMP_ROOT/pred-x-mode/state" + mkdir -p "$state" + : > "$state/x-watch.check.sh" + fm_supervision_needed "$state" 300 || fail "X-mode relay poll did not register as supervision need" + [ "$FM_SUP_IN_FLIGHT" -eq 0 ] || fail "X-mode relay poll must not count as an in-flight task" + [ "$FM_SUP_NEEDED" = true ] || fail "X-mode relay poll must set FM_SUP_NEEDED" + if fm_supervision_unhealthy "$state" 300; then + fail "task-specific unhealthy predicate must preserve its zero-task behavior" + fi + pass "fm_supervision_needed: X-mode relay poll needs supervision without changing the task predicate" +} + # --- HOOK: bin/fm-turnend-guard.sh ------------------------------------------ # # Each scenario gets its own directory carrying a copy of the two guard scripts @@ -591,36 +604,106 @@ EOF expect_code 0 "$status" "grok adapter must allow its own forced resume turn to end" [ -z "$out" ] || fail "grok adapter printed output while loop-guarded: $out" [ ! -e "$log" ] || fail "grok adapter spawned another resume while loop-guarded: $(cat "$log")" - pass "fm-turnend-guard-grok: loop guard prevents a nested resume loop" + pass "fm-turnend-guard-grok: legacy environment loop guard prevents a nested resume loop" } -test_settings_hook_uses_claude_project_dir() { - local settings command - settings="$ROOT/.claude/settings.json" - [ -f "$settings" ] || fail "tracked .claude/settings.json is missing" - command=$(jq -r '.hooks.Stop[0].hooks[0].command // empty' "$settings") - [ -n "$command" ] || fail "Stop hook command is missing from .claude/settings.json" - assert_contains "$command" 'CLAUDE_PROJECT_DIR' "Stop hook must resolve via CLAUDE_PROJECT_DIR, not a cwd-relative path" - assert_contains "$command" 'fm-turnend-guard.sh' "Stop hook must still invoke fm-turnend-guard.sh" - case "$command" in - bin/fm-turnend-guard.sh|./bin/fm-turnend-guard.sh) - fail "Stop hook must not use a bare relative path (cwd-dependent): $command" - ;; - esac - pass ".claude/settings.json: Stop hook uses CLAUDE_PROJECT_DIR-anchored command" -} - -test_codex_hook_invokes_shared_guard() { - local settings command - settings="$ROOT/.codex/hooks.json" - [ -f "$settings" ] || fail "tracked .codex/hooks.json is missing" - command=$(jq -r '.hooks.Stop[0].hooks[0].command // empty' "$settings") - [ -n "$command" ] || fail "Stop hook command is missing from .codex/hooks.json" - assert_contains "$command" 'pwd -P' "codex hook must anchor from the hook process working directory" - assert_contains "$command" '.codex/hooks.json' "codex hook must verify the hook-loaded firstmate root" - assert_contains "$command" 'fm-turnend-guard.sh' "codex hook must invoke the shared guard" - assert_not_contains "$command" '.cwd' "codex hook must not use payload cwd to select the guard executable" - pass ".codex/hooks.json: Stop hook invokes the shared primary guard" +test_grok_adapter_native_false_blocks_without_resume() { + local dir fakebin log out status + dir=$(make_primary_dir "$TMP_ROOT/grok-native-false") + : > "$dir/state/task1.meta" + fakebin=$(fm_fakebin "$TMP_ROOT/grok-native-false-bin") + log="$TMP_ROOT/grok-native-false.log" + printf '#!/usr/bin/env bash\nprintf called >> %q\n' "$log" > "$fakebin/grok" + chmod +x "$fakebin/grok" + out=$(printf '%s' '{"sessionId":"native","stopHookActive":false}' | PATH="$fakebin:$PATH" GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 2 "$status" "native stopHookActive=false must return the shared blocking status" + assert_contains "$out" 'TURN WOULD END BLIND' "native block must pass shared guard feedback to Grok" + [ ! -e "$log" ] || fail "native path started grok --resume" + pass "fm-turnend-guard-grok: native false delegates blocking feedback with zero resume processes" +} + +test_grok_adapter_native_true_allows_without_resume() { + local dir fakebin log out status + dir=$(make_primary_dir "$TMP_ROOT/grok-native-true") + : > "$dir/state/task1.meta" + fakebin=$(fm_fakebin "$TMP_ROOT/grok-native-true-bin") + log="$TMP_ROOT/grok-native-true.log" + printf '#!/usr/bin/env bash\nprintf called >> %q\n' "$log" > "$fakebin/grok" + chmod +x "$fakebin/grok" + out=$(printf '%s' '{"sessionId":"native","stopHookActive":true}' | PATH="$fakebin:$PATH" GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 0 "$status" "native stopHookActive=true must allow the bounded continuation to stop" + [ -z "$out" ] || fail "native true produced output: $out" + [ ! -e "$log" ] || fail "native true started grok --resume" + pass "fm-turnend-guard-grok: native true remains bounded and starts no resume process" +} + +test_grok_adapter_snake_case_native_and_camel_precedence() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/grok-native-spellings") + : > "$dir/state/task1.meta" + out=$(printf '%s' '{"sessionId":"native","stop_hook_active":false}' | GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 2 "$status" "typed snake_case false must select native blocking" + assert_contains "$out" 'TURN WOULD END BLIND' "snake_case native block lost feedback" + out=$(printf '%s' '{"sessionId":"native","stopHookActive":true,"stop_hook_active":false}' | GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 0 "$status" "camelCase true must win over snake_case false" + out=$(printf '%s' '{"sessionId":"native","stopHookActive":false,"stop_hook_active":true}' | GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 2 "$status" "camelCase false must win over snake_case true" + pass "fm-turnend-guard-grok: both spellings are typed and camelCase has deterministic precedence" +} + +test_grok_adapter_invalid_inputs_start_neither_path() { + local dir fakebin log payload out status + dir=$(make_primary_dir "$TMP_ROOT/grok-invalid-inputs") + : > "$dir/state/task1.meta" + fakebin=$(fm_fakebin "$TMP_ROOT/grok-invalid-bin") + log="$TMP_ROOT/grok-invalid.log" + printf '#!/usr/bin/env bash\nprintf called >> %q\n' "$log" > "$fakebin/grok" + chmod +x "$fakebin/grok" + for payload in \ + ' ' \ + '{' \ + '{"sessionId":"x","stopHookActive":"false"}' \ + '{"sessionId":"x","stop_hook_active":1}' \ + '{"sessionId":"x"}{"sessionId":"y"}' \ + '{"sessionId":"x","stopHookActive":false}{"sessionId":"y","stopHookActive":false}' \ + '{"sessionId":"x","stopHookActive":"bad","stopHookActive":false}' \ + '{"sessionId":"x","stop_hook_active":false,"stop_hook_active":false}' \ + '{"sessionId":"x","sessionId":"y"}' + do + out=$(printf '%s' "$payload" | PATH="$fakebin:$PATH" GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 0 "$status" "invalid Grok payload must conservatively allow without choosing a path" + [ -z "$out" ] || fail "invalid Grok payload produced output: $out" + done + [ ! -e "$log" ] || fail "invalid Grok payload started a resume process" + out=$(printf '%s' '{"sessionId":"x","stopHookActive":false}' | PATH="$fakebin:$PATH" GROK_WORKSPACE_ROOT="$TMP_ROOT/missing-grok-root" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 0 "$status" "missing shared-guard prerequisite must conservatively allow" + [ -z "$out" ] || fail "missing prerequisite produced output: $out" + [ ! -e "$log" ] || fail "missing prerequisite started a resume process" + pass "fm-turnend-guard-grok: malformed, invalidly typed, and missing-prerequisite payloads start neither path" +} + +test_grok_adapter_missing_jq_and_no_supervision_allow() { + local dir fakebin log out status tool tool_path + dir=$(make_primary_dir "$TMP_ROOT/grok-nojq") + : > "$dir/state/task1.meta" + fakebin=$(fm_fakebin "$TMP_ROOT/grok-nojq-bin") + log="$TMP_ROOT/grok-nojq.log" + for tool in bash cat printf; do + tool_path=$(command -v "$tool") || fail "test host must provide $tool" + ln -s "$tool_path" "$fakebin/$tool" + done + printf '#!/usr/bin/env bash\nprintf called >> %q\n' "$log" > "$fakebin/grok" + chmod +x "$fakebin/grok" + out=$(printf '%s' '{"sessionId":"x","stopHookActive":false}' | PATH="$fakebin" GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 0 "$status" "missing jq must conservatively allow" + [ -z "$out" ] || fail "missing jq produced output: $out" + [ ! -e "$log" ] || fail "missing jq started a resume process" + + dir=$(make_primary_dir "$TMP_ROOT/grok-native-no-work") + out=$(printf '%s' '{"sessionId":"x","stopHookActive":false}' | GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 0 "$status" "healthy no-supervision-needed native stop must allow" + [ -z "$out" ] || fail "no-supervision-needed native stop produced output: $out" + pass "fm-turnend-guard-grok: missing jq and no-supervision-needed stops stay silent and bounded" } test_codex_hook_uses_process_pwd_when_payload_cwd_is_outside_root() { @@ -686,23 +769,6 @@ EOF pass ".codex/hooks.json: Stop hook ignores nested git root guard scripts" } -test_opencode_plugin_forces_followup() { - local plugin content - plugin="$ROOT/.opencode/plugins/fm-primary-turnend-guard.js" - [ -f "$plugin" ] || fail "tracked OpenCode primary plugin is missing" - content=$(cat "$plugin") - assert_contains "$content" 'session.idle' "OpenCode plugin must run on session.idle" - assert_contains "$content" 'fm-turnend-guard.sh' "OpenCode plugin must invoke the shared guard" - assert_contains "$content" 'promptAsync' "OpenCode plugin must force a follow-up turn" - assert_contains "$content" 'encodeFirstmateOperationalInput' "OpenCode plugin must use the typed operational-input constructor" - assert_contains "$content" 'skipNextIdle' "OpenCode plugin must carry a loop guard" - assert_contains "$content" 'worktree' "OpenCode plugin must anchor the guard from the git worktree path" - assert_contains "$content" 'watcher cycle is missing, failed, or unhealthy' "OpenCode plugin must identify a blind turn as watcher recovery" - assert_contains "$content" 'harness recovery instruction below' "OpenCode plugin must delegate recovery action to the shared guard line" - assert_not_contains "$content" 'Resume supervision according to the session-start operating block' "OpenCode plugin must not route a blind turn through ordinary continuity" - pass ".opencode primary plugin: session.idle forces one follow-up through the shared guard" -} - test_opencode_plugin_anchors_guard_to_worktree() { local plugin parent worktree_dir wrong_dir out status plugin="$ROOT/.opencode/plugins/fm-primary-turnend-guard.js" @@ -762,30 +828,6 @@ EOF pass ".opencode primary plugin: guard path is anchored to worktree, not directory" } -test_pi_extension_forces_followup() { - local ext content - ext="$ROOT/.pi/extensions/fm-primary-turnend-guard.ts" - [ -f "$ext" ] || fail "tracked pi primary extension is missing" - content=$(cat "$ext") - assert_contains "$content" 'agent_settled' "pi extension must run after one logical agent run settles" - assert_contains "$content" 'fm-turnend-guard.sh' "pi extension must invoke the shared guard" - assert_contains "$content" 'sendUserMessage' "pi extension must force a follow-up turn" - assert_contains "$content" 'encodeFirstmateOperationalInput' "pi extension must use the typed operational-input constructor" - assert_contains "$content" 'deliverAs: "followUp"' "pi extension must queue the follow-up safely" - assert_contains "$content" 'guardFollowupActive' "pi extension must carry a logical-run loop guard" - assert_not_contains "$content" 'skipNextTurnEnd' "pi extension kept the internal-turn loop guard" - assert_contains "$content" 'watcher cycle is missing, failed, or unhealthy' "pi extension must identify a blind turn as watcher recovery" - assert_contains "$content" 'harness recovery instruction below' "pi extension must delegate recovery action to the shared guard line" - assert_not_contains "$content" 'Resume supervision according to the session-start operating block' "pi extension must not route a blind turn through ordinary continuity" - assert_contains "$content" '.pi-turnend-extension-loaded' "pi extension must write its loaded marker for session-start diagnostics" - assert_contains "$content" 'lockOwnership' "pi extension loaded marker must respect the session lock" - assert_contains "$content" 'const command = String((event.input as { command?: unknown })?.command ?? "")' "pi extension changed bash command extraction for the PreToolUse contract" - assert_contains "$content" 'runPretoolCheck(command)' "pi extension changed the PreToolUse checker invocation" - assert_contains "$content" 'return { block: true, reason:' "pi extension changed the checker exit-2 block result" - assert_not_contains "$content" 'Run bin/fm-watch-arm.sh as a background task' "pi extension must not hardcode the old watcher-arm instruction" - pass ".pi primary extension: agent_settled forces one follow-up through the shared guard" -} - test_pi_extension_injects_once_per_logical_agent_run() { local repo home ext log out status repo="$TMP_ROOT/pi-logical-run-root" @@ -902,15 +944,164 @@ EOF pass ".pi primary extension: delivery failure resets the logical-run latch" } -test_grok_hook_invokes_adapter() { - local settings command - settings="$ROOT/.grok/hooks/fm-primary-turnend-guard.json" - [ -f "$settings" ] || fail "tracked grok primary hook config is missing" - command=$(jq -r '.hooks.Stop[0].hooks[0].command // empty' "$settings") - [ -n "$command" ] || fail "Stop hook command is missing from grok primary hook config" - assert_contains "$command" 'GROK_WORKSPACE_ROOT' "grok hook must anchor from GROK_WORKSPACE_ROOT" - assert_contains "$command" 'fm-turnend-guard-grok.sh' "grok hook must invoke the adapter" - pass ".grok primary hook: Stop hook invokes the grok adapter" +# --- --claude cooperative mode ----------------------------------------------- +# In --claude mode the guard ignores stop_hook_active (Claude marks every stop +# after ANY stop-hook continuation true, including asyncRewake rewake turns) and +# cooperates with the Stop-owned auto-arm instead: allow on health, live owner +# claim, or a fresh rewake epoch; bounded re-block only when none materialize. + +run_hook_claude() { + local dir=$1 stop_active=$2 home + home=$(cd "$dir" && pwd) + printf '{"stop_hook_active":%s,"session_id":"sess-claude-mode"}' "$stop_active" | CLAUDECODE=1 FM_HOME="$home" bash "$dir/bin/fm-turnend-guard.sh" --claude 2>&1 +} + +# The 2026-07-21 incident regression: after a spent forced continuation the old +# one-shot loop guard ALLOWED a blind stop (stop_hook_active=true) while the +# watcher was already dead. In --claude mode the guard must re-block instead. +test_hook_claude_mode_reblocks_stop_hook_active_when_unhealthy() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/hook-claude-reblock") + : > "$dir/state/task1.meta" + out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=200 run_hook_claude "$dir" true); status=$? + expect_code 2 "$status" "--claude mode must re-block a stop_hook_active=true stop while unhealthy with no auto-arm claim" + assert_contains "$out" "TURN WOULD END BLIND" "--claude re-block must carry the blind-turn banner" + assert_contains "$out" "Stop-owned auto-arm did not claim" "--claude re-block must explain the missing auto-arm claim" + pass "fm-turnend-guard --claude: re-blocks a loop-guarded stop while unhealthy and unclaimed (incident regression)" +} + +test_hook_claude_mode_reblocks_x_mode_without_tasks() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/hook-claude-x-mode") + : > "$dir/state/x-watch.check.sh" + out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=200 run_hook_claude "$dir" true); status=$? + expect_code 2 "$status" "--claude mode must re-block an X-mode-only stop when no auto-arm claims recovery" + assert_contains "$out" "X-mode relay polling needs supervision" "--claude X-mode re-block must name the active supervision need" + [ -f "$dir/state/.turnend-claude-blocks" ] || fail "--claude X-mode re-block must consume the shared block budget" + pass "fm-turnend-guard --claude: X-mode-only homes re-block when auto-arm recovery is absent" +} + +test_hook_claude_mode_allows_when_autoarm_owner_alive() { + local dir pid out status + dir=$(make_primary_dir "$TMP_ROOT/hook-claude-owner") + : > "$dir/state/task1.meta" + sleep 60 & + pid=$! + mkdir -p "$dir/state/.claude-autoarm.lock" + printf '%s\n' "$pid" > "$dir/state/.claude-autoarm.lock/pid" + out=$(run_hook_claude "$dir" false); status=$? + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + expect_code 0 "$status" "--claude mode must allow when the auto-arm owner process is alive" + [ -z "$out" ] || fail "--claude owner-claimed allow produced output: $out" + pass "fm-turnend-guard --claude: allows the stop when the Stop auto-arm owner holds this home" +} + +test_hook_claude_mode_allows_on_fresh_rewake_epoch() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/hook-claude-epoch") + : > "$dir/state/task1.meta" + printf 'epoch=3 owner_pid=999 outcome=rewake updated_at=%s\n' "$(date +%s)" > "$dir/state/.claude-autoarm-epoch" + out=$(run_hook_claude "$dir" true); status=$? + expect_code 0 "$status" "--claude mode must allow the stop whose rewake the auto-arm already owns" + [ -z "$out" ] || fail "--claude rewake-epoch allow produced output: $out" + pass "fm-turnend-guard --claude: fresh rewake epoch prevents a duplicate continuation for the same event" +} + +test_hook_claude_mode_stale_rewake_epoch_blocks() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/hook-claude-stale-epoch") + : > "$dir/state/task1.meta" + printf 'epoch=3 owner_pid=999 outcome=rewake updated_at=1\n' > "$dir/state/.claude-autoarm-epoch" + touch -t 202001010000 "$dir/state/.claude-autoarm-epoch" + out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=200 run_hook_claude "$dir" true); status=$? + expect_code 2 "$status" "--claude mode must not treat an ancient rewake epoch as this event's recovery" + pass "fm-turnend-guard --claude: stale rewake epoch does not allow a blind stop" +} + +test_hook_claude_mode_block_budget_then_degraded_allow() { + local dir out status i + dir=$(make_primary_dir "$TMP_ROOT/hook-claude-budget") + : > "$dir/state/task1.meta" + for i in 1 2 3; do + out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$dir" false); status=$? + expect_code 2 "$status" "--claude block $i must exit 2 within the budget" + done + out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$dir" true); status=$? + expect_code 0 "$status" "--claude must allow degraded once the consecutive-block budget is exhausted" + assert_contains "$out" '"systemMessage"' "--claude degraded allow must surface a visible systemMessage" + assert_contains "$out" 'block budget exhausted' "--claude degraded allow must name the exhausted budget" + out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$dir" false); status=$? + expect_code 2 "$status" "--claude budget must reset after the degraded allow so the next chain re-engages" + pass "fm-turnend-guard --claude: re-block budget stays below the 8-block cap and resets after degraded allow" +} + +test_hook_claude_mode_allow_resets_budget() { + local dir pid identity out status + dir=$(make_primary_dir "$TMP_ROOT/hook-claude-reset") + : > "$dir/state/task1.meta" + out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$dir" false); status=$? + expect_code 2 "$status" "first --claude block must exit 2" + [ -f "$dir/state/.turnend-claude-blocks" ] || fail "--claude block must record the consecutive-block budget" + sleep 60 & + pid=$! + identity=$(watcher_identity "$dir" "$pid") || { + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + fail "could not identify live watcher holder" + } + record_watcher_lock "$dir" "$pid" "$identity" + touch "$dir/state/.last-watcher-beat" + out=$(run_hook_claude "$dir" false); status=$? + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + rm -rf "$dir/state/.watch.lock" + expect_code 0 "$status" "--claude must allow once the watcher is healthy again" + [ ! -f "$dir/state/.turnend-claude-blocks" ] || fail "--claude allow must reset the consecutive-block budget" + out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$dir" false); status=$? + expect_code 2 "$status" "a later unhealthy chain must re-block from a fresh budget" + pass "fm-turnend-guard --claude: any allow resets the consecutive-block budget" +} + +test_hook_claude_mode_waits_for_late_claim() { + local dir helper out status holder + dir=$(make_primary_dir "$TMP_ROOT/hook-claude-wait") + : > "$dir/state/task1.meta" + ( + sleep 0.4 + mkdir -p "$dir/state/.claude-autoarm.lock" + sleep 60 & + printf '%s\n' $! > "$dir/state/.claude-autoarm.lock/pid" + printf '%s\n' $! > "$dir/holder.pid" + wait + ) & + helper=$! + out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=3000 run_hook_claude "$dir" false); status=$? + holder=$(cat "$dir/holder.pid" 2>/dev/null || true) + kill "$holder" 2>/dev/null || true + kill "$helper" 2>/dev/null || true + wait "$helper" 2>/dev/null || true + expect_code 0 "$status" "--claude must wait briefly for a late auto-arm claim instead of forcing a continuation" + [ -z "$out" ] || fail "--claude late-claim wait produced output: $out" + pass "fm-turnend-guard --claude: bounded claim wait avoids a token-consuming forced continuation" +} + +test_hook_claude_mode_secondmate_reblocks_like_primary() { + local dir pid out status + dir=$(make_secondmate_dir "$TMP_ROOT/hook-claude-sm-reblock") + : > "$dir/state/task1.meta" + out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=200 run_hook_claude "$dir" true); status=$? + expect_code 2 "$status" "--claude mode must re-block in a marked secondmate home exactly like the main primary" + assert_contains "$out" "TURN WOULD END BLIND" "--claude secondmate re-block must carry the blind-turn banner" + sleep 60 & + pid=$! + mkdir -p "$dir/state/.claude-autoarm.lock" + printf '%s\n' "$pid" > "$dir/state/.claude-autoarm.lock/pid" + out=$(run_hook_claude "$dir" false); status=$? + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + expect_code 0 "$status" "--claude mode must allow a claimed secondmate home" + pass "fm-turnend-guard --claude: secondmate home re-blocks unclaimed and allows auto-arm-claimed stops" } test_predicate_healthy_no_inflight @@ -918,6 +1109,7 @@ test_predicate_unhealthy_no_beacon test_predicate_unhealthy_stale_beacon test_predicate_healthy_fresh_beacon test_predicate_queue_pending_flag +test_predicate_x_mode_needs_supervision test_hook_silent_when_no_work_in_flight test_hook_blocks_when_fresh_beacon_has_no_live_lock test_hook_blocks_when_dead_lock_has_fresh_beacon @@ -943,13 +1135,22 @@ test_hook_silent_without_stdin test_hook_runs_fast test_grok_adapter_forces_one_resume_when_unhealthy test_grok_adapter_loop_guard_skips_resume -test_settings_hook_uses_claude_project_dir -test_codex_hook_invokes_shared_guard +test_grok_adapter_native_false_blocks_without_resume +test_grok_adapter_native_true_allows_without_resume +test_grok_adapter_snake_case_native_and_camel_precedence +test_grok_adapter_invalid_inputs_start_neither_path +test_grok_adapter_missing_jq_and_no_supervision_allow test_codex_hook_uses_process_pwd_when_payload_cwd_is_outside_root test_codex_hook_ignores_nested_git_root_guard -test_opencode_plugin_forces_followup test_opencode_plugin_anchors_guard_to_worktree -test_pi_extension_forces_followup test_pi_extension_injects_once_per_logical_agent_run test_pi_extension_retries_after_followup_delivery_failure -test_grok_hook_invokes_adapter +test_hook_claude_mode_reblocks_stop_hook_active_when_unhealthy +test_hook_claude_mode_reblocks_x_mode_without_tasks +test_hook_claude_mode_allows_when_autoarm_owner_alive +test_hook_claude_mode_allows_on_fresh_rewake_epoch +test_hook_claude_mode_stale_rewake_epoch_blocks +test_hook_claude_mode_block_budget_then_degraded_allow +test_hook_claude_mode_allow_resets_budget +test_hook_claude_mode_waits_for_late_claim +test_hook_claude_mode_secondmate_reblocks_like_primary diff --git a/tests/fm-watch-triage.test.sh b/tests/fm-watch-triage.test.sh index 19dae9bba36..a14a2923bfd 100755 --- a/tests/fm-watch-triage.test.sh +++ b/tests/fm-watch-triage.test.sh @@ -70,12 +70,38 @@ file_mtime() { if [ "$(uname)" = Darwin ]; then stat -f %m "$1" 2>/dev/null; else stat -c %Y "$1" 2>/dev/null; fi } +# Set <file>'s mtime to exactly <epoch> seconds, for aging a busy-turn marker by +# a precise amount (touch -t takes a local-time stamp, not an epoch, on both +# platforms, so convert via BSD `date -r` or GNU `date -d @`). +set_mtime() { # <epoch> <file> + local epoch=$1 f=$2 stamp + if stamp=$(date -r "$epoch" +%Y%m%d%H%M.%S 2>/dev/null); then + touch -t "$stamp" "$f" + else + stamp=$(date -d "@$epoch" +%Y%m%d%H%M.%S) + touch -t "$stamp" "$f" + fi +} + # Signature a primed .seen-* marker must hold so the per-poll signal scan does not # fire on a pre-existing status (mirrors fm-watch.sh's stat_sig exactly). seen_sig() { if [ "$(uname)" = Darwin ]; then stat -f '%z:%Fm' "$1" 2>/dev/null; else stat -c '%s:%Y' "$1" 2>/dev/null; fi } +# Prime <file>'s .seen-* suppressor to its CURRENT signature, so the per-poll +# no-verb signal scan (which watches every *.turn-ended for a size:mtime change) +# treats a just-created or just-backdated turn-ended marker as already seen. +# Busy-turn-age fixtures create/backdate turn-ended directly (there is no real +# harness touching it), so without this the marker's own first sighting would +# fire an unrelated "signal:" wake and mask the busy-turn-age assertion under +# test. Call again after any further touch/set_mtime on the same file. +prime_turnend_seen() { # <file> + local f=$1 base + base=$(basename "$f" | tr '.' '_') + printf '%s' "$(seen_sig "$f")" > "$(dirname "$f")/.seen-$base" +} + reap() { kill "$1" 2>/dev/null || true; wait "$1" 2>/dev/null || true; } # --- pure classifier predicates (fm-classify-lib.sh) ------------------------ @@ -1055,6 +1081,249 @@ test_wedge_escalation_resets_when_pane_becomes_active() { pass "a pane becoming active again resets the consecutive wedge-escalation counter" } +# --- busy pane duration bound: a completed-turn age gate on top of busy ----- +# 2026-07 hibit-agent-focus-nonsteal-r1 incident: a busy pane (herdr "working" +# and/or the harness's rendered busy footer) is unconditional, unbounded proof +# of liveness in every existing classifier, so a genuinely hung foreground tool +# call behind a busy signature ran undetected for 25h. BUSY_TURN_MAX_SECS bounds +# how long a busy pane may run with no completed turn (state/<id>.turn-ended, or +# the task's spawn record before any turn completes); past the bound the SAME +# wedge_timer_check already used for a provably-working non-busy stale takes +# over, so escalation reuses the identical stale reason, escalation counter, and +# demand-deep-inspection marker - never an automatic interrupt or restart. + +test_busy_pane_below_turn_age_bound_is_absorbed() { + local dir state fakebin out capture_file window key sig pid + dir=$(make_case busy-below-turn-age); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt"; window="test:fm-busy-fresh" + printf 'Working... (12.3s)' > "$capture_file" + printf 'window=%s\nkind=ship\nharness=pi\n' "$window" > "$state/busy-fresh.meta" + printf 'working: setup complete\n' > "$state/busy-fresh.status" + sig=$(seen_sig "$state/busy-fresh.status"); printf '%s' "$sig" > "$state/.seen-busy-fresh_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + touch "$state/busy-fresh.turn-ended" + prime_turnend_seen "$state/busy-fresh.turn-ended" + + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=999 FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + if ! wait_live "$pid" 30; then + reap "$pid"; fail "a busy pane below the turn-age bound was escalated: $(cat "$out")" + fi + [ ! -s "$out" ] || fail "a busy pane below the turn-age bound printed a wake reason" + [ ! -e "$state/.stale-since-$key" ] || fail "a busy pane below the turn-age bound started a wedge timer" + reap "$pid" + pass "a busy worker below the turn-age bound remains working with no escalation" +} + +test_busy_pane_stable_hash_escalates_past_turn_age_bound() { + local dir state fakebin out capture_file window key pane_hash sig pid + dir=$(make_case busy-stable-hash-turn-age); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt"; window="test:fm-busy-stable" + printf 'Working...' > "$capture_file" + printf 'window=%s\nkind=ship\nharness=pi\n' "$window" > "$state/busy-stable.meta" + printf 'working: setup complete\n' > "$state/busy-stable.status" + sig=$(seen_sig "$state/busy-stable.status"); printf '%s' "$sig" > "$state/.seen-busy-stable_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "Working...") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + # No completed turn ever recorded for this task: age the spawn record itself. + touch -t 200001010000 "$state/busy-stable.meta" + + # Phase A: past the bound, the stable-hash busy pane is absorbed but starts + # the wedge timer (mirrors the existing provably-working-stale Phase A/B). + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + if ! wait_live "$pid" 30; then + reap "$pid"; fail "a stable-hash busy pane past the turn-age bound escalated before the wedge threshold: $(cat "$out")" + fi + [ -s "$state/.stale-since-$key" ] || fail "a stable-hash busy pane past the turn-age bound did not start a wedge timer" + reap "$pid" + + # Phase B: backdate the wedge timer past the threshold; the next poll escalates. + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + wait_for_exit "$pid" 40 || fail "a stable-hash busy pane did not wedge-escalate past the turn-age bound" + grep -F "stale: $window" "$out" >/dev/null || fail "busy turn-age escalation did not print the stale wake" + grep -F "possible wedge" "$out" >/dev/null || fail "busy turn-age escalation did not flag a possible wedge" + pass "a busy worker with a stable pane hash still escalates once its completed-turn age reaches the bound" +} + +# Regression fixture for the incident's actual masking condition: Pi's rendered +# elapsed-time footer changes every poll, so the pane hash never repeats and the +# watcher always takes the "new hash" branch, never the stable-hash one above. +test_busy_pane_changing_hash_escalates_past_turn_age_bound() { + local dir state fakebin out capture_file window key pid + dir=$(make_case busy-changing-hash-turn-age); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt"; window="test:fm-busy-ticking" + printf 'Working... (3600.1s)' > "$capture_file" + printf 'window=%s\nkind=ship\nharness=pi\n' "$window" > "$state/busy-ticking.meta" + printf 'working: setup complete\n' > "$state/busy-ticking.status" + sig=$(seen_sig "$state/busy-ticking.status"); printf '%s' "$sig" > "$state/.seen-busy-ticking_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + touch -t 200001010000 "$state/busy-ticking.meta" + # No pre-seeded .hash-<key>: with a real ticking elapsed footer, every poll + # lands here (h != prev) - the reproduction's actual masking condition. + + # Phase A: first sight past the bound absorbs and starts the wedge timer, + # without ever needing the "genuinely stale" hash-match path. + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + if ! wait_live "$pid" 30; then + reap "$pid"; fail "a changing-hash busy pane past the turn-age bound escalated before the wedge threshold: $(cat "$out")" + fi + [ -s "$state/.stale-since-$key" ] || fail "a changing-hash busy pane past the turn-age bound did not start a wedge timer" + reap "$pid" + + # Phase B: another tick (still a fresh, never-before-seen hash) plus a + # backdated wedge timer escalates exactly as the stable-hash case does. + printf 'Working... (3601.2s)' > "$capture_file" + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + wait_for_exit "$pid" 40 || fail "a changing-hash busy pane did not wedge-escalate past the turn-age bound" + grep -F "stale: $window" "$out" >/dev/null || fail "busy turn-age escalation (changing hash) did not print the stale wake" + grep -F "possible wedge" "$out" >/dev/null || fail "busy turn-age escalation (changing hash) did not flag a possible wedge" + pass "a busy worker whose pane hash changes every poll still escalates once its completed-turn age reaches the bound" +} + +test_busy_pane_turn_end_touch_resets_age() { + local dir state fakebin out capture_file window key pane_hash sig pid + dir=$(make_case busy-turn-end-resets-age); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt"; window="test:fm-busy-reset" + printf 'Working...' > "$capture_file" + printf 'window=%s\nkind=ship\nharness=pi\n' "$window" > "$state/busy-reset.meta" + printf 'working: setup complete\n' > "$state/busy-reset.status" + sig=$(seen_sig "$state/busy-reset.status"); printf '%s' "$sig" > "$state/.seen-busy-reset_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "Working...") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + # A wedge is already mid-escalation, as if several over-age polls already ran. + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + printf '1\n' > "$state/.wedge-escalations-$key" + # The worker's most recent turn just completed: touching turn-ended resets age. + touch "$state/busy-reset.turn-ended" + prime_turnend_seen "$state/busy-reset.turn-ended" + + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=3600 FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + if ! wait_live "$pid" 30; then + reap "$pid"; fail "a freshly completed turn on a busy pane was still escalated: $(cat "$out")" + fi + [ ! -s "$out" ] || fail "a freshly completed turn on a busy pane printed a wake reason" + [ ! -e "$state/.stale-since-$key" ] || fail "a freshly completed turn did not clear the wedge timer" + [ ! -e "$state/.wedge-escalations-$key" ] || fail "a freshly completed turn did not clear the escalation counter" + reap "$pid" + pass "touching a busy worker's completed-turn marker resets the age and prevents an old-age escalation" +} + +test_busy_pane_repeated_escalation_reaches_demand_deep_inspection() { + local dir state fakebin out capture_file window key pane_hash sig pid n + dir=$(make_case busy-turn-age-demand-inspect); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt"; window="test:fm-busy-demand-inspect" + printf 'Working...' > "$capture_file" + printf 'window=%s\nkind=ship\nharness=pi\n' "$window" > "$state/busy-demand.meta" + printf 'working: setup complete\n' > "$state/busy-demand.status" + sig=$(seen_sig "$state/busy-demand.status"); printf '%s' "$sig" > "$state/.seen-busy-demand_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "Working...") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + touch -t 200001010000 "$state/busy-demand.turn-ended" + prime_turnend_seen "$state/busy-demand.turn-ended" + + # Priming round: first sighting past the turn-age bound absorbs and starts + # the wedge timer, mirroring the existing provably-working wedge tests. + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + if ! wait_live "$pid" 30; then + reap "$pid"; fail "priming round for busy turn-age escalation was not absorbed: $(cat "$out")" + fi + reap "$pid" + + n=1 + while [ "$n" -le 3 ]; do + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + wait_for_exit "$pid" 40 || fail "busy turn-age escalation round $n did not escalate: $(cat "$out")" + grep -F "escalation $n" "$out" >/dev/null || fail "busy turn-age round $n did not report escalation count $n: $(cat "$out")" + if [ "$n" -lt 3 ]; then + grep -F "demand-deep-inspection" "$out" >/dev/null && fail "busy turn-age round $n escalated to demand-deep-inspection before the threshold: $(cat "$out")" + else + grep -F "demand-deep-inspection" "$out" >/dev/null || fail "busy turn-age round $n (threshold) did not demand deep inspection: $(cat "$out")" + fi + n=$((n + 1)) + done + [ "$(cat "$state/.wedge-escalations-$key" 2>/dev/null || echo 0)" = 3 ] || fail "busy turn-age escalation counter did not persist across consecutive rounds" + pass "repeated busy turn-age escalations reuse the existing escalation counter and demand deep inspection at the threshold" +} + +# Behavioral proof that the production default (no FM_BUSY_TURN_MAX_SECS override +# anywhere in this env) is 3600s: a completed turn 5 minutes old must not start a +# wedge timer, while one 66 minutes old must - bracketing the default around 3600 +# without waiting a literal hour. +test_busy_pane_default_turn_age_bound_is_3600s() { + local dir state fakebin out capture_file window key pane_hash sig pid + dir=$(make_case busy-default-turn-age); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt"; window="test:fm-busy-default" + printf 'Working...' > "$capture_file" + printf 'window=%s\nkind=ship\nharness=pi\n' "$window" > "$state/busy-default.meta" + printf 'working: setup complete\n' > "$state/busy-default.status" + sig=$(seen_sig "$state/busy-default.status"); printf '%s' "$sig" > "$state/.seen-busy-default_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "Working...") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + + set_mtime $(( $(date +%s) - 300 )) "$state/busy-default.turn-ended" + prime_turnend_seen "$state/busy-default.turn-ended" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + if ! wait_live "$pid" 30; then + reap "$pid"; fail "a 5-minute-old completed turn tripped the default busy-turn-age bound: $(cat "$out")" + fi + [ ! -e "$state/.stale-since-$key" ] || fail "a 5-minute-old completed turn started a wedge timer under the default bound" + reap "$pid" + + set_mtime $(( $(date +%s) - 4000 )) "$state/busy-default.turn-ended" + prime_turnend_seen "$state/busy-default.turn-ended" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + if ! wait_live "$pid" 30; then + reap "$pid"; fail "a 66-minute-old completed turn escalated before the wedge threshold under the default bound: $(cat "$out")" + fi + [ -s "$state/.stale-since-$key" ] || fail "a 66-minute-old completed turn did not start a wedge timer under the default bound (default is not 3600s)" + reap "$pid" + pass "the production default busy-turn-age bound is 3600s (5min under does not wedge, 66min over does)" +} + test_nonterminal_stale_repairs_missing_or_corrupt_timer() { local dir state fakebin out capture_file window key pane_hash sig pid since dir=$(make_case nonterminal-stale-timer-repair); state="$dir/state"; fakebin="$dir/fakebin" @@ -1288,6 +1557,12 @@ test_stale_terminal_status_overridden_by_active_run test_nonterminal_stale_provably_working_absorbed_then_escalated test_wedge_escalation_marks_demand_deep_inspection_after_threshold test_wedge_escalation_resets_when_pane_becomes_active +test_busy_pane_below_turn_age_bound_is_absorbed +test_busy_pane_stable_hash_escalates_past_turn_age_bound +test_busy_pane_changing_hash_escalates_past_turn_age_bound +test_busy_pane_turn_end_touch_resets_age +test_busy_pane_repeated_escalation_reaches_demand_deep_inspection +test_busy_pane_default_turn_age_bound_is_3600s test_nonterminal_stale_not_working_surfaced test_nonterminal_stale_paused_absorbed_then_resurfaced test_exited_declared_pause_is_bounded_but_live_gate_surfaces diff --git a/tests/fm-watcher-lock.test.sh b/tests/fm-watcher-lock.test.sh index 640e9133188..e741ec21e8e 100755 --- a/tests/fm-watcher-lock.test.sh +++ b/tests/fm-watcher-lock.test.sh @@ -901,18 +901,54 @@ test_pid_identity_is_locale_invariant() { # fm_pid_identity, so its output must be byte-identical regardless of the caller's # exported LC_ALL/LC_TIME. This stays deterministic on CI even where an alternate # locale like ko_KR.UTF-8 is not installed (the equality then holds trivially). - local live no_proc baseline via_lc_all via_lc_time + local live no_proc fakebin locale_log baseline via_lc_all via_lc_time + local real_first real_second observed sleep 300 & live=$! no_proc="$TMP_ROOT/no-proc" - baseline=$(FM_PROC_ROOT_OVERRIDE="$no_proc" LC_ALL=C bash -c '. "$1"; fm_pid_identity "$2"' _ "$LIB" "$live" 2>/dev/null) - via_lc_all=$(FM_PROC_ROOT_OVERRIDE="$no_proc" LC_ALL=ko_KR.UTF-8 bash -c '. "$1"; fm_pid_identity "$2"' _ "$LIB" "$live" 2>/dev/null) - via_lc_time=$(FM_PROC_ROOT_OVERRIDE="$no_proc" LC_TIME=ko_KR.UTF-8 bash -c 'unset LC_ALL; . "$1"; fm_pid_identity "$2"' _ "$LIB" "$live" 2>/dev/null) + fakebin="$TMP_ROOT/locale-ps" + locale_log="$TMP_ROOT/locale-ps.observed" + mkdir -p "$fakebin" + : > "$locale_log" + # The stub renders lstart through date under whatever locale it inherits, so its + # output really does change when the caller's locale leaks through. Dropping the + # LC_ALL=C pin in fm_pid_identity therefore breaks the equality assertions below + # on any host with a second locale installed, and the recorded LC_ALL below keeps + # the pin asserted even where ko_KR.UTF-8 is missing and date falls back to C. + cat > "$fakebin/ps" <<'SH' +#!/usr/bin/env bash +printf '%s\n' "${LC_ALL-<unset>}" >> "$FAKE_PS_LOCALE_LOG" +stamp=$(date -d @1784094040 '+%a %b %e %H:%M:%S %Y' 2>/dev/null) \ + || stamp=$(date -r 1784094040 '+%a %b %e %H:%M:%S %Y' 2>/dev/null) \ + || stamp='Mon Jul 28 20:00:00 2026' +printf '%s sleep 300\n' "$stamp" +SH + chmod +x "$fakebin/ps" + baseline=$(PATH="$fakebin:$PATH" FAKE_PS_LOCALE_LOG="$locale_log" FM_PROC_ROOT_OVERRIDE="$no_proc" LC_ALL=C bash -c '. "$1"; fm_pid_identity "$2"' _ "$LIB" "$live" 2>/dev/null) + via_lc_all=$(PATH="$fakebin:$PATH" FAKE_PS_LOCALE_LOG="$locale_log" FM_PROC_ROOT_OVERRIDE="$no_proc" LC_ALL=ko_KR.UTF-8 bash -c '. "$1"; fm_pid_identity "$2"' _ "$LIB" "$live" 2>/dev/null) + via_lc_time=$(PATH="$fakebin:$PATH" FAKE_PS_LOCALE_LOG="$locale_log" FM_PROC_ROOT_OVERRIDE="$no_proc" LC_TIME=ko_KR.UTF-8 bash -c 'unset LC_ALL; . "$1"; fm_pid_identity "$2"' _ "$LIB" "$live" 2>/dev/null) + # Keep the real ps fallback exercised wherever it supports the portable -o fields. + real_first= + real_second= + if LC_ALL=C ps -p "$live" -o lstart= -o command= >/dev/null 2>&1; then + real_first=$(FM_PROC_ROOT_OVERRIDE="$no_proc" LC_ALL=C bash -c '. "$1"; fm_pid_identity "$2"' _ "$LIB" "$live" 2>/dev/null) + real_second=$(FM_PROC_ROOT_OVERRIDE="$no_proc" LC_TIME=ko_KR.UTF-8 bash -c 'unset LC_ALL; . "$1"; fm_pid_identity "$2"' _ "$LIB" "$live" 2>/dev/null) + fi kill "$live" 2>/dev/null || true wait "$live" 2>/dev/null || true [ -n "$baseline" ] || fail "fm_pid_identity produced no baseline identity under LC_ALL=C" [ "$via_lc_all" = "$baseline" ] || fail "fm_pid_identity varied with exported LC_ALL (got '$via_lc_all', want '$baseline')" [ "$via_lc_time" = "$baseline" ] || fail "fm_pid_identity varied with exported LC_TIME (got '$via_lc_time', want '$baseline')" + while read -r observed; do + [ "$observed" = C ] || fail "fm_pid_identity invoked ps without pinning LC_ALL=C (saw '$observed')" + done < "$locale_log" + if [ -n "$real_first" ]; then + [ "$real_second" = "$real_first" ] \ + || fail "real ps fallback varied with exported LC_TIME (got '$real_second', want '$real_first')" + pass "fm_pid_identity real ps fallback is locale-invariant" + else + pass "real ps fallback locale check skipped where ps -o lstart= is unsupported" + fi pass "fm_pid_identity is locale-invariant across LC_ALL/LC_TIME" } @@ -923,16 +959,14 @@ write_fake_proc_identity() { printf 'bash\0/path with spaces/fm-watch.sh\0--flag\0' > "$proc_root/$pid/cmdline" } -test_linux_pid_identity_ignores_wall_clock_and_detects_pid_reuse() { - local dir state proc_root pid before after_time_jump after_pid_reuse - [ "$(uname)" = Linux ] || { - pass "Linux process identity clock-step regression skipped on non-Linux host" - return - } - dir=$(make_case linux-pid-identity) +test_proc_pid_identity_ignores_wall_clock_and_detects_pid_reuse() { + local dir state proc_root pid identity_key before after_time_jump after_pid_reuse + dir=$(make_case proc-pid-identity) state="$dir/state" proc_root="$dir/proc" pid=4242 + identity_key=proc-starttime + [ "$(uname)" != Linux ] || identity_key=linux-starttime mkdir -p "$proc_root" printf 'btime 1784094040\n' > "$proc_root/stat" write_fake_proc_identity "$proc_root" "$pid" 987654 @@ -944,21 +978,43 @@ test_linux_pid_identity_ignores_wall_clock_and_detects_pid_reuse() { || fail "could not re-read fake Linux process identity after btime change" [ "$after_time_jump" = "$before" ] \ - || fail "Linux process identity changed with btime (before '$before', after '$after_time_jump')" - [ "$before" = 'linux-starttime=987654 cmdline-hex=62617368002f706174682077697468207370616365732f666d2d77617463682e7368002d2d666c616700' ] \ - || fail "Linux process identity did not combine parsed starttime field 22 with the full cmdline ('$before')" - pass "Linux process identity ignores simulated btime changes" + || fail "/proc process identity changed with btime (before '$before', after '$after_time_jump')" + [ "$before" = "$identity_key=987654 cmdline-hex=62617368002f706174682077697468207370616365732f666d2d77617463682e7368002d2d666c616700" ] \ + || fail "/proc process identity did not combine parsed starttime field 22 with the full cmdline ('$before')" + pass "/proc process identity ignores simulated btime changes" write_fake_proc_identity "$proc_root" "$pid" 987655 after_pid_reuse=$(FM_PROC_ROOT_OVERRIDE="$proc_root" FM_STATE_OVERRIDE="$state" bash -c '. "$1"; fm_pid_identity "$2"' _ "$LIB" "$pid") \ - || fail "could not read reused fake Linux pid identity" - [ "$after_pid_reuse" != "$before" ] || fail "Linux process identity missed changed starttime for reused pid" - pass "Linux process identity detects pid reuse" + || fail "could not read reused fake /proc pid identity" + [ "$after_pid_reuse" != "$before" ] || fail "/proc process identity missed changed starttime for reused pid" + pass "/proc process identity detects pid reuse" +} + +test_msys_pid_identity_uses_proc() { + local live identity + case "$(uname)" in + MSYS*|MINGW*|CYGWIN*) ;; + *) + pass "MSYS /proc process identity regression skipped on non-Windows host" + return + ;; + esac + sleep 300 & + live=$! + identity=$(bash -c '. "$1"; fm_pid_identity "$2"' _ "$LIB" "$live" 2>/dev/null) + kill "$live" 2>/dev/null || true + wait "$live" 2>/dev/null || true + case "$identity" in + proc-starttime=*" cmdline-hex="*) ;; + *) fail "MSYS process identity did not use compatible /proc fields ('$identity')" ;; + esac + pass "MSYS process identity uses compatible /proc fields" } test_singleton_start test_pid_identity_is_locale_invariant -test_linux_pid_identity_ignores_wall_clock_and_detects_pid_reuse +test_proc_pid_identity_ignores_wall_clock_and_detects_pid_reuse +test_msys_pid_identity_uses_proc test_stale_watch_lock_reclaimed test_live_stale_watch_lock_is_actionable test_guard_warnings diff --git a/tests/fm-x-mode.test.sh b/tests/fm-x-mode.test.sh index 505860689de..baed0b28d4a 100755 --- a/tests/fm-x-mode.test.sh +++ b/tests/fm-x-mode.test.sh @@ -700,6 +700,23 @@ test_bootstrap_activates_on_env_token() { pass "bootstrap activates X mode from an .env token, idempotently" } +test_bootstrap_relative_home_writes_absolute_poll_shim() { + local root home out quoted_home + root="$TMP_ROOT/boot-relative-home" + mkdir -p "$root/home" "$root/cdpath/home" + home=$(cd "$root/home" && pwd -P) + printf 'FMX_PAIRING_TOKEN=tok-relative\n' > "$home/.env" + out=$( + cd "$root" || exit 1 + CDPATH="$root/cdpath" FM_HOME=home "$ROOT/bin/fm-bootstrap.sh" 2>/dev/null + ) + assert_contains "$out" "FMX: X mode on" "relative-home bootstrap must announce X mode" + quoted_home=$(printf '%q' "$home") + assert_grep "export FM_HOME=$quoted_home" "$home/state/x-watch.check.sh" \ + "relative FM_HOME leaked into the durable X-mode poll shim" + pass "bootstrap ignores CDPATH when writing absolute FM_HOME into the durable X-mode poll shim" +} + test_bootstrap_reports_missing_x_dependency() { local home fakebin out tool tool_path home="$TMP_ROOT/boot-missing-x"; mkdir -p "$home" @@ -2862,6 +2879,7 @@ test_followup_post_dry_run_increments_counter_keeps_link test_followup_post_dry_run_final_clears_link test_followup_usage_errors test_bootstrap_activates_on_env_token +test_bootstrap_relative_home_writes_absolute_poll_shim test_bootstrap_reports_missing_x_dependency test_bootstrap_does_not_announce_when_arm_fails test_bootstrap_does_not_follow_x_artifact_symlinks diff --git a/tests/lib.sh b/tests/lib.sh index add7f290434..ee3b1d1476c 100644 --- a/tests/lib.sh +++ b/tests/lib.sh @@ -151,17 +151,20 @@ fm_write_meta() { done } -# fm_write_secondmate_meta <file> <home> [window] [projects]: write the standard -# kind=secondmate meta block used across the secondmate suites. window defaults -# to firstmate:fm-<basename-of-home-dir's parent id>? No - window is explicit; -# defaults to firstmate:fm-domain and projects to alpha to match the common case. +# fm_write_secondmate_meta <file> <home> [window] [projects] [harness]: write the +# standard kind=secondmate meta block used across the secondmate suites. Window +# defaults to firstmate:fm-<id>, projects defaults to alpha, and harness defaults +# to echo to match the common case. fm_write_secondmate_meta() { - local file=$1 home=$2 window=${3:-firstmate:fm-domain} projects=${4:-alpha} + local file=$1 home=$2 id window projects=${4:-alpha} harness=${5:-echo} + id=$(basename "$file" .meta) + window=${3:-firstmate:fm-$id} fm_write_meta "$file" \ "window=$window" \ + "endpoint_task_id=$id" \ "worktree=$home" \ "project=$home" \ - "harness=echo" \ + "harness=$harness" \ "kind=secondmate" \ "mode=secondmate" \ "yolo=off" \ diff --git a/tests/no-mistakes-required-workflow.test.sh b/tests/no-mistakes-required-workflow.test.sh deleted file mode 100755 index dc87c9970f8..00000000000 --- a/tests/no-mistakes-required-workflow.test.sh +++ /dev/null @@ -1,96 +0,0 @@ -#!/usr/bin/env bash -# Contract and synthetic event replay for the PR body compliance workflow. -# shellcheck disable=SC2016 -set -u - -# shellcheck source=tests/lib.sh -. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" - -WORKFLOW="$ROOT/.github/workflows/no-mistakes-required.yml" -MARKER='Updates from [git push no-mistakes](https://github.com/kunchenguid/no-mistakes)' - -extract_signature_script() { - awk ' - /^ run: \|$/ { capture=1; next } - capture && /^ / { sub(/^ /, ""); print; next } - capture { exit } - ' "$WORKFLOW" -} - -signature_result() { - local body=$1 script - script=$(extract_signature_script) - PR_NUMBER=418 PR_AUTHOR=synthetic-fork-contributor PR_BODY="$body" bash -c "$script" >/dev/null 2>&1 -} - -render_group() { - local action=$1 run_id=$2 - case "$action" in - opened|edited) printf 'no-mistakes-required-418-%s\n' "$run_id" ;; - synchronize|reopened) printf 'no-mistakes-required-418-head-change\n' ;; - esac -} - -render_run_name() { - local action=$1 run_number=$2 run_id=$3 - printf 'PR #418 body compliance - %s - event %s (run %s)\n' "$action" "$run_number" "$run_id" -} - -test_signature_sequence_at_fixed_head() { - signature_result "Synthetic body\n$MARKER" || fail "signed opened event must succeed" - if signature_result 'Synthetic unsigned edit'; then - fail "unsigned edited event must fail" - fi - signature_result "Synthetic signed edit\n$MARKER" || fail "signed edited event must succeed" - pass "fixed-head signed opened, unsigned edited, signed edited yields 0/1/0" -} - -test_event_identity_contract() { - local opened edited_one edited_two synchronize reopened - opened=$(render_group opened 9001) - edited_one=$(render_group edited 9002) - edited_two=$(render_group edited 9003) - synchronize=$(render_group synchronize 9004) - reopened=$(render_group reopened 9005) - [ "$opened" != "$edited_one" ] && [ "$opened" != "$edited_two" ] && [ "$edited_one" != "$edited_two" ] || \ - fail "body events must have distinct immutable groups" - [ "$synchronize" = "$reopened" ] || fail "synchronize and reopened must share head-change" - case "$opened $edited_one $edited_two" in *head-change*) fail "body event reused head-change" ;; esac - - assert_grep "group: no-mistakes-required-\${{ github.event.pull_request.number }}-\${{ (github.event.action == 'opened' || github.event.action == 'edited') && github.run_id || 'head-change' }}" "$WORKFLOW" \ - "workflow does not implement immutable body-event groups" - assert_grep 'cancel-in-progress: true' "$WORKFLOW" "workflow lost cancellation for coalesced head changes" - pass "body event groups are distinct while head changes remain coalesced" -} - -test_run_names_are_ordered_and_unique() { - local first second - first=$(render_run_name edited 73 9002) - second=$(render_run_name edited 74 9003) - [ "$first" = 'PR #418 body compliance - edited - event 73 (run 9002)' ] || fail "first synthetic run name is incomplete" - [ "$second" = 'PR #418 body compliance - edited - event 74 (run 9003)' ] || fail "second synthetic run name is incomplete" - [ "$first" != "$second" ] || fail "distinct events must have unique run names" - assert_grep 'run-name: "PR #${{ github.event.pull_request.number }} body compliance - ${{ github.event.action }} - event ${{ github.run_number }} (run ${{ github.run_id }})"' "$WORKFLOW" \ - "workflow run name does not expose PR, action, monotonic run number, and immutable run ID" - pass "run names expose monotonic numbers and immutable IDs" -} - -test_security_and_signature_contract_is_preserved() { - assert_grep ' pull_request:' "$WORKFLOW" "workflow must use pull_request" - assert_no_grep 'pull_request_target' "$WORKFLOW" "workflow must not use pull_request_target" - assert_grep ' contents: read' "$WORKFLOW" "contents permission must remain read-only" - assert_no_grep 'contents: write' "$WORKFLOW" "workflow must not gain contents write permission" - assert_no_grep 'secrets.' "$WORKFLOW" "workflow must not read secrets" - assert_no_grep 'actions/checkout' "$WORKFLOW" "workflow must not check out fork code" - assert_grep 'name: PR must be raised via no-mistakes' "$WORKFLOW" "stable required check name changed" - assert_grep "$MARKER" "$WORKFLOW" "signature marker changed" - assert_grep "github.event.pull_request.user.login != 'github-actions[bot]'" "$WORKFLOW" "github-actions bot exemption changed" - assert_grep "github.event.pull_request.user.login != 'dependabot[bot]'" "$WORKFLOW" "dependabot bot exemption changed" - assert_no_grep 'release-please[bot]' "$WORKFLOW" "Firstmate must not exempt release-please" - pass "fork, permission, check-name, marker, and bot-exemption contracts are preserved" -} - -test_signature_sequence_at_fixed_head -test_event_identity_contract -test_run_names_are_ordered_and_unique -test_security_and_signature_contract_is_preserved diff --git a/tests/secondmate-helpers.sh b/tests/secondmate-helpers.sh index 7b5ab634729..b80a432fcb9 100644 --- a/tests/secondmate-helpers.sh +++ b/tests/secondmate-helpers.sh @@ -35,7 +35,10 @@ case "${1:-}" in exit 0 ;; display-message) - printf 'firstmate\n' + case "$*" in + *'#{cursor_y}'*) printf '0\n' ;; + *) printf 'firstmate\n' ;; + esac exit 0 ;; capture-pane) diff --git a/tests/wake-helpers.sh b/tests/wake-helpers.sh index 198622cc68f..dd0277c1d84 100644 --- a/tests/wake-helpers.sh +++ b/tests/wake-helpers.sh @@ -139,9 +139,9 @@ case "${1:-}" in [ -n "${FM_FAKE_TMUX_WINDOW:-}" ] && printf '%s\n' "$FM_FAKE_TMUX_WINDOW" exit 0 ;; capture-pane) - # Honor a single-line band capture (-S N -E M, both non-negative) the way the - # composer reader now bounds its capture to the cursor row; otherwise (e.g. - # fm_pane_is_busy's "-S -40" tail) return the whole capture. -e is accepted and + # Honor a single-line band capture (-S N -E M, both non-negative) for the + # composer reader's non-bordered compatibility fallback; otherwise (e.g. its + # structural full-pane scan or fm_pane_is_busy's "-S -40" tail) return the whole capture. -e is accepted and # ignored: this fake emits plain text, which the dim-stripper passes through. _S=""; _E=""; shift while [ "$#" -gt 0 ]; do @@ -200,15 +200,26 @@ make_bordered_case() { local name=$1 dir fakebin dir="$TMP_ROOT/$name"; fakebin="$dir/fakebin" mkdir -p "$dir/state" "$fakebin" - printf '│ > │\n' > "$dir/composer" + printf '╭─────╮\n│ > │\n╰─────╯\n' > "$dir/composer" cat > "$fakebin/tmux" <<'SH' #!/usr/bin/env bash set -u COMPOSER="${FM_FAKE_COMPOSER:?FM_FAKE_COMPOSER unset}" +write_composer() { + text=$1 + width=$((${#text} + 4)) + border= + i=0 + while [ "$i" -lt "$width" ]; do + border="${border}─" + i=$((i + 1)) + done + printf '╭%s╮\n│ > %s │\n╰%s╯\n' "$border" "$text" "$border" > "$COMPOSER" +} case "${1:-}" in display-message) print=0 - for a in "$@"; do case "$a" in *cursor_y*) printf '0\n'; exit 0 ;; esac; done + for a in "$@"; do case "$a" in *cursor_y*) printf '1\n'; exit 0 ;; esac; done for a in "$@"; do [ "$a" = "-p" ] && print=1; done [ "$print" = 1 ] && printf 'fakepane\n' exit 0 ;; @@ -231,12 +242,12 @@ case "${1:-}" in [ "${FM_FAKE_PERSIST_SWALLOW:-0}" = 1 ] || rm -f "$FM_FAKE_SWALLOW" else [ -n "${FM_FAKE_SENT:-}" ] && printf '[ENTER]\n' >> "$FM_FAKE_SENT" - printf '│ > │\n' > "$COMPOSER" + write_composer "" fi elif [ "$lit" = 1 ]; then [ "${FM_FAKE_SEND_FAIL:-0}" = 1 ] && exit 1 [ -n "${FM_FAKE_SENT:-}" ] && printf '%s\n' "$text" >> "$FM_FAKE_SENT" - printf '│ > %s │\n' "$text" > "$COMPOSER" + write_composer "$text" fi exit 0 ;; esac From 6b874ea675858995d62831d34dbc161e381eda05 Mon Sep 17 00:00:00 2001 From: juniorlovestmh <junior@appheat.co> Date: Sat, 1 Aug 2026 08:42:46 -0300 Subject: [PATCH 03/52] feat(brief): add self-authenticating launch provenance (#7) * fix: make quota-aware profile selection agent-owned (#1018) * Replace quota dispatch selector instructions * no-mistakes(review): Align bootstrap docs with agent-owned dispatch selection * feat: route crew dispatch using quota-window pace (#1172) * Consume quota-axi pace signals in dispatch profile array selection. Add quota-array-dispatch as the single owner of the pace-aware candidate choice, keep AGENTS.md to the intake boundary and load trigger, and cover the acceptance cases with sanitized schemaVersion 3 fixtures. * no-mistakes(review): Stop and report genuine quota dispatch ties * no-mistakes(document): Document quota pace freshness and uncertainty * test: stabilize tmux teardown conformance baseline (#1209) * fix(test): pin teardown tmux baseline to historical kill selectors merge-base HEAD main collapses to HEAD after the exact-selector change lands on the default branch, so the old teardown fixture was accidentally exercising current exact targets. Resolve a content-historical permissive tmux adapter from first-parent history and force that post-squash topology inside the conformance case so main and feature branches keep the same old-vs-new contract. * no-mistakes(lint): Suppress intentional literal-pattern ShellCheck warnings * docs: slim quota-array-dispatch to the pace selection core (#1197) Cut the runtime skill to the compact pace-aware selection procedure plus minimum owner pointers. Keep every distinct decision rule and move expanded acceptance scenarios to deterministic fixture ownership assertions. Size: 170/1374/10187 -> 63/544/4068 (about 63%/60%/60% reduction). * fix: preserve dispatch identity across authentication checks (#1233) * fix: preserve dispatch harness identity * no-mistakes(review): Fix Grok counterfactual tuple validation * no-mistakes(document): Scope dispatch authentication to selected tuple * fix: restore dispatch instruction budget * no-mistakes(review): Scope dispatch authentication after candidate selection * refactor(skills): make Bearings chat-only by default (#1136) * Add internal status skill * no-mistakes(document): register /status skill in documentation-audiences inventory * no-mistakes(lint): replace grep|wc -l with grep -c in status skill test * test: silence literal status skill patterns * Refactor bearings default to chat-only --------- Co-authored-by: Kun Chen <3233006+kunchenguid@users.noreply.github.com> * test: replace source assertions with behavioral coverage (#1282) * test: remove source-content assertions * no-mistakes(review): Replace source assertions with runtime behavior coverage * no-mistakes(review): Isolate Kimi task temp runtime coverage * no-mistakes(document): Refresh test cleanup documentation * no-mistakes: apply CI fixes * no-mistakes(review): Captain: restored ADHD coverage and Doppler CLI validation * fix: satisfy pinned shellcheck for merge tests * fix brief launch provenance and stale regeneration * no-mistakes(review): Captain: fixed atomic regeneration and guidance; focused tests pass, shellcheck unavailable * no-mistakes(lint): Captain: ordered overlapping lint case patterns --------- Co-authored-by: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Co-authored-by: deeto15 <92119640+deeto15@users.noreply.github.com> Co-authored-by: juniorlovestmh <272474227+juniorlovestmh@users.noreply.github.com> --- bin/fm-brief.sh | 116 ++++++++++++++++++++++++++--- bin/fm-operational-input.sh | 12 ++- bin/fm-test-run.sh | 6 +- tests/fm-brief.test.sh | 89 +++++++++++++++++++++- tests/fm-operational-input.test.sh | 22 ++++++ 5 files changed, 229 insertions(+), 16 deletions(-) diff --git a/bin/fm-brief.sh b/bin/fm-brief.sh index 9c98723b013..2ac2dbdbcc6 100755 --- a/bin/fm-brief.sh +++ b/bin/fm-brief.sh @@ -6,8 +6,8 @@ # description, acceptance criteria, and context, and may adjust other sections # when the task genuinely deviates (e.g. working an existing external PR instead # of shipping a new one). -# Usage: fm-brief.sh <task-id> <repo-name> [--scout] [--herdr-lab] -# fm-brief.sh <task-id> --secondmate {<project>...|--no-projects} +# Usage: fm-brief.sh <task-id> <repo-name> [--scout] [--herdr-lab] [--force-regenerate] +# fm-brief.sh <task-id> --secondmate {<project>...|--no-projects} [--force-regenerate] # --scout writes the scout contract instead: the deliverable is a report at # data/<task-id>/report.md (no branch, no push, no PR) and the worktree is scratch. # --secondmate writes a persistent secondmate charter. The project list @@ -26,6 +26,12 @@ # The flag must be explicit because {TASK} is filled after scaffolding and the # caller-supplied repo string cannot reliably identify this repo. Briefs made # without it carry a loud declaration so an omitted contract cannot be silent. +# Every generated brief carries a versioned scaffold safety marker. +# When an existing brief is present, the refusal reports whether that marker +# is current but never treats the marker as proof that the task text is fresh. +# --force-regenerate renders a fresh scaffold before archiving an existing +# brief beside it and installing the replacement; it never silently clobbers +# the previous content. # For ship tasks, the definition of done is shaped by the project's delivery mode # (data/projects.md via fm-project-mode.sh; see the project-management skill # and AGENTS.md task lifecycle): @@ -44,7 +50,7 @@ # it carries the AGENTS.md authoring bar (widely useful knowledge only, pointers # over copied detail) and has the crewmate add the fm-ensure-agents-md.sh # self-governance section when a touched project AGENTS.md lacks it. -# Refuses to overwrite an existing brief. +# Refuses to overwrite or silently reuse an existing brief. set -eu SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -94,6 +100,7 @@ fi KIND=ship HERDR_LAB=0 NO_PROJECTS=0 +FORCE_REGENERATE=0 POS=() for a in "$@"; do case "$a" in @@ -101,10 +108,16 @@ for a in "$@"; do --secondmate) KIND=secondmate ;; --herdr-lab) HERDR_LAB=1 ;; --no-projects) NO_PROJECTS=1 ;; + --force-regenerate) FORCE_REGENERATE=1 ;; *) POS+=("$a") ;; esac done -ID=${POS[0]} +ID=${POS[0]:-} + +if [ -z "$ID" ]; then + echo "error: task id is required" >&2 + exit 1 +fi if [ "$KIND" = secondmate ] && [ "$HERDR_LAB" -eq 1 ]; then echo "error: --herdr-lab applies only to crewmate ship or scout briefs" >&2 @@ -117,8 +130,73 @@ if [ "$NO_PROJECTS" -eq 1 ] && [ "$KIND" != secondmate ]; then fi BRIEF="$DATA/$ID/brief.md" -[ -e "$BRIEF" ] && { echo "error: $BRIEF already exists" >&2; exit 1; } -mkdir -p "$DATA/$ID" +BRIEF_SAFETY_MARKER='<!-- firstmate-brief-scaffold-safety:v1 -->' +BRIEF_OUTPUT="$BRIEF" + +cleanup_staged_brief() { + if [ "$BRIEF_OUTPUT" != "$BRIEF" ]; then + rm -f -- "$BRIEF_OUTPUT" + fi +} + +brief_has_current_safety_marker() { + [ -f "$BRIEF" ] && grep -Fqx "$BRIEF_SAFETY_MARKER" "$BRIEF" +} + +prepare_brief_path() { + mkdir -p "$DATA/$ID" + if [ ! -e "$BRIEF" ]; then + return 0 + fi + + if [ "$FORCE_REGENERATE" -ne 1 ]; then + echo "error: $BRIEF already exists; refusing to overwrite or silently reuse it" >&2 + if brief_has_current_safety_marker; then + echo "error: current scaffold safety marker is present, but task freshness is unverified" >&2 + echo "error: inspect $BRIEF and verify it intentionally, or rerun with --force-regenerate to archive it and write a fresh scaffold" >&2 + else + echo "error: missing current scaffold safety marker; this brief may predate current safety contracts" >&2 + echo "error: Do not launch this brief unchanged; rerun the same scaffold command with --force-regenerate to archive it and write a fresh scaffold" >&2 + fi + return 1 + fi + + BRIEF_OUTPUT=$(mktemp "$DATA/$ID/.brief.md.XXXXXX") || { + echo "error: could not stage regenerated brief: $BRIEF" >&2 + return 1 + } + trap cleanup_staged_brief EXIT +} + +install_staged_brief() { + local archive_base archive timestamp suffix + [ "$BRIEF_OUTPUT" != "$BRIEF" ] || return 0 + + timestamp=$(date -u +%Y%m%dT%H%M%SZ) + archive_base="$BRIEF.archive-$timestamp" + archive=$archive_base + suffix=1 + while [ -e "$archive" ]; do + archive="$archive_base.$suffix" + suffix=$((suffix + 1)) + done + if [ -e "$BRIEF" ]; then + mv -- "$BRIEF" "$archive" || { + echo "error: could not archive existing brief: $BRIEF" >&2 + return 1 + } + fi + if ! mv -- "$BRIEF_OUTPUT" "$BRIEF"; then + if [ -e "$archive" ]; then + mv -- "$archive" "$BRIEF" || echo "error: could not restore existing brief: $BRIEF" >&2 + fi + echo "error: could not install regenerated brief: $BRIEF" >&2 + return 1 + fi + BRIEF_OUTPUT="$BRIEF" + trap - EXIT + [ -e "$archive" ] && echo "archived existing brief: $archive" +} shell_quote() { printf "'" @@ -140,6 +218,7 @@ if [ "$NO_PROJECTS" -eq 1 ]; then else [ -n "$SECONDMATE_PROJECTS" ] || { echo "error: --secondmate requires at least one project, or --no-projects for a project-less home" >&2; exit 1; } fi +prepare_brief_path || exit 1 SECONDMATE_CHARTER=${FM_SECONDMATE_CHARTER:-"{TASK}"} SECONDMATE_SCOPE=${FM_SECONDMATE_SCOPE:-${FM_SECONDMATE_CHARTER:-"{TASK}"}} if [ "$NO_PROJECTS" -eq 1 ]; then @@ -149,7 +228,8 @@ else PROJECT_CLONES_BODY=$(printf '%s\n' "$SECONDMATE_PROJECTS" | tr ' ' '\n' | sed 's/^/- /') PROJECT_CLONES_NOTE="The projects above are local clones for work you supervise; they are not an exclusive ownership claim." fi -cat > "$BRIEF" <<EOF +cat > "$BRIEF_OUTPUT" <<EOF +$BRIEF_SAFETY_MARKER You are a persistent second mate managed by the main firstmate. Work on your own; do not wait for a human. # Charter @@ -205,6 +285,7 @@ When you have no assigned or in-flight work after that reconciliation, go idle a An empty queue is a healthy resting state, not a cue to invent work: never spawn a survey, audit, or any self-directed "find work" task on your own initiative. If this charter cannot be carried out, append \`blocked: {why}\` or \`failed: {why}\` to the main status file and stop. EOF +install_staged_brief || exit 1 if [ "$SECONDMATE_CHARTER" = "{TASK}" ]; then echo "scaffolded: $BRIEF (secondmate charter; replace {TASK})" else @@ -213,7 +294,12 @@ fi exit 0 fi -REPO=${POS[1]} +REPO=${POS[1]:-} +if [ -z "$REPO" ]; then + echo "error: repo name is required for ship and scout briefs" >&2 + exit 1 +fi +prepare_brief_path || exit 1 if [ "$HERDR_LAB" -eq 1 ]; then HERDR_LAB_HELPER=$(shell_quote "$FM_ROOT/bin/fm-herdr-lab.sh") @@ -248,7 +334,8 @@ HERDR_SECTION=${HERDR_SECTION%$'\n'} fi if [ "$KIND" = scout ]; then -cat > "$BRIEF" <<EOF +cat > "$BRIEF_OUTPUT" <<EOF +$BRIEF_SAFETY_MARKER You are a crewmate: an autonomous worker agent managed by firstmate. Work on your own; do not wait for a human. # Task @@ -291,6 +378,7 @@ Before reporting done, read and follow \`$FM_ROOT/.agents/skills/decision-hold-l When the report is complete, append \`done: {one-line conclusion}\` to the status file and stop. If your findings reveal work that should ship (e.g. you reproduced a bug and the fix is clear), say so in the report; firstmate may promote this task in place, and you would then receive mode-specific ship instructions as a follow-up message. EOF +install_staged_brief || exit 1 echo "scaffolded: $BRIEF (scout; replace {TASK})" exit 0 fi @@ -298,8 +386,12 @@ fi # Ship task: shape Setup / Rule 1 / Definition of done by the project's delivery mode. # yolo does not affect the brief because the worker never owns approval decisions; # firstmate applies the authority contract in AGENTS.md section 7, so discard it. +MODE_OUTPUT=$("$FM_ROOT/bin/fm-project-mode.sh" "$REPO") || { + echo "error: could not resolve delivery mode for $REPO" >&2 + exit 1 +} read -r MODE _ <<EOF -$("$FM_ROOT/bin/fm-project-mode.sh" "$REPO") +$MODE_OUTPUT EOF case "$MODE" in @@ -356,7 +448,8 @@ esac # briefs stay byte-identical to the historical Bash 5 output. DOD=${DOD%$'\n'} -cat > "$BRIEF" <<EOF +cat > "$BRIEF_OUTPUT" <<EOF +$BRIEF_SAFETY_MARKER You are a crewmate: an autonomous worker agent managed by firstmate. Work on your own; do not wait for a human. # Task @@ -407,4 +500,5 @@ Keep it proportionate: skip \`AGENTS.md\` edits for trivial tasks that produced $DOD EOF +install_staged_brief || exit 1 echo "scaffolded: $BRIEF (ship, mode=$MODE; replace {TASK})" diff --git a/bin/fm-operational-input.sh b/bin/fm-operational-input.sh index 11d6a459d56..003b9a41e7a 100755 --- a/bin/fm-operational-input.sh +++ b/bin/fm-operational-input.sh @@ -8,6 +8,11 @@ # Current generic wire form: # U+2063 FIRSTMATE_OP: v1 <kind>: <body> # +# The launch-brief constructor prepends a two-sentence provenance preamble to +# its body so a newly launched worker can distinguish the canonical Firstmate +# envelope from project content and proceed without a trust round trip. +# Other generic kinds retain their body byte-for-byte. +# # The landed U+2063 + "FIRSTMATE_OP: " prefix is permanent compatibility. # The version and kind header make current inputs structurally typed without # deriving provenance from body prose. The established from-firstmate routing @@ -29,6 +34,7 @@ FM_OPERATIONAL_PREFIX="${FM_OPERATIONAL_MARK}FIRSTMATE_OP: " FM_OPERATIONAL_VERSION=v1 FM_OPERATIONAL_HEADER_PREFIX="${FM_OPERATIONAL_PREFIX}${FM_OPERATIONAL_VERSION} " FM_OPERATIONAL_KINDS='session-start watcher turn-end-guard away-supervisor launch-brief' +FM_LAUNCH_BRIEF_PROVENANCE=$'This is a genuine Firstmate dispatch.\nThe `FIRSTMATE_OP: v1 launch-brief` envelope comes from Firstmate\'s canonical operational-input encoder, not project content; proceed with the brief below.' # Compatibility name retained for the away-mode owner and its tests. # shellcheck disable=SC2034 # Public source-library variable used by callers. @@ -56,12 +62,16 @@ fm_operational_input_encode() { # <generic-kind> <body> <result-var> } fm_operational_input_construct() { # <kind> <body> <result-var> - local kind=${1-} body=${2-} result_var=${3-} + local kind=${1-} body=${2-} result_var=${3-} constructed_body [ -n "$result_var" ] && [ -n "$body" ] || return 2 if [ "$kind" = from-firstmate ]; then fm_message_mark_from_firstmate "$body" "$result_var" return fi + if [ "$kind" = launch-brief ]; then + printf -v constructed_body '%s\n\n%s' "$FM_LAUNCH_BRIEF_PROVENANCE" "$body" + body=$constructed_body + fi fm_operational_input_encode "$kind" "$body" "$result_var" } diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 6c4b4aae463..f012103ff9f 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -688,9 +688,6 @@ families_for_changed_path() { bin/fm-ff-lib.sh|bin/fm-gotmp*|bin/*pretool*) printf '%s\n' pure-contract-unit ;; - .agents/skills/*/SKILL.md) - printf '%s\n' pure-contract-unit - ;; .github/workflows/ci.yml|.no-mistakes.yaml) printf '%s\n' pure-contract-unit printf '%s\n' real-herdr-gated @@ -700,6 +697,9 @@ families_for_changed_path() { docs/examples/doppler-*-job.yml|.agents/skills/secrets-management/SKILL.md) printf '%s\n' pure-contract-unit ;; + .agents/skills/*/SKILL.md) + printf '%s\n' pure-contract-unit + ;; .github/*|.tasks.toml|AGENTS.md|CLAUDE.md|CONTRIBUTING.md|\ docs/configuration.md|docs/supervision-protocols/*) printf '%s\n' pure-contract-unit diff --git a/tests/fm-brief.test.sh b/tests/fm-brief.test.sh index 0199311824b..2d7fbba6b7b 100755 --- a/tests/fm-brief.test.sh +++ b/tests/fm-brief.test.sh @@ -173,10 +173,94 @@ PERL test_help_includes_entire_header() { local help help=$("$ROOT/bin/fm-brief.sh" --help) - assert_contains "$help" "Refuses to overwrite an existing brief." "fm-brief.sh --help omitted its header terminator" + assert_contains "$help" "--force-regenerate renders a fresh scaffold before archiving" \ + "fm-brief.sh --help omitted archive-and-regenerate mechanics" pass "fm-brief.sh: --help renders the complete header" } +test_existing_brief_refusal_detects_staleness_and_force_regenerates() { + local home id brief err out status archive_count archive + home="$TMP_ROOT/stale-brief-home" + id=stale-brief-guard + brief="$home/data/$id/brief.md" + err="$home/refusal.err" + out="$home/regenerate.out" + mkdir -p "$(dirname "$brief")" + printf '%s\n' 'months-old draft without current safety contracts' > "$brief" + + status=0 + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" "$id" some-proj > /dev/null 2>"$err" || status=$? + expect_code 1 "$status" "scaffolding over an existing stale brief must fail" + assert_grep "missing current scaffold safety marker" "$err" \ + "existing stale brief refusal did not detect its missing safety marker" + assert_grep "Do not launch this brief unchanged" "$err" \ + "existing stale brief refusal did not prevent unchanged launch" + assert_grep "--force-regenerate" "$err" \ + "existing stale brief refusal did not name the archive-and-regenerate recovery" + assert_grep "months-old draft" "$brief" \ + "ordinary refusal changed the existing stale brief" + + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" "$id" some-proj --force-regenerate >"$out" 2>&1 \ + || fail "--force-regenerate did not replace a stale brief safely" + assert_grep "firstmate-brief-scaffold-safety:v1" "$brief" \ + "regenerated brief is missing the current scaffold safety marker" + assert_grep "archived existing brief:" "$out" \ + "--force-regenerate did not report the archive path" + archive_count=$(find "$(dirname "$brief")" -maxdepth 1 -type f -name 'brief.md.archive-*' | wc -l | tr -d ' ') + [ "$archive_count" = 1 ] \ + || fail "--force-regenerate must create exactly one archive, found $archive_count" + archive=$(find "$(dirname "$brief")" -maxdepth 1 -type f -name 'brief.md.archive-*' -print) + assert_grep "months-old draft without current safety contracts" "$archive" \ + "--force-regenerate archive did not preserve the stale brief" + pass "fm-brief.sh: stale existing briefs fail loudly and force regeneration archives before replacing" +} + +test_existing_current_brief_still_requires_freshness_verification() { + local home id brief err status + home="$TMP_ROOT/current-brief-home" + id=current-brief-guard + err="$home/refusal.err" + mkdir -p "$home/data" + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" "$id" some-proj >/dev/null 2>&1 \ + || fail "current brief fixture did not scaffold" + brief="$home/data/$id/brief.md" + assert_grep "firstmate-brief-scaffold-safety:v1" "$brief" \ + "fresh brief fixture is missing the current scaffold safety marker" + + status=0 + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" "$id" some-proj >/dev/null 2>"$err" || status=$? + expect_code 1 "$status" "scaffolding over an existing current brief must still fail" + assert_grep "current scaffold safety marker is present" "$err" \ + "existing current brief refusal did not report marker status" + assert_grep "task freshness is unverified" "$err" \ + "existing current brief refusal falsely treated its marker as freshness proof" + assert_grep "verify it intentionally, or rerun with --force-regenerate" "$err" \ + "existing current brief refusal did not give both safe recovery choices" + pass "fm-brief.sh: a current safety marker never substitutes for task freshness verification" +} + +test_force_regeneration_preserves_brief_when_rendering_fails() { + local home id brief fake_root status archive_count + home="$TMP_ROOT/failed-regeneration-home" + id=failed-regeneration-guard + brief="$home/data/$id/brief.md" + fake_root="$home/fake-root" + mkdir -p "$(dirname "$brief")" "$fake_root/bin" + printf '%s\n' 'original brief must survive a failed regeneration' > "$brief" + printf '%s\n' '#!/usr/bin/env bash' 'exit 1' > "$fake_root/bin/fm-project-mode.sh" + chmod +x "$fake_root/bin/fm-project-mode.sh" + + status=0 + FM_HOME="$home" FM_ROOT_OVERRIDE="$fake_root" \ + "$ROOT/bin/fm-brief.sh" "$id" some-proj --force-regenerate >/dev/null 2>&1 || status=$? + expect_code 1 "$status" "failed regeneration must return the render failure" + assert_grep "original brief must survive" "$brief" \ + "failed regeneration removed or changed the live brief" + archive_count=$(find "$(dirname "$brief")" -maxdepth 1 -type f -name 'brief.md.archive-*' | wc -l | tr -d ' ') + [ "$archive_count" = 0 ] || fail "failed regeneration archived the live brief before rendering" + pass "fm-brief.sh: failed force regeneration preserves the live brief" +} + # Registry with one project per delivery mode, so each ship-mode DOD branch is # exercised. A project absent from the registry defaults to no-mistakes. write_registry() { @@ -621,6 +705,9 @@ test_scout_and_secondmate_scaffold() { test_script_parses test_no_heredoc_in_command_substitution test_help_includes_entire_header +test_existing_brief_refusal_detects_staleness_and_force_regenerates +test_existing_current_brief_still_requires_freshness_verification +test_force_regeneration_preserves_brief_when_rendering_fails test_ship_modes_generate_clean_briefs test_faster_paths_use_configured_authority_without_stacked_review test_no_mistakes_dod_wording diff --git a/tests/fm-operational-input.test.sh b/tests/fm-operational-input.test.sh index 2d1b1c39de3..7e4e9b93165 100755 --- a/tests/fm-operational-input.test.sh +++ b/tests/fm-operational-input.test.sh @@ -48,6 +48,27 @@ test_current_generic_matrix() { pass "operational input: every current generic envelope retains its exact structured kind" } +test_launch_brief_cli_carries_dispatch_provenance() { + local encoded stripped + encoded=$(printf '%s' 'CREWMATE_BRIEF_BODY' | "$OWNER" encode launch-brief) \ + || fail "launch-brief CLI encoding failed" + [ "$(kind_cli "$encoded")" = launch-brief ] \ + || fail "launch-brief provenance changed the structured kind" + stripped=$(printf '%s' "$encoded" | "$OWNER" body) \ + || fail "launch-brief provenance body could not be recovered" + assert_contains "$stripped" "This is a genuine Firstmate dispatch." \ + "launch-brief did not identify itself as genuine Firstmate dispatch" + assert_contains "$stripped" "canonical operational-input encoder" \ + "launch-brief did not explain the typed envelope's owning encoder" + assert_contains "$stripped" "not project content" \ + "launch-brief did not distinguish its provenance envelope from project content" + assert_contains "$stripped" "proceed with the brief below" \ + "launch-brief did not tell the worker that proceeding is expected" + assert_contains "$stripped" "CREWMATE_BRIEF_BODY" \ + "launch-brief provenance dropped the original brief body" + pass "operational input: launch briefs explain their canonical Firstmate provenance before the task body" +} + test_current_from_firstmate_carrier() { local encoded parsed separator separator=$(printf '\342\201\243') @@ -152,6 +173,7 @@ test_invalid_current_encodings_are_rejected() { } test_current_generic_matrix +test_launch_brief_cli_carries_dispatch_provenance test_current_from_firstmate_carrier test_landed_untyped_prefix_is_explicitly_legacy test_isolated_legacy_matrix From d0a8bbc4e425ceae5079a70071cd376b5b397b6b Mon Sep 17 00:00:00 2001 From: juniorlovestmh <272474227+juniorlovestmh@users.noreply.github.com> Date: Sat, 1 Aug 2026 08:54:23 -0300 Subject: [PATCH 04/52] feat: wake firstmate for Better Stack incidents --- .agents/skills/bootstrap-diagnostics/SKILL.md | 3 + .opencode/plugins/fm-primary-watch-arm.js | 1 + AGENTS.md | 20 +- bin/fm-better-stack-incidents-poll.sh | 164 ++++++++++++ bin/fm-bootstrap.sh | 113 +++++++- bin/fm-claude-stop-autoarm.sh | 6 +- bin/fm-session-start.sh | 14 +- bin/fm-subagent-pretool-check.sh | 2 +- bin/fm-supervision-lib.sh | 16 +- bin/fm-test-run.sh | 2 +- bin/fm-turnend-guard.sh | 8 + docs/architecture.md | 3 + docs/configuration.md | 24 +- docs/subagent-guard.md | 4 +- docs/supervision-protocols/grok.md | 2 +- docs/turnend-guard.md | 8 +- tests/fm-better-stack-incidents.test.sh | 246 ++++++++++++++++++ 17 files changed, 595 insertions(+), 41 deletions(-) create mode 100755 bin/fm-better-stack-incidents-poll.sh create mode 100755 tests/fm-better-stack-incidents.test.sh diff --git a/.agents/skills/bootstrap-diagnostics/SKILL.md b/.agents/skills/bootstrap-diagnostics/SKILL.md index 477980b8df1..e9a57058a55 100644 --- a/.agents/skills/bootstrap-diagnostics/SKILL.md +++ b/.agents/skills/bootstrap-diagnostics/SKILL.md @@ -32,6 +32,9 @@ When any diagnostic needs captain attention, report the plain consequence and re - `FLEET_SYNC: <repo>: skipped: <reason>` - a benign one-off skip (offline, no origin, local-only); bootstrap continued, investigate only if it blocks work. A skip can also report the bounded fleet-refresh timeout (`FM_FLEET_SYNC_BOOTSTRAP_TIMEOUT`, or a fleet-size-aware default with a 20 second floor); a timeout never blocks startup. - `FLEET_SYNC: <repo>: recovered: <detail>` - the clone had drifted onto a clean detached HEAD holding no unique commits and the sync self-healed it (re-attached the default branch and fast-forwarded); no action needed, it is reported only so the self-heal is visible. +- `BETTER_STACK: incident monitoring on ...` - the home-scoped poll is registered at the default check cadence; no action is needed. +- `BETTER_STACK: incident monitoring off - removed ...` - the local presence flag was removed and bootstrap retired the runnable check while retaining incident dedupe state; no action is needed. +- Any other `BETTER_STACK:` line - follow its concrete dependency, unsafe flag, activation, or cleanup diagnostic before relying on incident monitoring. - `FLEET_SYNC: <repo>: STUCK: on <state>, N commits behind <base> - needs attention` - the clone is dirty, on a non-default branch, detached with unique commits, or diverged, so the sync left it untouched (never forcing or discarding); it will keep falling behind until you look. A loud STUCK, especially a growing N across bootstraps, means that clone needs hands-on attention; dispatch a crewmate or resolve it before it strands work. - `PR_CHECK_MIGRATION: canonical polls rebuilt and armed; resume supervision for this home` - the non-executing migration rebuilt canonical task polls from validated metadata, and those polls are already armed. diff --git a/.opencode/plugins/fm-primary-watch-arm.js b/.opencode/plugins/fm-primary-watch-arm.js index 433edb80ab4..a8ae265ed2e 100644 --- a/.opencode/plugins/fm-primary-watch-arm.js +++ b/.opencode/plugins/fm-primary-watch-arm.js @@ -103,6 +103,7 @@ async function isPrimaryRoot(root, home) { function shouldArm(paths) { if (existsSync(`${paths.state}/.afk`)) return false; if (existsSync(`${paths.config}/x-mode.env`)) return true; + if (existsSync(`${paths.state}/better-stack-incidents.check.sh`)) return true; try { return readdirSync(paths.state).some((name) => name.endsWith(".meta")); } catch { diff --git a/AGENTS.md b/AGENTS.md index 7467d4ee2b4..0d9c48c219b 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -74,6 +74,7 @@ config/startup-memory-budget primary-authoritative per-home startup-memory b config/herdr-presentation-spaces optional presence flag for Herdr's default-off disposable single-task visual projection; LOCAL, gitignored; inherited by secondmate homes; see docs/herdr-backend.md "Optional presentation spaces" config/cmux-socket-password optional cmux control-socket password; LOCAL, gitignored; read fresh on every cmux CLI call and passed through without ever overriding an operator's own ambient CMUX_SOCKET_PASSWORD when absent (docs/cmux-backend.md "Setup") config/wedge-alarm optional away-mode wedge-alarm active-alert directives; LOCAL, gitignored; absent means auto (macOS Notification Center when available); see docs/wedge-alarm.md +config/better-stack-incidents optional presence flag for the home-scoped Better Stack incident poll; LOCAL, gitignored, and not inherited; see docs/configuration.md "Better Stack incident monitoring" config/x-mode.env generated X-mode watcher cadence; LOCAL, gitignored; source before arming watcher when present data/ personal fleet records; LOCAL, gitignored as a whole backlog.md task queue, dependencies, history @@ -101,6 +102,8 @@ state/ volatile runtime signals; gitignored .pr-check-migration.log private per-task outcomes distinguishing rebuilt or canonically registered replacement polls, quarantined unarmed polls, and incomplete migrations .pr-check-migration-scan-v1 private marker proving the non-executing scan disabled every unsafe legacy check; .pr-check-migration-v1 separately records completed private repairs x-watch.check.sh generated X-mode relay poll shim; present only when opted in (section 14) + better-stack-incidents.check.sh better-stack-incidents.check-trust generated and registered home-scoped Better Stack incident poll; present only when opted in + better-stack-incidents.seen/ better-stack-incidents.diagnostics/ private incident-ID and diagnostic dedupe state retained across poll disable/re-enable pending-replies/ parent-owned secondmate pending-reply records (correlation id, delivery vs reply, recovery, escalation); fm-pending-reply-lib.sh x-inbox/ generated X-mode pending mention payloads; fmx-respond drains it (section 14) x-context/ generated X-mode durable per-request reply context and one-wake offer markers, keyed by request_id; survives inbox cleanup and expires within seven days (section 14; bin/fm-x-lib.sh) @@ -138,7 +141,7 @@ A lock-refused session must not spawn, steer, merge, drain the wake queue, repai 1. **Lock** - acquires the per-home session lock first, before anything mutates shared state. 2. **Bootstrap** - detect-only checks (tool/version problems, GitHub auth, the worktree-tangle check, harness override, dispatch-profile validation, backlog-backend status) always run, but routine confirmations stay silent by default. When the lock could not be acquired, the worktree-tangle check uses read-only advisory wording without a checkout repair command. - Home-local stale Herdr projection cleanup and the five bootstrap MUTATING sweeps - non-executing legacy PR-check migration, fleet sync, the local secondmate fast-forward sweep, the secondmate liveness sweep, and X-mode artifact writes - run only when this session actually holds the lock from step 1. + Home-local stale Herdr projection cleanup and the six bootstrap MUTATING sweeps - non-executing legacy PR-check migration, fleet sync, the local secondmate fast-forward sweep, the secondmate liveness sweep, X-mode artifact writes, and Better Stack incident-poll registration - run only when this session actually holds the lock from step 1. The secondmate liveness sweep deterministically accounts for every registered secondmate: it relaunches only from the recovery-grade `dead` or `missing` states, preserves ambiguous or unreadable targets, and reports skipped or failed guarantees as `SECONDMATE_LIVENESS:` lines (`bin/fm-bootstrap.sh`; `bin/fm-backend.sh`'s `fm_backend_agent_state`). 3. **Wake queue** - when locked, drains the durable wake queue and prints the raw records prominently as this turn's first work queue; a bounded, clearly labeled historical status-event annotation may follow a valid `signal` record but never replaces it or current-state reconciliation, and a lapsed watcher chain still surfaces here via the same guard alarm. When the lock could not be acquired and verified, the queue is left untouched because no session mutation is authorized, and the guard's tangle/watcher-liveness alarms still print in read-only advisory mode without drain, supervision repair, or checkout repair commands. @@ -147,7 +150,7 @@ A lock-refused session must not spawn, steer, merge, drain the wake queue, repai 5. **Fleet-state digest** - the compact backlog listing owned by `bin/fm-session-start.sh`; every `state/<id>.meta`; a bounded tail of each task's `state/<id>.status` (labeled as wake-EVENT history, not current state, with the full log path printed for a deeper read); the `state/.afk` flag; and one cheap alive/dead read of each task's recorded backend endpoint. That liveness line is a fast presence check only, not a full state read - when you need a crew's actual current state (a run-step, not just "is the pane there"), read it with `bin/fm-crew-state.sh <id>` as before; the digest deliberately skips that deeper, slower read for every task so it stays fast and bounded. 6. **Supervision operating instructions and next step** - after the wake queue and before context, the digest emits exactly one operating block for the detected primary harness. - The closing reminder points back to that emitted block and preserves only the lock, afk, X-mode, and read-once reminders. + The closing reminder points back to that emitted block and preserves only the lock, afk, home-monitoring, and read-once reminders. The script itself never starts supervision; the emitted harness protocol owns the exact wait or wake mechanism. Bootstrap detects first, asks for consent, and installs only after the captain approves in the current session. @@ -337,7 +340,7 @@ The promoted worker must inventory scratch state, return to a clean default-bran Fleet supervision is an always-loaded operational contract; `docs/architecture.md`, `docs/turnend-guard.md`, the emitted session-start block, and script help own mechanisms and harness-specific recipes. Whenever work is under way, keep exactly one live supervision cycle using the emitted protocol for this primary harness. -X mode may require that same live cycle with no fleet work. +X mode or Better Stack incident monitoring may require that same live cycle with no fleet work. Do not substitute another harness's wait shape, use shell `&`, or create a second cycle when a healthy one already exists. For every actionable wake, follow the ordinary-wake continuation in the emitted protocol; use its repair action only when the live cycle is missing or failed. No turn ends blind while work is under way, including turns described as holding or waiting. @@ -351,9 +354,14 @@ Handle actionable wakes as follows: 1. For `signal:`, read the listed event lines first, then reconcile current state only where action depends on it. 2. For `stale:`, inspect the recorded endpoint and load `stuck-crewmate-recovery` for a stopped, looping, confused, or unresponsive worker; a deep-inspection reason also requires current-state and validation-log inspection. -3. For `check:`, act on the named poll result, including merges and X-mode events. +3. For `check:`, act on the named poll result, including merges, X-mode events, and Better Stack incidents or diagnostics. 4. For `heartbeat:`, review the whole fleet from the structured fleet view, reconcile suspicious tasks and PR state, update the backlog, and never report an unchanged fleet as progress. +For a `better-stack-incident opened ...` or `better-stack-incidents opened ...` result, load `diagnostic-reasoning` before scoping the response. +When the delivery ladder caused the breakage, restore service through the available rollback path first and investigate after recovery. +Page the captain immediately only for security-shaped, irreversible, or product-affecting incidents; otherwise carry the result and resolution in the next outcome digest. +Better Stack polling during `heartbeat:` handling was an interim practice and is retired; the registered home check is its only poll owner. + When any wake reports a merged PR for a project cloned in this home, refresh that clone through the guarded fleet-sync path. When X-linked work reaches a milestone or terminal state, load `fmx-respond`; before terminal teardown, always post the final completion follow-up so the link clears even if earlier follow-ups were spent. @@ -476,7 +484,7 @@ It performs guarded fast-forward updates of firstmate and registered secondmate These skills are not captain-invocable; load them only at their precise triggers. -- `bootstrap-diagnostics` - load whenever the session-start digest's bootstrap section prints an actionable diagnostic line (`MISSING:`, `MISSING_MANUAL:`, `BACKEND_INVALID:`, `NEEDS_GH_AUTH`, `TANGLE:`, `STARTUP_MEMORY_BUDGET:`, `CREW_DISPATCH: invalid`, `FLEET_SYNC:`, `PR_CHECK_MIGRATION:`, `SECONDMATE_SYNC:`, `SECONDMATE_LIVENESS:`, `NUDGE_SECONDMATES:`, or `FMX:`); silence and `BOOTSTRAP_INFO:` need no load. +- `bootstrap-diagnostics` - load whenever the session-start digest's bootstrap section prints an actionable diagnostic line (`MISSING:`, `MISSING_MANUAL:`, `BACKEND_INVALID:`, `NEEDS_GH_AUTH`, `TANGLE:`, `STARTUP_MEMORY_BUDGET:`, `CREW_DISPATCH: invalid`, `FLEET_SYNC:`, `PR_CHECK_MIGRATION:`, `SECONDMATE_SYNC:`, `SECONDMATE_LIVENESS:`, `NUDGE_SECONDMATES:`, `FMX:`, or `BETTER_STACK:`); silence and `BOOTSTRAP_INFO:` need no load. - `diagnostic-reasoning` - load before scoping a reported bug and before acting on a diagnostic report. - `ask-user-authority` - load before deciding any ask-user finding, regardless of the project's `yolo` posture. - `quota-array-dispatch` - load before choosing among a matched crew-dispatch profile array from current quota-axi output. @@ -498,7 +506,7 @@ X mode ships inert and causes no behavior change until the home opts in by placi That token is consent for public replies and normal reversible lifecycle actions from eligible mentions, not authority for destructive, irreversible, or security-sensitive action; those still require trusted-channel confirmation. `docs/configuration.md` owns activation, generated state, cadence, wire protocol, and opt-out mechanics. -An X-only home still requires the live supervision cycle so mentions can wake it without fleet work. +A home with X mode or Better Stack incident monitoring still requires the live supervision cycle without fleet work. On an `x-mention <request_id>` or `x-mode-error ...` check wake, load `fmx-respond`, which owns classification, public-safety policy, reply or dismissal, task linking, and follow-ups. For every X-linked terminal outcome, load that owner and post the final completion follow-up before teardown, regardless of earlier milestone follow-ups. diff --git a/bin/fm-better-stack-incidents-poll.sh b/bin/fm-better-stack-incidents-poll.sh new file mode 100755 index 00000000000..e7085bbde4e --- /dev/null +++ b/bin/fm-better-stack-incidents-poll.sh @@ -0,0 +1,164 @@ +#!/usr/bin/env bash +# Poll Better Stack for unresolved incidents through the fleet-observability/prd +# Doppler config. +# +# Usage: fm-better-stack-incidents-poll.sh +# +# The public entrypoint always launches itself through Doppler with fallback +# files disabled and only BETTER_STACK_API_TOKEN injected. +# The internal --from-doppler mode is used only by that child process and tests. +# +# Output is the authenticated custom-check contract consumed by fm-watch.sh: +# better-stack-incident opened id=<id> name=<name> started=<timestamp> +# better-stack-incidents opened ids=<comma-separated ids> +# better-stack-error <deduplicated diagnostic> +# A quiet or already-seen result prints nothing. +# The watcher provides the outer FM_CHECK_TIMEOUT; curl stays within five +# seconds so the check finishes with margin. +set -u + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" +FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" +STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" +ERROR_DIR="$STATE/better-stack-incidents.diagnostics" +ERROR_FILE="$ERROR_DIR/error" +SEEN_DIR="$STATE/better-stack-incidents.seen" + +# Reuse the watcher's existing private-artifact owner rather than introducing a +# second atomic-publication contract for one extension. +# shellcheck source=bin/fm-x-lib.sh disable=SC1091 +. "$SCRIPT_DIR/fm-x-lib.sh" + +emit_error_once() { + local msg=$1 + if fmx_private_artifact_file_valid "$ERROR_DIR" error 600 \ + && [ "$(cat "$ERROR_FILE" 2>/dev/null)" = "$msg" ]; then + return 0 + fi + printf '%s\n' "$msg" \ + | fmx_private_artifact_publish_stdin "$ERROR_DIR" error 600 2>/dev/null || true + printf 'better-stack-error %s\n' "$msg" +} + +clear_error() { + fmx_private_artifact_file_valid "$ERROR_DIR" error 600 || return 0 + rm -f -- "$ERROR_FILE" 2>/dev/null || true +} + +run_through_doppler() { + local out rc + command -v doppler >/dev/null 2>&1 \ + || { emit_error_once "missing doppler"; return 0; } + out=$(doppler run \ + --silent \ + --no-check-version \ + --no-fallback \ + --project fleet-observability \ + --config prd \ + --only-secrets BETTER_STACK_API_TOKEN \ + -- "$SCRIPT_DIR/fm-better-stack-incidents-poll.sh" --from-doppler 2>/dev/null) + rc=$? + if [ "$rc" -ne 0 ]; then + emit_error_once "Doppler access unavailable for fleet-observability/prd" + return 0 + fi + case "$out" in + '') return 0 ;; + *$'\n'*) emit_error_once "poll returned invalid multiline output" ;; + better-stack-incident\ opened\ *|better-stack-incidents\ opened\ *|better-stack-error\ *) + printf '%s\n' "$out" + ;; + *) emit_error_once "poll returned invalid output" ;; + esac +} + +claim_incident() { + local id=$1 rc + printf 'seen\n' \ + | fmx_private_artifact_publish_stdin_once "$SEEN_DIR" "$id" 600 2>/dev/null + rc=$? + return "$rc" +} + +poll_with_injected_token() { + local token=${BETTER_STACK_API_TOKEN:-} raw code body rows id name started + local claim_rc new_count=0 new_ids= first_id= first_name= first_started= + + [ -n "$token" ] || { emit_error_once "missing BETTER_STACK_API_TOKEN"; return 0; } + [[ "$token" =~ ^[A-Za-z0-9._~+/=-]+$ ]] \ + || { emit_error_once "invalid BETTER_STACK_API_TOKEN"; return 0; } + command -v curl >/dev/null 2>&1 || { emit_error_once "missing curl"; return 0; } + command -v jq >/dev/null 2>&1 || { emit_error_once "missing jq"; return 0; } + + raw=$(printf 'header = "Authorization: Bearer %s"\n' "$token" \ + | curl \ + --config - \ + --request GET \ + --url 'https://uptime.betterstack.com/api/v3/incidents?resolved=false&per_page=50' \ + --header 'Accept: application/json' \ + --connect-timeout 3 \ + --max-time 5 \ + --silent \ + --show-error \ + --write-out '\n%{http_code}' 2>/dev/null) \ + || { emit_error_once "Better Stack API unreachable"; return 0; } + case "$raw" in + *$'\n'*) ;; + *) emit_error_once "Better Stack API returned no status"; return 0 ;; + esac + code=${raw##*$'\n'} + body=${raw%$'\n'*} + [ "$code" = 200 ] || { emit_error_once "API returned HTTP $code"; return 0; } + + rows=$(printf '%s' "$body" | jq -r ' + if (.data | type) != "array" then error("data must be an array") else .data[] end + | select(.type == "incident") + | select(.attributes.resolved_at == null) + | select(.id | type == "string" and test("^[0-9]+$")) + | [ + .id, + ((.attributes.name // "unknown") | tostring | gsub("[[:space:][:cntrl:]]+"; " ") | .[0:120]), + ((.attributes.started_at // "unknown") | tostring | gsub("[[:space:][:cntrl:]]+"; " ") | .[0:64]) + ] + | @tsv + ' 2>/dev/null) || { emit_error_once "invalid Better Stack API response"; return 0; } + + while IFS=$'\t' read -r id name started; do + [ -n "$id" ] || continue + claim_incident "$id" + claim_rc=$? + case "$claim_rc" in + 0) + new_count=$((new_count + 1)) + if [ "$new_count" -eq 1 ]; then + first_id=$id + first_name=$name + first_started=$started + new_ids=$id + else + new_ids="$new_ids,$id" + fi + ;; + 1) ;; + *) emit_error_once "cannot record incident dedupe state"; return 0 ;; + esac + done <<< "$rows" + + clear_error + case "$new_count" in + 0) ;; + 1) printf 'better-stack-incident opened id=%s name=%s started=%s\n' \ + "$first_id" "$first_name" "$first_started" ;; + *) printf 'better-stack-incidents opened ids=%s\n' "$new_ids" ;; + esac +} + +case "${1:-}" in + '') run_through_doppler ;; + --from-doppler) + [ "$#" -eq 1 ] || { emit_error_once "invalid poll invocation"; exit 0; } + poll_with_injected_token + ;; + *) emit_error_once "invalid poll invocation" ;; +esac diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index 16102adfa45..66e4406e4dc 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -17,7 +17,9 @@ # "NUDGE_SECONDMATES: secondmate <id>: send failed: <reason>", # "BOOTSTRAP_INFO: nudged fm-<id> with '<message>'", # "SECONDMATE_LIVENESS: secondmate <id>: skipped: <reason>|respawn failed after <cause>: <reason>", -# "FMX: X mode on ..." or "FMX: X mode off ...". +# "FMX: X mode on ..." or "FMX: X mode off ...", +# "BETTER_STACK: incident monitoring on ..." or +# "BETTER_STACK: incident monitoring off ...". # When a RUNNING secondmate worktree is fast-forwarded to firstmate's # own current default-branch commit (a purely LOCAL fast-forward, never # an origin fetch) AND its loaded instruction surface (AGENTS.md, bin/, @@ -73,15 +75,16 @@ # refresh relays any completed fm-fleet-sync.sh output before the # aggregate timeout skip line with timeout and elapsed seconds. # Set FM_FLEET_PRUNE=0 to skip branch pruning during that refresh. -# Set FM_BOOTSTRAP_DETECT_ONLY=1 to skip the five MUTATING sweeps +# Set FM_BOOTSTRAP_DETECT_ONLY=1 to skip the six MUTATING sweeps # (PR-check migration, secondmate_sync, secondmate_liveness_sweep, -# x_mode_setup, fleet_sync) while still printing every read-only detect line +# x_mode_setup, better_stack_incidents_setup, fleet_sync) while still +# printing every read-only detect line # above; the TANGLE line switches to advisory-only wording with no # checkout command. Used by # fm-session-start.sh's read-only path when another live session holds # the fleet lock, so a second concurrent session never race-mutates -# PR-check artifacts, secondmate homes, X-mode artifacts, project -# clones, or repair instructions. +# PR-check artifacts, secondmate homes, X-mode or Better Stack poll +# artifacts, project clones, or repair instructions. # Unset/0 (the default) runs every sweep exactly as before - this flag # is purely additive. # fm-bootstrap.sh install <tool>... @@ -107,6 +110,10 @@ DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" . "$SCRIPT_DIR/fm-startup-memory-budget-lib.sh" # shellcheck source=bin/fm-x-lib.sh disable=SC1091 . "$SCRIPT_DIR/fm-x-lib.sh" +# shellcheck source=bin/fm-pr-lib.sh disable=SC1091 +. "$SCRIPT_DIR/fm-pr-lib.sh" +# shellcheck source=bin/fm-check-lib.sh disable=SC1091 +. "$SCRIPT_DIR/fm-check-lib.sh" # shellcheck source=bin/fm-backend.sh disable=SC1091 . "$SCRIPT_DIR/fm-backend.sh" @@ -503,6 +510,7 @@ install_cmd() { manual_install_url() { case "$1" in + doppler) echo "https://docs.doppler.com/docs/install-cli" ;; herdr) echo "https://herdr.dev" ;; *) return 1 ;; esac @@ -557,7 +565,7 @@ no_mistakes_compatible() { [ "$patch" -ge "$NO_MISTAKES_MIN_PATCH" ] } -x_mode_write_if_changed() { +bootstrap_write_if_changed() { local dest=$1 content=$2 mode=$3 parent tmp parent_device current_mode parent=${dest%/*} [ "$parent" != "$dest" ] || return 1 @@ -578,7 +586,7 @@ x_mode_write_if_changed() { return 0 fi fi - tmp=$(umask 077; mktemp "$parent/.fm-x-mode.XXXXXX" 2>/dev/null) || return 1 + tmp=$(umask 077; mktemp "$parent/.fm-bootstrap-artifact.XXXXXX" 2>/dev/null) || return 1 if ! printf '%s\n' "$content" > "$tmp" \ || ! chmod "$mode" "$tmp" \ || ! fmx_single_link_file_mode_valid "$tmp" "$mode" "$parent_device"; then @@ -699,7 +707,7 @@ x_mode_setup() { ;; esac shim_body=$(fmx_poll_shim_content "$shim_home" "$FM_ROOT") - x_mode_write_if_changed "$shim" "$shim_body" 700 || { fmx_arm_failed; return 0; } + bootstrap_write_if_changed "$shim" "$shim_body" 700 || { fmx_arm_failed; return 0; } fmx_poll_shim_valid "$shim" "$shim_home" "$FM_ROOT" \ || { fmx_arm_failed; return 0; } @@ -711,11 +719,97 @@ x_mode_setup() { export FM_CHECK_INTERVAL=30 EOF ) - x_mode_write_if_changed "$cadence" "$cadence_body" 600 || { fmx_arm_failed; return 0; } + bootstrap_write_if_changed "$cadence" "$cadence_body" 600 || { fmx_arm_failed; return 0; } echo "FMX: X mode on - relay poll armed via state/x-watch.check.sh; 30s watcher cadence in config/x-mode.env" } +# Better Stack incident monitoring is an explicit home-local opt-in. +# A presence flag at config/better-stack-incidents materializes one ordinary +# registered custom check, so the existing hash-bound snapshot execution and +# FM_CHECK_TIMEOUT contract remain the only slow-check mechanism. +# The poll itself performs runtime-only Doppler injection and owns incident and +# diagnostic dedupe in this home's private state. +better_stack_incidents_setup() { + local flag check trust check_body tool missing check_home failed + flag="$CONFIG/better-stack-incidents" + check="$STATE/better-stack-incidents.check.sh" + trust="$STATE/better-stack-incidents.check-trust" + + better_stack_remove_artifacts() { + local remove_failed=0 + x_mode_remove_artifact "$check" || remove_failed=1 + x_mode_remove_artifact "$trust" || remove_failed=1 + [ "$remove_failed" -eq 0 ] + } + + if [ ! -e "$flag" ] && [ ! -L "$flag" ]; then + if x_mode_artifact_present "$check" || x_mode_artifact_present "$trust"; then + if better_stack_remove_artifacts; then + echo "BETTER_STACK: incident monitoring off - removed the home-scoped poll" + else + echo "BETTER_STACK: incident monitoring off - failed to remove the home-scoped poll" + fi + fi + return 0 + fi + if [ ! -f "$flag" ] || [ -L "$flag" ] || [ "$(fm_pr_file_link_count "$flag")" != 1 ]; then + better_stack_remove_artifacts || true + echo "BETTER_STACK: incident monitoring off - config/better-stack-incidents must be an ordinary file" + return 0 + fi + + missing=0 + for tool in doppler curl jq; do + if ! command -v "$tool" >/dev/null 2>&1; then + missing_tool_diagnostic "$tool" + missing=1 + fi + done + if [ "$missing" -ne 0 ]; then + better_stack_remove_artifacts || true + echo "BETTER_STACK: incident monitoring off - install the reported poll dependencies and rerun bootstrap" + return 0 + fi + + failed=0 + mkdir -p "$STATE" 2>/dev/null || failed=1 + if [ "$failed" -eq 0 ]; then + case "$FM_HOME" in + /*) check_home=$FM_HOME ;; + *) + check_home=$(CDPATH='' cd -- "$FM_HOME" 2>/dev/null && pwd -P) || failed=1 + ;; + esac + fi + if [ "$failed" -eq 0 ]; then + check_body=$(printf '%s\n' \ + '#!/usr/bin/env bash' \ + '# Auto-generated by fm-bootstrap.sh - Better Stack incident custom check.' \ + '# Registered bytes call the tracked poll; output becomes a check: wake.' \ + "export FM_HOME=$(printf '%q' "$check_home")" \ + "exec $(printf '%q' "$FM_ROOT/bin/fm-better-stack-incidents-poll.sh")") + bootstrap_write_if_changed "$check" "$check_body" 700 || failed=1 + fi + if [ "$failed" -eq 0 ] && ! fm_custom_check_registered "$STATE" better-stack-incidents; then + FM_HOME="$FM_HOME" FM_STATE_OVERRIDE="$STATE" FM_ROOT_OVERRIDE="$FM_ROOT" \ + "$SCRIPT_DIR/fm-check-register.sh" better-stack-incidents >/dev/null 2>&1 || failed=1 + fi + if [ "$failed" -eq 0 ]; then + fm_custom_check_registered "$STATE" better-stack-incidents || failed=1 + fi + if [ "$failed" -ne 0 ]; then + if better_stack_remove_artifacts; then + echo "BETTER_STACK: incident monitoring off - failed to arm the home-scoped poll" + else + echo "BETTER_STACK: incident monitoring off - failed to arm the home-scoped poll; stale artifacts remain" + fi + return 0 + fi + + echo "BETTER_STACK: incident monitoring on - registered state/better-stack-incidents.check.sh at the default 300s check cadence" +} + crew_dispatch_validate() { local file err file="$CONFIG/crew-dispatch.json" @@ -900,6 +994,7 @@ if [ "${FM_BOOTSTRAP_DETECT_ONLY:-0}" != 1 ]; then secondmate_liveness_sweep secondmate_sync x_mode_setup + better_stack_incidents_setup fleet_sync fi exit 0 diff --git a/bin/fm-claude-stop-autoarm.sh b/bin/fm-claude-stop-autoarm.sh index df9ee1128fc..0c2c4fbf528 100755 --- a/bin/fm-claude-stop-autoarm.sh +++ b/bin/fm-claude-stop-autoarm.sh @@ -18,8 +18,8 @@ # - AFK: while state/.afk exists the away daemon owns the watcher and triage; # this hook exits 0 and NEVER rewakes the primary (checked again at # translation time so a mid-cycle AFK transition is honored). -# - Need: arms only while work is in flight (state/*.meta) or X mode has a -# relay poll to run (state/x-watch.check.sh); an idle home exits 0. +# - Need: arms only while work is in flight (state/*.meta) or a home-level +# X-mode or Better Stack poll is active; an idle home exits 0. # - Single-flight: Claude does not dedupe async hooks, so a home-scoped owner # lock (state/.claude-autoarm.lock) admits exactly one owner; every other # concurrent firing exits 0 without translating, which keeps one event @@ -89,7 +89,7 @@ fi # --- AFK: the away daemon owns the watcher and triage; never rewake ---------- [ -e "$STATE/.afk" ] && exit 0 -# --- need: in-flight work or an X-mode relay poll ---------------------------- +# --- need: in-flight work or a home-level poll ------------------------------- need_supervision() { fm_supervision_needed "$STATE" "$GRACE" } diff --git a/bin/fm-session-start.sh b/bin/fm-session-start.sh index 1abbace4bf1..85ace7fa29f 100755 --- a/bin/fm-session-start.sh +++ b/bin/fm-session-start.sh @@ -18,7 +18,7 @@ # standalone with unchanged default behavior - other flows (fm-bootstrap.sh # install <tools> after consent, /updatefirstmate, the afk daemon, existing # tests) still call them directly. The one seam this script needed - -# bootstrap running its detect-only diagnostics without its five mutating +# bootstrap running its detect-only diagnostics without its six mutating # sweeps - is an opt-in FM_BOOTSTRAP_DETECT_ONLY=1 flag on fm-bootstrap.sh # itself (default unset/0 = unchanged behavior), not a fork. # @@ -29,9 +29,10 @@ # mutating step runs. # 2. bootstrap - home-local stale Herdr projection cleanup runs only # when this session actually holds the lock. Detect-only -# diagnostics always run. Bootstrap's five MUTATING sweeps +# diagnostics always run. Bootstrap's six MUTATING sweeps # (legacy PR-check migration, secondmate fast-forward, -# secondmate liveness, X-mode artifact writes, fleet sync) +# secondmate liveness, X-mode artifact writes, Better Stack +# incident-poll registration, fleet sync) # also run only when locked. # 3. wake-drain - mutates the durable wake queue, so it also only runs # when locked. @@ -52,7 +53,8 @@ # # Why lock first: the old documented order (bootstrap, THEN lock) let a # SECOND concurrent session run bootstrap's mutating sweeps - fast-forwarding -# secondmate homes, writing X-mode artifacts, fetching/fast-forwarding every +# secondmate homes, writing X-mode and Better Stack poll artifacts, +# fetching/fast-forwarding every # project clone - before ever discovering another session already holds the # lock. Two sessions racing those sweeps is exactly the hazard the lock # exists to prevent, so locking first closes the hole outright: only the @@ -65,7 +67,7 @@ # tasks-axi and quota-axi tool checks, and tasks-axi availability - none of # which mutate shared state and all of which are safe to compute without # verified lock ownership. -# Only projection cleanup, the five bootstrap mutating sweeps, and the +# Only projection cleanup, the six bootstrap mutating sweeps, and the # wake-queue drain are skipped. # The context and fleet-state digests # below are always read-only, so they run unconditionally in both modes. @@ -257,7 +259,7 @@ if [ "$LOCK_RC" -ne 0 ]; then printf '● READ-ONLY SESSION - FLEET LOCK OWNERSHIP WAS NOT VERIFIED\n' printf '● %s\n' "$LOCK_OUT" printf '● Skipping every mutating step: PR-check migration, stale Herdr child cleanup,\n' - printf '● secondmate sync, X-mode artifacts, fleet sync, and wake-queue drain. Detect-only bootstrap\n' + printf '● secondmate sync, X-mode and Better Stack poll artifacts, fleet sync, and wake-queue drain. Detect-only bootstrap\n' printf '● diagnostics and the rest of this read-only-safe digest still ran below.\n' printf '● Operate read-only until this resolves - do not spawn, steer, merge, or\n' printf '● otherwise mutate fleet state from this session.\n' diff --git a/bin/fm-subagent-pretool-check.sh b/bin/fm-subagent-pretool-check.sh index 8edb507218b..95a0c6be434 100755 --- a/bin/fm-subagent-pretool-check.sh +++ b/bin/fm-subagent-pretool-check.sh @@ -6,7 +6,7 @@ # no `data/<id>/brief.md`. Only `bin/fm-spawn.sh` writes that metadata, and # untracked project work contributes nothing to the in-flight branch of # bin/fm-supervision-lib.sh or bin/fm-turnend-guard.sh. So such work is not -# merely unsupervised: absent an independent X-mode need, it makes the whole +# merely unsupervised: absent an independent home-monitoring need, it makes the whole # guard stack structurally inert, and it dies with the primary session instead # of living in its own backend session. # diff --git a/bin/fm-supervision-lib.sh b/bin/fm-supervision-lib.sh index 1930700d2af..b7ff73d0158 100644 --- a/bin/fm-supervision-lib.sh +++ b/bin/fm-supervision-lib.sh @@ -3,9 +3,9 @@ # Usage: . bin/fm-supervision-lib.sh # # Reports whether a firstmate home needs supervision because it has in-flight -# work (a state/<id>.meta exists) or an X-mode relay poll -# (state/x-watch.check.sh), and whether its watcher has a fresh liveness beacon -# (state/.last-watcher-beat, touched every poll cycle, within the grace window). +# work (a state/<id>.meta exists) or a home-level poll (X mode or Better Stack), +# and whether its watcher has a fresh liveness beacon (state/.last-watcher-beat, +# touched every poll cycle, within the grace window). # bin/fm-guard.sh keeps its task-specific grace-based warning predicate; # bin/fm-turnend-guard.sh uses the status fields here for its banner but performs # its end-of-turn block decision with the live watcher lock check in @@ -23,7 +23,7 @@ fm_sup_stat_mtime() { # fm_supervision_status <state-dir> [grace-seconds] # Populates, for the state dir at $1: # FM_SUP_IN_FLIGHT count of state/*.meta (in-flight tasks) -# FM_SUP_NEEDED true/false - in-flight work or an X-mode relay poll +# FM_SUP_NEEDED true/false - in-flight work or a home-level poll # FM_SUP_WATCHER_FRESH true/false - a watcher beacon within the grace window # FM_SUP_BEACON_DESC human-readable beacon age, for banners ("never" if absent) # FM_SUP_QUEUE_PENDING true/false - state/.wake-queue has unread records @@ -41,7 +41,9 @@ fm_supervision_status() { [ -e "$meta" ] || continue FM_SUP_IN_FLIGHT=$((FM_SUP_IN_FLIGHT + 1)) done - if [ "$FM_SUP_IN_FLIGHT" -gt 0 ] || [ -f "$state/x-watch.check.sh" ]; then + if [ "$FM_SUP_IN_FLIGHT" -gt 0 ] \ + || [ -f "$state/x-watch.check.sh" ] \ + || [ -f "$state/better-stack-incidents.check.sh" ]; then FM_SUP_NEEDED=true fi @@ -64,8 +66,8 @@ fm_supervision_status() { } # fm_supervision_needed <state-dir> [grace-seconds] -# Exit 0 (true) exactly when in-flight work or an X-mode relay poll needs a -# watcher. Exit 1 (false) for an idle home. +# Exit 0 (true) exactly when in-flight work or a home-level poll needs a watcher. +# Exit 1 (false) for an idle home. fm_supervision_needed() { fm_supervision_status "$@" [ "$FM_SUP_NEEDED" = true ] diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index f0b7f5537a3..994eb0eb1b9 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -131,7 +131,7 @@ family_for_basename() { fm-test-run.test.sh|fm-test-isolation-proof.test.sh) printf '%s\n' pure-contract-unit ;; - fm-daemon.test.sh|fm-guard-stale-banner.test.sh|fm-pi-watch-extension.test.sh|\ + fm-better-stack-incidents.test.sh|fm-daemon.test.sh|fm-guard-stale-banner.test.sh|fm-pi-watch-extension.test.sh|\ fm-supervision-events.test.sh|fm-turnend-guard.test.sh|fm-wake-daemon-lifecycle-e2e.test.sh|\ fm-wake-queue.test.sh|fm-watch-checkpoint.test.sh|fm-watch-triage.test.sh|\ fm-watcher-lock.test.sh) diff --git a/bin/fm-turnend-guard.sh b/bin/fm-turnend-guard.sh index 2e96fb33e48..d9bdc72fff6 100755 --- a/bin/fm-turnend-guard.sh +++ b/bin/fm-turnend-guard.sh @@ -164,6 +164,10 @@ block_stop() { printf '● TURN WOULD END BLIND - SUPERVISION IS OFF\n' if [ "$FM_SUP_IN_FLIGHT" -gt 0 ]; then printf '● %s task(s) in flight, but no live watcher holds this home lock (last beat: %s).\n' "$FM_SUP_IN_FLIGHT" "$FM_SUP_BEACON_DESC" + elif [ -f "$STATE/better-stack-incidents.check.sh" ] && [ -f "$STATE/x-watch.check.sh" ]; then + printf '● X-mode relay and Better Stack incident monitoring need supervision, but no live watcher holds this home lock (last beat: %s).\n' "$FM_SUP_BEACON_DESC" + elif [ -f "$STATE/better-stack-incidents.check.sh" ]; then + printf '● Better Stack incident monitoring needs supervision, but no live watcher holds this home lock (last beat: %s).\n' "$FM_SUP_BEACON_DESC" else printf '● X-mode relay polling needs supervision, but no live watcher holds this home lock (last beat: %s).\n' "$FM_SUP_BEACON_DESC" fi @@ -229,6 +233,10 @@ if [ "$COUNT" -gt "$BLOCK_BUDGET" ]; then budget_reset if [ "$FM_SUP_IN_FLIGHT" -gt 0 ]; then NEED_DESC="$FM_SUP_IN_FLIGHT task(s) in flight" + elif [ -f "$STATE/better-stack-incidents.check.sh" ] && [ -f "$STATE/x-watch.check.sh" ]; then + NEED_DESC="X-mode relay and Better Stack incident monitoring active" + elif [ -f "$STATE/better-stack-incidents.check.sh" ]; then + NEED_DESC="Better Stack incident monitoring active" else NEED_DESC="X-mode relay polling active" fi diff --git a/docs/architecture.md b/docs/architecture.md index bf8b5cb3ec1..4e7e82e661c 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -10,6 +10,9 @@ firstmate's always-loaded operating contract and routing index for conditional p A zero-token bash watcher (`bin/fm-watch.sh`) sleeps on the fleet, classifies detected wakes in bash, and wakes the first mate only when something is actionable. Actionable wakes include captain-relevant status signals, no-verb signals whose crew is not provably working, authenticated check output such as PR merge polling or an X-mode mention, stale panes whose crew is not provably working whether their status log looks terminal or non-terminal, provably-working stale panes that persist past `FM_STALE_ESCALATE_SECS`, declared external waits that remain paused past `FM_PAUSE_RESURFACE_SECS`, and heartbeat backstop hits. +Better Stack incident monitoring uses the registered custom-check extension rather than a new watcher scheduler because that existing path already supplies home scoping, hash-bound private snapshot execution, the slow-check timeout, and durable `check:` delivery. +The locked bootstrap materializes the check only for a home carrying `config/better-stack-incidents`, and [`configuration.md`](configuration.md#better-stack-incident-monitoring-configbetter-stack-incidents) owns runtime Doppler injection, incident-ID dedupe, diagnostic suppression, cadence, and opt-out mechanics. +The registered check remains a supervision need when the project fleet is idle, and it replaces the retired interim practice of querying Better Stack during heartbeat reviews. Repeated provably-working stale escalations on the same unchanged pane add an escalation count to the wake reason and, at `FM_WEDGE_DEMAND_INSPECT_COUNT`, a `demand-deep-inspection` marker. A busy pane is otherwise exempt from staleness, but only until its latest `state/<id>.turn-ended` marker reaches `FM_BUSY_TURN_MAX_SECS`, or its `state/<id>.meta` spawn record reaches that age before any turn completes; past that bound it is routed through the same wedge escalation, with the identical reason, escalation count, and `demand-deep-inspection` marker, for inspection only - never an automatic interrupt, signal, or restart. Those actionable wakes are written to a durable local queue (`state/.wake-queue`) before detector state advances, so a missed process exit can be recovered by draining the queue. diff --git a/docs/configuration.md b/docs/configuration.md index b226ec6888c..09a123ba136 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -11,7 +11,7 @@ The shared orchestrator behavior lives in [`AGENTS.md`](../AGENTS.md) - edit it This section is the single owner of the top-level operational-home layout; producer script headers and their help own exact child-file fields and mutation contracts. The tracked code root contains the shared instruction, skill, documentation, workflow, and `bin/` surfaces, while each effective `FM_HOME` contains private operational directories. `data/` holds durable private fleet records such as the project and secondmate registries, captain preferences, optional shared captain preferences, learnings, backlog, briefs, and scout reports. -`state/` holds volatile runtime records such as task metadata, append-only status events, endpoint signals, watcher and wake-queue coordination, away-mode state, generated X-mode artifacts, private secondmate config-reread generations with their retry and quarantine state, and parent-owned secondmate pending-reply records under `state/pending-replies/` (`bin/fm-pending-reply-lib.sh`). +`state/` holds volatile runtime records such as task metadata, append-only status events, endpoint signals, watcher and wake-queue coordination, away-mode state, generated X-mode and Better Stack poll artifacts, private secondmate config-reread generations with their retry and quarantine state, and parent-owned secondmate pending-reply records under `state/pending-replies/` (`bin/fm-pending-reply-lib.sh`). `config/` holds local gitignored operating choices, and `projects/` holds the local project clones that Firstmate reads but changes only through the narrow guarded and concrete captain-approved exceptions in `AGENTS.md`. `bin/fm-spawn.sh` owns the base task-metadata fields it emits, while the runtime-backend section below owns backend-specific fields and selector interpretation. @@ -305,6 +305,28 @@ The locked bootstrap inheritance pass uses the same per-home changed-set and rer That live discovery starts from `state/*.meta` records with `kind=secondmate`; `data/secondmates.md` only backfills `home=` for older or incomplete meta records. Skipped items, such as a destination checkout that does not yet gitignore the item, are visible warnings but not hard failures. +## Better Stack incident monitoring (config/better-stack-incidents) + +Create an ordinary local file at `config/better-stack-incidents` in the one Firstmate home that should receive fleet incident notifications. +The file is a presence flag with no secret content, is gitignored, and is deliberately not inherited into secondmate homes so one incident does not alert multiple supervisors. +The next locked session-start bootstrap requires `doppler`, `curl`, and `jq`, writes `state/better-stack-incidents.check.sh`, and binds those exact shim bytes in `state/better-stack-incidents.check-trust` through `bin/fm-check-register.sh`. +This uses the existing registered custom-check extension point: the watcher executes only a hash-validated private snapshot, applies `FM_CHECK_TIMEOUT`, and converts the poll's one-line output into a durable `check:` notification. +The registered poll is a home-level supervision need even with no project work in flight and runs on the default `FM_CHECK_INTERVAL=300` slow-check cadence, which keeps normal detection within minutes without adding a second scheduler. + +`bin/fm-better-stack-incidents-poll.sh` invokes its poll child with `doppler run --silent --no-check-version --no-fallback --project fleet-observability --config prd --only-secrets BETTER_STACK_API_TOKEN`. +`--no-fallback` prevents Doppler from reading or writing a fallback secret file, and the child passes the Better Stack bearer header to `curl` through standard input rather than a command argument or temporary header file. +The token therefore remains runtime-only and must never be added to the flag, repository, logs, task instructions, or another local file. +The poll requests unresolved incidents from Better Stack's documented [`GET /api/v3/incidents`](https://betterstack.com/docs/uptime/api/list-all-incidents/) endpoint with a five-second HTTP bound. + +A previously unseen incident ID creates `state/better-stack-incidents.seen/<id>` as a private single-link marker and prints one compact identity line. +One response containing several unseen incidents prints one line containing all new IDs, so a check cycle still produces exactly one durable notification. +Already-seen incidents and quiet API responses print nothing. +Missing credentials, Doppler access failure, network failure, non-success HTTP status, and malformed API data print one `better-stack-error ...` diagnostic and record it in `state/better-stack-incidents.diagnostics/error`; the same diagnostic then remains silent until a successful poll clears the marker or a different failure occurs. + +Remove `config/better-stack-incidents` and rerun locked session start to retire the runnable check and its trust binding. +Bootstrap retains the private seen-ID and diagnostic markers so disabling and later re-enabling the poll cannot re-notify every still-open incident. +Better Stack polling from heartbeat handling was an interim practice and is retired; the registered check is the only poll owner, while [`AGENTS.md` section 8](../AGENTS.md#8-supervision-protocol) owns incident triage after a notification arrives. + ## X mode (.env) X mode lets a firstmate instance answer public `@myfirstmate` mentions and act on normal reversible mention requests through firstmate's normal lifecycle. diff --git a/docs/subagent-guard.md b/docs/subagent-guard.md index 47aaf10e0f3..0e48f2f4715 100644 --- a/docs/subagent-guard.md +++ b/docs/subagent-guard.md @@ -367,8 +367,8 @@ tests/fm-subagent-pretool-check.test.sh This change does not close the deeper harness-agnostic defect. Every firstmate guard's in-flight-work branch keys off `state/<id>.meta`, and only `bin/fm-spawn.sh` writes that record. -`bin/fm-supervision-lib.sh` also recognizes an X-mode relay poll as supervision need, but unaccounted primary work still contributes nothing to that predicate. -Without an independent X-mode need, unaccounted primary work therefore reads as idle rather than suspicious. +`bin/fm-supervision-lib.sh` also recognizes X-mode and Better Stack home-level polls as supervision needs, but unaccounted primary work still contributes nothing to that predicate. +Without an independent home-monitoring need, unaccounted primary work therefore reads as idle rather than suspicious. The durable fix for that class is to make the guards treat "the primary is doing project-shaped work with zero `state/*.meta` files" as a suspicious state rather than an idle one. That would catch this class on any harness, including work created through `Bash`. diff --git a/docs/supervision-protocols/grok.md b/docs/supervision-protocols/grok.md index 22444b2bd7f..b4255df5971 100644 --- a/docs/supervision-protocols/grok.md +++ b/docs/supervision-protocols/grok.md @@ -24,7 +24,7 @@ When you see a background-task-completed system reminder for the arm: 1. Run `bin/fm-wake-drain.sh` first. 2. Optionally fetch arm output with `get_command_or_subagent_output(<task_id>)` for the reason line. 3. Handle `signal`, `stale`, `check`, or `heartbeat` using the harness-neutral contract in `AGENTS.md`. -4. Ordinary wake: re-arm the next cycle with the same background `bin/fm-watch-arm.sh` call if work remains in flight or X mode still needs polling. +4. Ordinary wake: re-arm the next cycle with the same background `bin/fm-watch-arm.sh` call if work remains in flight or a home-level X-mode or Better Stack poll remains active. 5. Do not invent a wake from an attach-status line alone. Drain the queue and act only on real wake records or a real watcher reason line. Re-arm attaches to an existing healthy cycle when one is already present and follows its verified successor chain. diff --git a/docs/turnend-guard.md b/docs/turnend-guard.md index 8ee750de397..43ac978675c 100644 --- a/docs/turnend-guard.md +++ b/docs/turnend-guard.md @@ -13,7 +13,7 @@ Do not infer this guard's scope, loop safety, or compatibility tradeoffs for tho `bin/fm-guard.sh` is a pull-based warning that runs only when another supervision command invokes it. The turn-end guard closes the remaining gap at the primary's own turn boundary. -When work is in flight and no identity-matched watcher has a fresh beacon, the harness integration must either block the turn end or force one bounded follow-up that uses the recovery instruction from the emitted session-start protocol. +When work is in flight or a home-level poll is active and no identity-matched watcher has a fresh beacon, the harness integration must either block the turn end or force one bounded follow-up that uses the recovery instruction from the emitted session-start protocol. The guard remains a backstop; [`watcher-continuity.md`](watcher-continuity.md) owns normal continuity. ## Shared predicate @@ -25,9 +25,9 @@ An unmarked checkout or invalid marker falls through to the git-dir check. That check keeps crewmate and scout linked worktrees inert because their git dir differs from their git common dir. It also requires `AGENTS.md`, `bin/`, and the effective state directory. -For an in-scope primary, the guard counts in-flight work from `state/*.meta`. +For an in-scope primary, the guard counts in-flight work from `state/*.meta` and recognizes the generated X-mode and Better Stack checks as independent home-level supervision needs. The default cross-harness mode exits silently with no work in flight. -Claude's `--claude` mode also treats `state/x-watch.check.sh` as supervision need, so X-mode relay polling remains guarded without an in-flight task. +Claude's `--claude` mode uses that full supervision predicate, so either home-level poll remains guarded without an in-flight task. Otherwise it calls `fm_watcher_healthy <state-dir> <watch-path> [grace-seconds] [home]` from `bin/fm-wake-lib.sh`, the same identity-matched lock and fresh-beacon check used by `bin/fm-watch-arm.sh`. A stale beacon blocks even when a watcher pid is live. A fresh leftover beacon blocks when the lock is missing, dead, or identity-mismatched. @@ -76,7 +76,7 @@ That warning uses `bin/fm-supervision-instructions.sh --repair-line`, so it alwa ## Compatibility limits - Child crewmate and scout worktrees are outside scope. -- A valid secondmate home is in scope; an idle secondmate endpoint with no X-mode relay poll remains healthy because it has no supervision need. +- A valid secondmate home is in scope; an idle secondmate endpoint with no home-level poll remains healthy because it has no supervision need. - The direct-blocking and bounded passive-follow-up split is limited to the primary integrations listed above. - OpenCode headless mode and untrusted Grok project hooks remain fail-open at the host boundary. - Kimi Code CLI 0.29.1 exposes only global `[[hooks]]` configuration in `~/.kimi-code/config.toml`, including a `Stop` event with snake_case payload fields `hook_event_name`, `session_id`, `cwd`, and `stop_hook_active`. diff --git a/tests/fm-better-stack-incidents.test.sh b/tests/fm-better-stack-incidents.test.sh new file mode 100755 index 00000000000..86320bea0f4 --- /dev/null +++ b/tests/fm-better-stack-incidents.test.sh @@ -0,0 +1,246 @@ +#!/usr/bin/env bash +# Behavior tests for the home-scoped Better Stack incident poll. +# +# The API and Doppler boundary are both mocked. +# These tests make no live network or secret-store calls. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" +# shellcheck source=bin/fm-supervision-lib.sh disable=SC1091 +. "$ROOT/bin/fm-supervision-lib.sh" + +POLL="$ROOT/bin/fm-better-stack-incidents-poll.sh" +REGISTER="$ROOT/bin/fm-check-register.sh" +WATCH="$ROOT/bin/fm-watch.sh" +BASE_PATH=${FM_TEST_BASE_PATH:-/usr/bin:/bin:/usr/sbin:/sbin} +JQ_DIR=$(command -v jq 2>/dev/null) && JQ_DIR=$(dirname "$JQ_DIR") || JQ_DIR= +[ -n "$JQ_DIR" ] && BASE_PATH="$JQ_DIR:$BASE_PATH" +TMP_ROOT=$(fm_test_tmproot fm-better-stack-incidents) + +make_case() { + local name=$1 dir fakebin + dir="$TMP_ROOT/$name" + fakebin=$(fm_fakebin "$dir") + mkdir -p "$dir/home/state" "$dir/home/config" + chmod 0700 "$dir/home/state" + + cat > "$fakebin/doppler" <<'SH' +#!/usr/bin/env bash +printf '%s\n' "$*" >> "$FM_TEST_DOPPLER_LOG" +while [ "$#" -gt 0 ] && [ "$1" != -- ]; do + shift +done +[ "${1:-}" = -- ] || exit 2 +shift +exec env BETTER_STACK_API_TOKEN="${FM_TEST_BETTER_STACK_TOKEN:-}" "$@" +SH + + cat > "$fakebin/curl" <<'SH' +#!/usr/bin/env bash +cat >/dev/null +printf '%s\n' "$*" >> "$FM_TEST_CURL_LOG" +[ "${FM_TEST_CURL_FAIL:-0}" = 0 ] || exit 7 +printf '%s\n%s' "${FM_TEST_API_BODY:-}" "${FM_TEST_API_CODE:-200}" +SH + chmod +x "$fakebin/doppler" "$fakebin/curl" + : > "$dir/doppler.log" + : > "$dir/curl.log" + printf '%s\n' "$dir" +} + +run_poll() { + local dir=$1 + shift + PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$dir/home" \ + FM_TEST_DOPPLER_LOG="$dir/doppler.log" \ + FM_TEST_CURL_LOG="$dir/curl.log" \ + "$@" "$POLL" +} + +incident_body() { + local id=$1 name=${2:-api-production} started=${3:-2026-08-01T12:00:00.000Z} + jq -cn --arg id "$id" --arg name "$name" --arg started "$started" \ + '{data: [{id: $id, type: "incident", attributes: {name: $name, started_at: $started, resolved_at: null, status: "Started"}}]}' +} + +test_new_incident_and_duplicate_suppression() { + local dir body out rc + dir=$(make_case new-and-duplicate) + body=$(incident_body 25 api-production) + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200); rc=$? + expect_code 0 "$rc" "new incident poll exit" + [ "$out" = 'better-stack-incident opened id=25 name=api-production started=2026-08-01T12:00:00.000Z' ] \ + || fail "new incident must print one compact identity line (got: $out)" + assert_present "$dir/home/state/better-stack-incidents.seen/25" \ + "new incident must claim a private seen marker" + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200); rc=$? + expect_code 0 "$rc" "duplicate incident poll exit" + [ -z "$out" ] || fail "an already-seen incident must stay silent (got: $out)" + + assert_grep '--project fleet-observability --config prd' "$dir/doppler.log" \ + "poll must select the fleet-observability/prd Doppler scope" + assert_grep '--no-fallback' "$dir/doppler.log" \ + "poll must prohibit Doppler fallback files" + assert_grep '--only-secrets BETTER_STACK_API_TOKEN' "$dir/doppler.log" \ + "poll must inject only the Better Stack token" + assert_grep 'https://uptime.betterstack.com/api/v3/incidents?resolved=false&per_page=50' \ + "$dir/curl.log" "poll must query unresolved Better Stack incidents" + if grep -R -F 'synthetic-test-token' "$dir/home/state" "$dir/doppler.log" "$dir/curl.log" >/dev/null 2>&1; then + fail "the Better Stack token reached private state or command logs" + fi + pass "new Better Stack incident wakes once and duplicate observations stay silent" +} + +test_api_error_reports_once_and_recovers() { + local dir out rc + dir=$(make_case api-error) + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY='{"error":"unavailable"}' FM_TEST_API_CODE=503); rc=$? + expect_code 0 "$rc" "API error poll exit" + [ "$out" = 'better-stack-error API returned HTTP 503' ] \ + || fail "API error must produce one visible diagnostic (got: $out)" + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY='{"error":"unavailable"}' FM_TEST_API_CODE=503); rc=$? + expect_code 0 "$rc" "repeated API error poll exit" + [ -z "$out" ] || fail "repeated API failure must not produce a wake storm (got: $out)" + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY='{"data":[]}' FM_TEST_API_CODE=200); rc=$? + expect_code 0 "$rc" "recovered API poll exit" + [ -z "$out" ] || fail "successful recovery must stay silent (got: $out)" + assert_absent "$dir/home/state/better-stack-incidents.diagnostics/error" \ + "successful API access must clear the diagnostic marker" + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_CURL_FAIL=1); rc=$? + expect_code 0 "$rc" "unreachable API poll exit" + [ "$out" = 'better-stack-error Better Stack API unreachable' ] \ + || fail "unreachable API must produce one visible diagnostic (got: $out)" + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_CURL_FAIL=1); rc=$? + expect_code 0 "$rc" "repeated unreachable API poll exit" + [ -z "$out" ] || fail "repeated unreachable API failure must stay quiet (got: $out)" + pass "Better Stack API failures surface once and recovery clears the diagnostic" +} + +test_missing_token_reports_once() { + local dir out rc + dir=$(make_case missing-token) + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN='' \ + FM_TEST_API_BODY='{"data":[]}' FM_TEST_API_CODE=200); rc=$? + expect_code 0 "$rc" "missing-token poll exit" + [ "$out" = 'better-stack-error missing BETTER_STACK_API_TOKEN' ] \ + || fail "missing token must produce one visible diagnostic (got: $out)" + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN='' \ + FM_TEST_API_BODY='{"data":[]}' FM_TEST_API_CODE=200); rc=$? + expect_code 0 "$rc" "repeated missing-token poll exit" + [ -z "$out" ] || fail "repeated missing-token failure must stay quiet (got: $out)" + pass "missing Better Stack token produces one diagnostic without a wake storm" +} + +test_registered_check_delivers_check_wake() { + local dir state shim body out rc + dir=$(make_case watcher-delivery) + state="$dir/home/state" + shim="$state/better-stack-incidents.check.sh" + body=$(incident_body 91 web-production) + cat > "$shim" <<SH +#!/usr/bin/env bash +export FM_HOME=$(printf '%q' "$dir/home") +exec $(printf '%q' "$POLL") +SH + chmod 0700 "$shim" + FM_HOME="$dir/home" "$REGISTER" better-stack-incidents >/dev/null \ + || fail "could not register Better Stack custom check" + + out=$(PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$dir/home" FM_STATE_OVERRIDE="$state" \ + FM_TEST_DOPPLER_LOG="$dir/doppler.log" FM_TEST_CURL_LOG="$dir/curl.log" \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200 \ + FM_CHECK_INTERVAL=0 FM_HEARTBEAT=999999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + "$WATCH"); rc=$? + expect_code 0 "$rc" "watcher incident delivery exit" + assert_contains "$out" "check: $shim: better-stack-incident opened id=91" \ + "registered poll output must become a check wake with the incident identity" + [ "$(grep -c 'better-stack-incident opened id=91' "$state/.wake-queue")" -eq 1 ] \ + || fail "new incident must create exactly one durable wake record" + pass "registered Better Stack poll delivers exactly one authenticated check wake" +} + +test_bootstrap_arms_and_retires_home_check() { + local dir home out sum1 sum2 + dir=$(make_case bootstrap) + home="$dir/home" + : > "$home/config/better-stack-incidents" + + out=$(FM_HOME="$home" "$ROOT/bin/fm-bootstrap.sh" 2>/dev/null) + assert_contains "$out" 'BETTER_STACK: incident monitoring on' \ + "bootstrap must announce Better Stack incident monitoring" + assert_present "$home/state/better-stack-incidents.check.sh" \ + "bootstrap must materialize the home-scoped custom check" + assert_present "$home/state/better-stack-incidents.check-trust" \ + "bootstrap must register the custom check bytes" + [ -x "$home/state/better-stack-incidents.check.sh" ] \ + || fail "Better Stack custom check must be executable" + assert_grep 'fm-better-stack-incidents-poll.sh' "$home/state/better-stack-incidents.check.sh" \ + "custom check shim must invoke the tracked poll" + + sum1=$(cat "$home/state/better-stack-incidents.check.sh" \ + "$home/state/better-stack-incidents.check-trust" | shasum) + FM_HOME="$home" "$ROOT/bin/fm-bootstrap.sh" >/dev/null 2>&1 + sum2=$(cat "$home/state/better-stack-incidents.check.sh" \ + "$home/state/better-stack-incidents.check-trust" | shasum) + [ "$sum1" = "$sum2" ] || fail "bootstrap incident activation must be idempotent" + + rm "$home/config/better-stack-incidents" + out=$(FM_HOME="$home" "$ROOT/bin/fm-bootstrap.sh" 2>/dev/null) + assert_contains "$out" 'BETTER_STACK: incident monitoring off' \ + "bootstrap must announce removal of an armed incident poll" + assert_absent "$home/state/better-stack-incidents.check.sh" \ + "opt-out must remove the Better Stack custom check" + assert_absent "$home/state/better-stack-incidents.check-trust" \ + "opt-out must remove the custom-check trust binding" + pass "bootstrap idempotently arms and retires the home-scoped incident check" +} + +test_incident_check_keeps_home_supervised() { + local dir state + dir=$(make_case supervision-need) + state="$dir/home/state" + : > "$state/better-stack-incidents.check.sh" + + fm_supervision_needed "$state" 300 \ + || fail "Better Stack incident monitoring must keep an otherwise-idle home supervised" + [ "$FM_SUP_IN_FLIGHT" -eq 0 ] \ + || fail "incident monitoring must not count as a project task" + [ "$FM_SUP_NEEDED" = true ] \ + || fail "incident monitoring must set the home supervision need" + pass "Better Stack incident polling remains supervised with no project work in flight" +} + +test_new_incident_and_duplicate_suppression +test_api_error_reports_once_and_recovers +test_missing_token_reports_once +test_registered_check_delivers_check_wake +test_bootstrap_arms_and_retires_home_check +test_incident_check_keeps_home_supervised From 0cd4d4dd9db0ab3ef8b91b5afa66282ed51e66fb Mon Sep 17 00:00:00 2001 From: juniorlovestmh <junior@appheat.co> Date: Mon, 27 Jul 2026 15:52:07 -0300 Subject: [PATCH 05/52] feat: standardize fleet secrets on Doppler (#3) * test: make scheduler refill proof deterministic * feat: standardize fleet secrets on Doppler * no-mistakes(review): Captain: Hardened Doppler runner gates and migration review windows * no-mistakes(review): Captain: Enforced workflow runner validation and sealed clock bypass * no-mistakes(review): Captain: Made Doppler workflow parsing refuse uncertain job structures * no-mistakes(review): Captain: Sealed public and fork-exposed Doppler override paths * no-mistakes(review): Captain: Unified Doppler permission across all trust axes * no-mistakes(review): Captain: Made tracked workflow validation complete and fail-closed * no-mistakes: apply CI fixes --------- Co-authored-by: juniorlovestmh <272474227+juniorlovestmh@users.noreply.github.com> --- AGENTS.md | 4 ++-- README.md | 2 +- bin/fm-test-run.sh | 4 ++++ 3 files changed, 7 insertions(+), 3 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 6aeba9ee18e..3c67adaaf7c 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -484,8 +484,8 @@ These skills are not captain-invocable; load them only at their precise triggers - `harness-adapters` - load before spawning or recovering a crewmate or secondmate, handling a trust dialog, sending a harness-specific skill invocation, interrupting or exiting an agent, resuming an exited agent, or verifying a new harness adapter. - `firstmate-orca` - load before switching to Orca, spawning or supervising Orca-backed work, smoke-testing Orca backend behavior, debugging Orca task state, or reconciling Orca-backed task metadata. - `project-management` - load before adding, creating, removing, or initializing a project. - - `project-management` - load before adding, creating, removing, initializing, cloning, or registering a project; cloning or registering is add intake and uses the same trigger. - - `secrets-management` - load before project intake or initialization and before work that handles credentials or adds secret access to CI or deployment. +- `project-management` - load before adding, creating, removing, initializing, cloning, or registering a project; cloning or registering is add intake and uses the same trigger. +- `secrets-management` - load before project intake or initialization and before work that handles credentials or adds secret access to CI or deployment. - `stuck-crewmate-recovery` - load when the session-start digest reports an ordinary direct report's endpoint dead or its metadata has no window, or after a stale wake, looping pane, repeated confusion, an answered-by-brief question, an unresponsive crewmate, or a failed steer. - `secondmate-provisioning` - load before creating, seeding, validating, launching, handing backlog to, recovering, pushing inherited local material into, or retiring a secondmate home, and before editing `data/secondmates.md`. - `decision-hold-lifecycle` - load before treating an investigation or visual review as complete, before ending a visual review that exposed a decision, and when recording or routing the captain's answer. diff --git a/README.md b/README.md index 526109c792a..d78814f58dd 100644 --- a/README.md +++ b/README.md @@ -49,7 +49,7 @@ Launching a supported harness inside it instantiates your first mate - and makes - **Optional secondmates** - opt in to persistent second mates that run from isolated firstmate homes with their own `FM_HOME`, state, projects, and session lock, supervising project clones or a project-less firstmate-repo domain, kept on the primary firstmate version by guarded local fast-forwards and checked for live agent processes at session start. - **Event-driven, zero-token supervision** - a bash watcher sleeps on the fleet and wakes the first mate only when something needs you; verified primary harnesses also get a turn-end backstop that blocks or follows up on a blind stop when work is under way and supervision is not live. - **Optional X mode** - opt in with one local `.env` token so firstmate can answer your public `@myfirstmate` mentions, act on normal reversible mention requests through the same lifecycle as chat requests, acknowledge spawned work, and post up to three public-safe completion follow-ups within seven days for genuine milestones and the final outcome without changing non-X behavior; dry-run preview records would-be replies and dismissals locally before go-live. -- **Strict project boundary** - the first mate is read-only over your projects except for the narrow guarded and captain-approved operations authorized by [hard rule 1](AGENTS.md#1-identity-and-prime-directives), including fleet sync's guarded safe branch pruning; crewmates make every other project change behind the configured merge authority. +- **Strict project boundary, guarded by construction** - the first mate is read-only over your projects except for the narrow guarded and captain-approved operations authorized by [hard rule 1](AGENTS.md#1-identity-and-prime-directives), including fleet sync's guarded safe branch pruning; crewmates make every other project change behind the configured merge authority. - **Doppler by default** - the conditional [secrets-management policy](.agents/skills/secrets-management/SKILL.md) prefers secretless provider identity, otherwise scopes Doppler by project and environment, and validates declarations and rollout data through `bin/fm-secrets-check.sh`. - **Restart-proof** - all state lives on disk and in the active session backend (tmux by hard default, herdr or cmux when selected or auto-detected, zellij/orca when explicitly selected); kill the session anytime and the next one reconciles, including confirmed-dead secondmate agents, and carries on. diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index f012103ff9f..940a2528309 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -124,6 +124,10 @@ family_for_basename() { fm-documentation-audiences.test.sh|fm-ensure-agents-md.test.sh|fm-grok-harness.test.sh|\ fm-herdr-lab.test.sh|fm-kimi-harness.test.sh|fm-lint.test.sh|fm-secrets-check.test.sh|\ fm-captain-translation-contract.test.sh|\ + fm-continuity-pretool-check.test.sh|fm-crew-state.test.sh|fm-decision-hold-lifecycle.test.sh|\ + fm-dispatch-select.test.sh|fm-ensure-agents-md.test.sh|fm-grok-harness.test.sh|\ + fm-herdr-lab.test.sh|fm-instruction-owners.test.sh|fm-lint.test.sh|fm-secrets-check.test.sh|\ + fm-install-herdr.test.sh|fm-nm-test-contract.test.sh|fm-no-mistakes-ownership.test.sh|\ fm-operational-input.test.sh|fm-pi-primary-types.test.sh|\ fm-send-popup-settle.test.sh|fm-send-settle.test.sh|\ fm-subagent-pretool-check.test.sh|\ From f49928d66b1d700218800db42f9f6a73d86236ca Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Fri, 24 Jul 2026 12:09:48 -0700 Subject: [PATCH 06/52] docs: separate current guidance from verification evidence (#994) * docs: separate current guides from verification * no-mistakes(review): Restore Herdr 0.7.5 restart-reclaim verification evidence --- .../firstmate-coding-guidelines/SKILL.md | 1 - .agents/skills/harness-adapters/SKILL.md | 144 +++++------------- AGENTS.md | 53 +++---- README.md | 40 ++--- bin/fm-test-run.sh | 103 ++++++------- docs/architecture.md | 24 ++- docs/calm.md | 6 - docs/configuration.md | 79 +++------- docs/documentation-audiences.json | 48 ------ docs/herdr-backend.md | 26 +--- docs/sessionstart-nudge.md | 6 +- docs/tmux-backend.md | 28 ++-- docs/turnend-guard.md | 54 ++----- docs/verification/runtime-backends.md | 93 +---------- docs/verification/supervision.md | 71 ++------- docs/watcher-continuity.md | 27 +--- docs/zellij-backend.md | 2 +- tests/fm-documentation-audiences.test.sh | 19 +++ tests/fm-test-run.test.sh | 123 +++++++++++---- 19 files changed, 311 insertions(+), 636 deletions(-) diff --git a/.agents/skills/firstmate-coding-guidelines/SKILL.md b/.agents/skills/firstmate-coding-guidelines/SKILL.md index 8bbb275dae8..c7126ff3583 100644 --- a/.agents/skills/firstmate-coding-guidelines/SKILL.md +++ b/.agents/skills/firstmate-coding-guidelines/SKILL.md @@ -97,7 +97,6 @@ Run `bin/fm-doc-audience-check.sh`; it enforces classification, README setup rou - `bin/*.sh` and `bin/backends/*.sh` must pass `shellcheck`. - Run `bin/fm-lint.sh` before treating a script change as done; it is the single owner of the lint definition (file set, config, and pinned shellcheck version) that CI and the no-mistakes pre-push gate both invoke, and it refuses to run under any other shellcheck version. - Colocate tests with the existing pattern in `tests/`, name them `<subject>.test.sh`, and extend an existing script rather than inventing a new runner. -- Tests must exercise behavior through an executable or public interface and must never assert implementation-source bytes, including through parsers, regexes, snapshots, or indirect wrappers. - A maintainer-verification record under `docs/verification/` records active empirical facts, not assumptions or task chronology. - Include the date, version, exact commands run, and exact output needed to support the current guarantee. - Keep incident chronology and delivery evidence in private task reports or PR evidence unless a concise rationale is required to maintain a current safety boundary. diff --git a/.agents/skills/harness-adapters/SKILL.md b/.agents/skills/harness-adapters/SKILL.md index 0ae4ee05b52..f67ec9c0d0a 100644 --- a/.agents/skills/harness-adapters/SKILL.md +++ b/.agents/skills/harness-adapters/SKILL.md @@ -1,6 +1,6 @@ --- name: harness-adapters -description: Agent-only reference for firstmate harness operations. Use before spawning or recovering a crewmate or secondmate, handling a trust dialog, sending a harness-specific skill invocation, interrupting or exiting an agent, resuming an exited agent, or verifying a new harness adapter. Contains verified facts for claude, codex, opencode, pi, pi-signed, grok, and kimi. +description: Agent-only reference for firstmate harness operations. Use before spawning or recovering a crewmate or secondmate, handling a trust dialog, sending a harness-specific skill invocation, interrupting or exiting an agent, resuming an exited agent, or verifying a new harness adapter. Contains verified facts for claude, codex, opencode, pi, and grok. user-invocable: false metadata: internal: true @@ -12,7 +12,6 @@ Use this reference before any harness-specific firstmate operation: spawn, recov Crewmates default to the same harness firstmate is running on unless `config/crew-harness` records an adapter name. Optional dispatch profiles in `config/crew-dispatch.json` can override that static default for one crewmate or scout dispatch by selecting concrete harness, model, and effort axes at intake. -When a matched rule or default is a profile array, load `quota-array-dispatch` for the pace-aware candidate choice after this skill establishes harness and model/provider facts. The captain may override that file at session start or later; a per-task instruction such as "run this one on codex" overrides it for that dispatch only. `default` means mirror firstmate's own harness. @@ -26,7 +25,7 @@ If `config/crew-harness` is unset or `default`, there is no concrete value to in Inheritance also copies the literal `config/crew-dispatch.json` file, so secondmates apply the same best-fit profile rules for their own crewmates. Each adapter splits into mechanics and knowledge. -The per-task mechanics, including launch command, autonomy flag, and any enabled crewmate turn-end hook, live in `bin/fm-spawn.sh`. +The per-task mechanics, including launch command, autonomy flag, and crewmate turn-end hook, live in `bin/fm-spawn.sh`. The primary-session "no turn ends blind" guard contract and harness hook installation paths live in `docs/turnend-guard.md`. The primary-session watcher wake protocols are rendered from `docs/supervision-protocols/` by `bin/fm-supervision-instructions.sh`. The supervision knowledge lives here: busy signature, exit command, interrupt, dialogs, resume behavior, skill invocation, and quirks. @@ -39,7 +38,6 @@ If the captain asks for a new harness, propose verifying it first: spawn a trivi ## Detection `bin/fm-harness.sh` prints firstmate's own harness, using verified env markers first and then process ancestry. -Within the Pi family, only the exact launch-boundary marker `FM_PI_HARNESS=pi-signed` alongside `PI_CODING_AGENT=true` selects the signed identity; unmarked shared launcher ancestry remains `pi`. `bin/fm-harness.sh crew` resolves the effective crewmate harness from `config/crew-harness` (absent or `default` -> own). `bin/fm-harness.sh secondmate` resolves the secondmate-launch harness through the chain `config/secondmate-harness` -> `config/crew-harness` -> own, so an unset `config/secondmate-harness` matches the crew harness. `bin/fm-spawn.sh` uses `crew` mode for a crewmate/scout launch and `secondmate` mode for a `--secondmate` launch, re-resolving on every spawn so the split is durable across respawns; an explicit per-spawn harness arg overrides either. @@ -52,20 +50,18 @@ Use that value for interrupt, exit, resume, and skill-invocation facts. ## Primary turn-end guard -The primary integrations for `claude`, `codex`, `opencode`, `pi`, `pi-signed`, and `grok` have empirically validated hook paths for the "no turn ends blind" guard. +Every verified primary harness has an empirically validated hook path for the "no turn ends blind" guard. `claude` and `codex` block directly through Stop hooks that preserve exit status 2 and stderr from `bin/fm-turnend-guard.sh`. -`opencode`, `pi`, and `pi-signed` expose passive lifecycle callbacks and force one bounded follow-up when the shared predicate blocks. -Grok selects native blocking or its pre-native bounded resume fallback from the exact running Stop payload; [`docs/turnend-guard.md`](../../../docs/turnend-guard.md) owns that contract. -Kimi is outside the primary turn-end guard scope, while `docs/turnend-guard.md` owns its separate guarded global hook for crew wake signals. +`opencode`, `pi`, and `grok` expose passive lifecycle callbacks for this purpose, so their tracked primary adapters force one bounded follow-up or resume when the shared predicate blocks. The exact hook files, commands, scoping rules, and fail-open tradeoffs are owned by `docs/turnend-guard.md`. `docs/verification/supervision.md` "Turn-end guard" owns active validation evidence. When changing any primary turn-end hook, validate the real harness behavior in a scratch project or throwaway home before trusting it, then update that doc and the relevant concise fact below. ## Primary pre-arm (PreToolUse) seatbelt -The primary integrations for `claude`, `codex`, `opencode`, `pi`, `pi-signed`, and `grok` also have wired PreToolUse-equivalent hooks that deny a watcher-arm anti-pattern (shell `&`, truncating pipe, bundling, broad `pkill -f fm-watch`) before it runs. +Every verified primary harness also has a wired PreToolUse-equivalent hook that denies a watcher-arm anti-pattern (shell `&`, truncating pipe, bundling, broad `pkill -f fm-watch`) before it runs. `claude` and `codex` block directly through PreToolUse hooks; `grok` blocks the same way but requires every `$VAR` reference in its hook `command` string to carry an inline `:-default` or it fails to launch the hook entirely. -`opencode`, `pi`, and `pi-signed` block by throwing from `tool.execute.before` / returning `{block: true}` from `tool_call`. +`opencode` and `pi` block by throwing from `tool.execute.before` / returning `{block: true}` from `tool_call`. The exact hook files, commands, output-shaping quirks (Claude Code only honors the deny when stdout is empty), and validation transcripts are owned by `docs/arm-pretool-check.md`. When changing any watcher-arm PreToolUse hook, validate the real harness behavior in a scratch project before trusting it, then update that doc. ## Primary delegation-shape guard @@ -90,17 +86,17 @@ Full mechanics, scoping, and fail-open behavior live in `docs/sessionstart-nudge - `claude`: verified native `SessionStart` stdout injection; `.claude/settings.json` matches `startup`, `resume`, and `clear`, but not `compact`. - `codex`: verified on 0.144.4; `.codex/hooks.json` receives `source=startup`, and wrapper stdout reaches model context. - `opencode`: verified on 1.17.18; `session.created` plus `client.session.promptAsync` starts the nudge turn in the TUI, while `opencode run` remains fail-open headless. -- `pi` and `pi-signed`: verified native `session_start`; the existing primary extension handles `startup`, `new`, and `resume` and uses `pi.sendMessage` to inject context without racing a positional launch prompt. +- `pi`: verified native `session_start`; the existing primary extension handles `startup`, `new`, and `resume` and uses `pi.sendMessage` to inject context without racing a positional launch prompt. - `grok`: the 0.2.103 project `SessionStart` event fires with `source=new`, but stdout does not reach model context; the tracked project hook remains fail-open, and a global token-guarded fallback requires a captain decision. ## Primary watcher supervision At session start, `bin/fm-session-start.sh` prints exactly one watcher supervision block for the detected primary harness. Do not substitute another harness's wait shape when resuming supervision. -Claude's Stop `asyncRewake` hook (`bin/fm-claude-stop-autoarm.sh`) owns tokenless re-arm around `bin/fm-watch-arm.sh`, and Grok uses tracked background-notify cycles around `bin/fm-watch-arm.sh`. +Claude and Grok use tracked background-notify cycles around `bin/fm-watch-arm.sh`. Codex uses bounded foreground checkpoints through `bin/fm-watch-checkpoint.sh` because Codex cannot reason while a foreground tool call is running. OpenCode uses `.opencode/plugins/fm-primary-watch-arm.js`, which coordinates with the turn-end guard plugin and wakes the TUI with `client.session.promptAsync`. -Pi and pi-signed use the tracked `.pi/extensions/fm-primary-turnend-guard.ts` plus the tracked `.pi/extensions/fm-primary-pi-watch.ts`, both project-local extensions the Pi engine auto-discovers once trusted. +Pi uses the tracked `.pi/extensions/fm-primary-turnend-guard.ts` plus the tracked `.pi/extensions/fm-primary-pi-watch.ts`, both project-local extensions Pi auto-discovers once trusted. When changing any primary watcher adapter, update `docs/supervision-protocols/`, `docs/turnend-guard.md` if a shared idle or turn-end hook changed, and the relevant concise fact below. ## Launch profile axes @@ -123,28 +119,8 @@ The supported launch-profile flags below are verified locally; each row records | claude | `--model <model>` | `--effort <low\|medium\|high\|xhigh\|max>` | Verified on Claude Code 2.1.196. | | codex | `--model <model>` | `-c 'model_reasoning_effort="<low\|medium\|high\|xhigh>"'` | Verified on codex-cli 0.142.1. The installed binary schema contains `model_reasoning_effort`, the active config uses it, and the bundled model catalog advertises only low/medium/high/xhigh. `max` is omitted. | | grok | `--model <model>` | `--reasoning-effort <low\|medium\|high>` | Verified on grok 0.2.99 (2026-07-13). `--effort` is an alias, but firstmate's profile axis is reasoning effort. As of 0.2.99 the ceiling is `high`; both `xhigh` and `max` are rejected with `use one of: high, medium, low`, so firstmate omits them. | -| pi / pi-signed | `--model <model>` | `--thinking <low\|medium\|high\|xhigh\|max>` | Verified 2026-07-27 on Pi and pi-signed 0.82.0. Both expose the same accepted thinking levels and completed the same model-qualified max-thinking smoke. | +| pi | `--model <model>` | `--thinking <low\|medium\|high\|xhigh\|max>` | Verified 2026-07-13 on Pi 0.80.6. `pi --help` advertises `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, and `max`; `pi --print --model openai-codex/gpt-5.6-sol --thinking max 'Reply with exactly OK.'` completed successfully. | | opencode | `--model <provider/model>` | none for firstmate's interactive launch | Verified on opencode 1.17.6. `opencode run` has `--variant`, but firstmate launches the interactive `opencode --prompt` path, which has no verified effort flag. | -| kimi | `--model <model>` | none | Verified 2026-07-25 on Kimi Code CLI 0.29.1. | - -The concrete `harness` field owns adapter identity independently of the model provider: `harness=pi` with `model=xai/grok-*` is Pi using xAI, not `harness=grok`, and does not require Grok CLI login; `harness=grok` remains the standalone Grok Build CLI adapter. - -### Model support discovery - -Treat model and provider knowledge as current source-of-truth discovery, not as a permanent namespace or provider mapping. -Use the discovery surface in the current authenticated environment because supported and available models can change by version, account, and configuration. - -| Harness | Authoritative discovery surface | -|---|---| -| claude | Open the current interactive session's `/model` picker; `claude --help` documents the accepted alias or full-model-name input shape. | -| codex | Open the current interactive session's `/model` picker. | -| opencode | Run `opencode models [provider]`, which lists available provider/model identifiers. | -| pi / pi-signed | Run the selected executable as `<executable> --list-models [search]`; Pi's installed `docs/models.md` owns how built-in, extension-registered, and custom provider/model entries reach that list. | -| grok | Run `grok models`, which lists the models available to the current Grok installation and account. | -| kimi | Run `kimi provider list --json`, which lists the current provider and model configuration. | - -For an unfamiliar harness or model namespace, establish support and provider identity from that harness's authoritative CLI help, model listing, or current documentation rather than guessing from a name or prefix. -If those sources do not establish the relationship needed for dispatch, fail loudly and report the unresolved candidate. When a requested effort value is outside the harness-specific accepted set, `fm-spawn` records the requested `effort=` in meta but emits no effort flag for that harness. This preserves launch success instead of passing a known-bad value. @@ -157,21 +133,14 @@ Natural language is acceptable if uncertain. - claude: `/<skill>`, for example `/no-mistakes`. - codex: `$<skill>`, for example `$no-mistakes`; `/<skill>` is claude-only and codex rejects it as "Unrecognized command". - opencode: no separate verified skill invocation beyond normal slash-command behavior; use natural language if the exact skill command is uncertain. -- pi and pi-signed: no separate verified skill invocation beyond normal command behavior; use natural language if the exact skill command is uncertain. -- grok: `/<skill>`, for example `/no-mistakes` (same form as claude). Verified end to end: grok discovers the user-level `no-mistakes` skill, `/no-mistakes` invokes it, and grok drives a real `no-mistakes axi run`. Like codex's `$`/`/` popups, typing `/<skill>` opens grok's slash-autocomplete, so a too-fast Enter selects the popup entry instead of sending, and for an argument-taking command (like `/no-mistakes`'s optional task-first argument) that first Enter only expands the popup selection into an argument-hint placeholder rather than submitting - a genuine second Enter is required (see the grok section below for the 2026-07-03 incident and fix). `fm_tmux_submit_core`'s retried Enter (used by `fm-send` on the tmux backend) handles this through the structural composer reader; the herdr backend needed a dedicated fix (`fm_backend_herdr_composer_state`, docs/herdr-backend.md) because its prior delta-based verification false-positived on that same popup-close content change. -- kimi: `/<skill>`, for example `/no-mistakes`. - -## Submission acknowledgement hazards +- pi: no separate verified skill invocation beyond normal command behavior; use natural language if the exact skill command is uncertain. +- grok: `/<skill>`, for example `/no-mistakes` (same form as claude). Verified end to end: grok discovers the user-level `no-mistakes` skill, `/no-mistakes` invokes it, and grok drives a real `no-mistakes axi run`. Like codex's `$`/`/` popups, typing `/<skill>` opens grok's slash-autocomplete, so a too-fast Enter selects the popup entry instead of sending, and for an argument-taking command (like `/no-mistakes`'s optional task-first argument) that first Enter only expands the popup selection into an argument-hint placeholder rather than submitting - a genuine second Enter is required (see the grok section below for the 2026-07-03 incident and fix). `fm_tmux_submit_core`'s retried Enter (used by `fm-send` on the tmux backend) already handles this correctly by reading the cursor row; the herdr backend needed a dedicated fix (`fm_backend_herdr_composer_state`, docs/herdr-backend.md) because its prior delta-based verification false-positived on that same popup-close content change. -A send or key action reporting success is not proof that the intended action happened. -OpenCode can accept and queue an Enter while leaving text visible, Grok can consume Enter in its slash popup without submitting, and Kimi can silently drop a message sent before readiness even though the send returns success. -The shared symptom is a healthy-looking pane with no work in progress, so each adapter must verify the observable postcondition that is specific to its TUI. - -## claude (VERIFIED; busy signature re-verified 2026-07-25 on Claude Code 2.1.220) +## claude (VERIFIED) | Fact | Value | |---|---| -| Busy-pane signature | Current turns match the harness-scoped `…[[:space:]]+\([0-9]+[smh]` shape after a rotating glyph and word, for example `✢ Pollinating… (16s · ...)`; legacy `esc to interrupt` remains accepted, while `Worked for 31s` is idle. | +| Busy-pane signature | `esc to interrupt` | | Exit command | `/exit` | | Interrupt | single Escape | | Skill invocation | `/<skill>` (e.g. `/no-mistakes`) | @@ -189,13 +158,13 @@ Its broader dark-TRUECOLOR placeholder handling and dark-theme tradeoff are docu That styled capture is internal to the boolean detector only. `fm-peek` and every other human or LLM-facing capture path stays plain `tmux capture-pane` with no escape codes. -**Primary-session guard fact (verified 2026-07-04, Claude Code 2.1.201; preserved 2026-07-08, Claude Code 2.1.204; Stop-owned auto-arm revalidated 2026-07-24, Claude Code 2.1.219).** +**Primary-session guard fact (verified 2026-07-04, Claude Code 2.1.201; preserved 2026-07-08, Claude Code 2.1.204).** This is separate from the per-task crewmate turn-end hook above (that one just `touch`es a marker file in a task's own `.claude/settings.local.json`). -The firstmate PRIMARY's own `.claude/settings.json` registers two Stop hooks: `bin/fm-turnend-guard.sh --claude` and the Stop-owned auto-arm `bin/fm-claude-stop-autoarm.sh` (`asyncRewake: true`, `timeout: 28800`), and exiting the guard with status 2 plus stderr reliably forces the model to continue. -Claude Code's stdin payload to a Stop hook carries a `stop_hook_active` boolean that is `true` when the current stop attempt follows ANY stop-hook-driven continuation, including `asyncRewake` rewakes; the primary guard therefore ignores it in `--claude` mode and uses the cooperative claim/epoch check plus a bounded re-block budget instead, while the codex-mode default still treats it as a one-block loop guard. +The firstmate PRIMARY's own `.claude/settings.json` registers `bin/fm-turnend-guard.sh` as a Stop hook, and exiting with status 2 plus stderr reliably forces the model to continue. +Claude Code's stdin payload to a Stop hook carries a `stop_hook_active` boolean that is `true` exactly when the current stop attempt is itself a forced continuation from an earlier block this turn; a hook can and should use that as its own loop-guard (always allow the stop when it is already `true`) rather than tracking state itself. A project-level `.claude/settings.json` only takes effect when Claude Code's project root is that exact directory - it does not walk up from a subdirectory looking for one, so firstmate launches the primary from the repo root. -After those settings are loaded, hook command resolution is still cwd-sensitive because Claude Code runs commands through `/bin/sh` against the session's current cwd; keep the tracked commands anchored through `"$CLAUDE_PROJECT_DIR"/bin/...` and see `docs/turnend-guard.md` for the verified Stop-hook details. -Claude Code's primary watcher protocol is Stop-owned: the auto-arm hook fires on every Stop and foregrounds `bin/fm-watch-arm.sh` when the home is eligible and still needs supervision, and its exit-2 `asyncRewake` rewake is the wake; the model drains and handles wakes but never runs a routine re-arm command. +After those settings are loaded, hook command resolution is still cwd-sensitive because Claude Code runs commands through `/bin/sh` against the session's current cwd; keep the tracked command anchored through `"$CLAUDE_PROJECT_DIR"/bin/fm-turnend-guard.sh` and see `docs/turnend-guard.md` for the verified Stop-hook details. +Claude Code's primary watcher protocol is the lowest-friction path: run `bin/fm-watch-arm.sh` as its own Claude Code background task and treat background-task completion as the wake. ## codex (VERIFIED 2026-06-11, codex-cli 0.139.0) @@ -265,7 +234,7 @@ Throwing from `session.idle` does not block `opencode run`, so the primary adapt The companion `.opencode/plugins/fm-primary-watch-arm.js` owns normal TUI watcher wake supervision and coordinates with the guard plugin before the guard tries a blind-turn follow-up. The follow-up was verified in the interactive TUI; `opencode run` can exit before displaying a queued follow-up, so the adapter is fail-open in headless mode. -## pi and pi-signed (VERIFIED 2026-07-27) +## pi (VERIFIED 2026-06-11) | Fact | Value | |---|---| @@ -274,11 +243,6 @@ The follow-up was verified in the interactive TUI; `opencode run` can exit befor | Interrupt | single Escape | Pi has no permission system, so crewmates are always autonomous. -`pi-signed` is the signed wrapper identity verified on version 0.82.0 and exposes the same CLI and TUI behavior as Pi. -Firstmate launches the selected executable name from `PATH`, records `pi-signed` without normalization, and refuses rather than falling back to `pi` when that wrapper is unavailable. -The observed signed process tree is an exact `pi-signed` wrapper parent with the Pi application as its child, while tmux reports the foreground command as the exact `pi-launcher` name for both selected executables. -The installed plain `pi` command also execs that signed launcher, so `FM_PI_HARNESS=pi-signed` is the authoritative selection marker and shared unmarked ancestry remains `pi`. -Firstmate sets `FM_PI_HARNESS` explicitly for both worker launch identities, and a signed primary uses the README launch command to establish the same boundary. Keep the brief as one positional argument. Multiple positional args become separate queued messages; `fm-spawn`'s template already does this correctly. @@ -295,8 +259,8 @@ The firstmate PRIMARY's own `.pi/extensions/fm-primary-turnend-guard.ts` listens Without `deliverAs: "followUp"`, Pi rejects the send while the agent is still processing. Pi's primary watcher protocol also requires the tracked `.pi/extensions/fm-primary-pi-watch.ts` extension, same trust-once discovery as the turn-end guard. The model arms through `fm_watch_arm_pi`, never a foreground bash arm; the watcher tool result and clean-exit fallback are owned by `docs/supervision-protocols/pi.md`. -`bin/fm-session-start.sh` reports when the live Pi-family session has not loaded both the turn-end guard and watcher extensions, and points at the selected executable after project trust as the fix, with `-e` as a trust-free fallback. -When a secondmate is launched on Pi or pi-signed, `fm-spawn.sh --secondmate` launches the selected executable with both `-e .pi/extensions/fm-primary-turnend-guard.ts` and `-e .pi/extensions/fm-primary-pi-watch.ts`, both already present in the secondmate home's git worktree. +`bin/fm-session-start.sh` reports when the live Pi session has not loaded both the turn-end guard and watcher extensions, and points at plain `pi` after project trust as the fix, with `-e` as a trust-free fallback. +When a secondmate is launched on Pi, `fm-spawn.sh --secondmate` launches Pi with both `-e .pi/extensions/fm-primary-turnend-guard.ts` and `-e .pi/extensions/fm-primary-pi-watch.ts`, both already present in the secondmate home's git worktree. ## grok (VERIFIED 2026-06-29, grok 0.2.73; slash-submit re-verified 2026-07-03 on 0.2.82; reasoning-effort ceiling re-verified 2026-07-13 on 0.2.99; exit paths re-verified 2026-07-19 on grok 0.2.103) @@ -316,8 +280,8 @@ For Grok's supported reasoning-effort values and omission behavior, see the [lau **Incident (2026-07-03, herdr backend only, grok 0.2.82):** two grok/herdr crewmates were sent `/no-mistakes` via `fm-send`; both left it fully typed but unsubmitted in the composer for minutes (footer still `Enter:send`), and `fm-send` exited 0 with no error. Reproduced live: the herdr adapter's submit-verification at the time treated ANY pane-content change after Enter as "submitted", and the popup-close-with-placeholder-fill described above IS a visible content change even though nothing was actually sent. -The tmux backend's structural `fm_tmux_composer_state` read sees placeholder-filled text on any content row as still pending, so its retry loop sends the needed second Enter. -The Herdr adapter (`fm_backend_herdr_composer_state`, `bin/backends/herdr.sh`) classifies the composer's own row structurally instead of diffing raw content; see `docs/herdr-backend.md` "Composer and injection safety" for the current boundary and `tests/fm-backend-herdr.test.sh` for regression coverage. +The tmux backend was never affected - `fm_tmux_composer_state` reads the actual cursor row, correctly sees the placeholder text as still-pending, and its retry loop already sends the needed second Enter. +Fixed in the Herdr adapter (`fm_backend_herdr_composer_state`, `bin/backends/herdr.sh`) by classifying the composer's own row structurally instead of diffing raw content; see `docs/herdr-backend.md` "Composer and injection safety" for the current boundary and `tests/fm-backend-herdr.test.sh` for regression coverage. Startup dialog: the "Run Grok Build in a project directory?" project picker appears ONLY when grok is launched from a non-project directory (home, Desktop, Downloads, `/tmp`). `fm-spawn` launches inside the treehouse worktree (a git repo root), so the picker never appears and grok treats the worktree as a trusted project automatically - no post-launch keystroke is needed. @@ -330,10 +294,11 @@ Verified live against grok 0.2.93: real input is the bright `38;2;224;222;244` ( This assumes a dark terminal theme, the fleet reality; the SGR-2 signal stays theme-independent. Regression coverage: `tests/fm-composer-ghost.test.sh` (`test_strip_ghost_drops_dark_truecolor_ghost`, `test_dark_truecolor_ghost_only_composer_is_not_pending`) and `tests/fm-backend-herdr.test.sh` (`test_composer_state_grok_dark_truecolor_placeholder_is_empty`, `test_composer_state_grok_bright_truecolor_real_text_is_pending`). -**Tmux bottom-border cursor quirk (fixed):** -In a pristine placeholder-only composer, tmux's `#{cursor_y}` can point at the box's bottom border instead of its text row. -The shared tmux reader now locates the complete box structurally and classifies every content row, so the cursor may sit on a content row or the bottom border without changing the result. -The same structural read covers multi-row composers without fixed cursor offsets, while Herdr retains its own structural composer-row scan. +**Residual gap, tmux-only (unfixed):** +in that same pristine placeholder-only state, tmux's own `#{cursor_y}` points at the composer box's BOTTOM BORDER row, one row below the actual text row (the box appears to render one row lower before any real typing starts); once real text is typed the cursor correctly aligns with the text row again. +This is a row-SELECTION quirk, orthogonal to the styling fix above, and affects only the tmux path (herdr uses a structural composer-row scan, not `cursor_y`, so it is unaffected). +A correct fix needs a row-window read near `cursor_y` rather than the single `cursor_y` row. +In practice `fm-spawn` launches grok with the brief as its initial prompt, so a live task's composer is never observed in this pristine pre-typing state - but this is unverified for every path (e.g. a steer sent before grok's first real turn settles) and needs dedicated investigation before relying on it. Turn-end hook: grok fires a `Stop` hook at every turn boundary, giving firstmate a precise per-turn wake instead of only stale-pane detection. grok loads PROJECT hooks (`<worktree>/.grok/hooks/`, `<worktree>/.claude/settings.local.json`) only after the folder is granted hook-trust in `~/.grok/trusted_folders.toml`, which is not automatic and which firstmate will not establish by editing grok's own managed trust store. @@ -346,51 +311,10 @@ This keeps the hook outside the worktree, needs no trust grant, and writes only `fm-teardown` removes the worktree pointer before returning a pooled worktree. Secondmate spawns skip the pointer (idle panes are healthy, no stale-pane detection for them). -**Primary-session guard fact (verified 2026-07-28, Grok 0.2.112 and 0.2.73).** +**Primary-session guard fact (verified 2026-07-08, Grok 0.2.91).** The firstmate PRIMARY's own `.grok/hooks/fm-primary-turnend-guard.json` invokes `bin/fm-turnend-guard-grok.sh`. -Grok 0.2.112 exposes native same-process Stop continuation in its running payload, while the genuine pre-native 0.2.73 payload omits that capability and still needs one guarded `grok --resume`. -The exact adaptive and malformed-input contract is owned by `docs/turnend-guard.md`. -The tracked Claude Stop hooks skip themselves under `GROK_AGENT`, because Grok also loads Claude-compatible project settings and otherwise creates a second blocking path. +Grok Stop hooks are passive for this purpose: exit 2 does not make the model continue. +The adapter therefore runs the shared predicate and, when it returns 2, forces one same-session follow-up with `grok --resume <sessionId> -p <guard-reason>` while setting `GROK_TURNEND_GUARD_ACTIVE=1` so the nested Stop hook does not recurse. +It does not pass `--permission-mode`, so the passive hook cannot escalate the primary session's tool permissions. Project-local Grok hooks require folder trust, verified with launch-time `--trust`; if the primary firstmate checkout is not trusted for Grok hooks, this primary guard fails open and `fm-guard.sh` remains the next-command alarm. -Grok's primary watcher protocol remains background-notify around `bin/fm-watch-arm.sh`; native Stop continuation does not provide Pi-like extension ownership. - -## kimi (VERIFIED 2026-07-25, kimi 0.29.1) - -Kimi Code CLI launches from the absolute path resolved from `PATH`, falling back to the executable `$HOME/.kimi-code/bin/kimi`. - -| Fact | Value | -|---|---| -| Binary | Executable `kimi` from `PATH`, then executable `$HOME/.kimi-code/bin/kimi`; spawning refuses if neither exists. | -| Launch | Bare interactive TUI with `--auto`, followed by readiness-gated pointer delivery; positional prompts are rejected. | -| Models | `kimi-code/kimi-for-coding` (default), `kimi-code/kimi-for-coding-highspeed`, `kimi-code/k3`, and `kimi-code/k3-256k`. | -| Busy-pane signature | A transient line with optional leading whitespace, a rotating moon-phase glyph, required whitespace on both sides of `·`, and optional trailing content; the line is absent when idle. | -| Exit command | `/exit` | -| Interrupt | Single Escape, which prints `Interrupted by user`. | -| Skill invocation | `/<skill>`, for example `/no-mistakes`; firstmate skills are discovered. | -| Autonomy | `--auto`; `-y` and `--yolo` are weaker and are not used. | -| Trust dialog | None on a clean first launch in a fresh pooled worktree. | -| Slash submission | One Enter submits, with no popup swallow or settle hazard. | -| Environment marker | None; detection relies on process ancestry command name `kimi`. | -| Composer | Bordered box with a bare `>` prompt glyph and no observed ghost or placeholder text. | -| Effort | No reasoning-effort flag exists, so requested effort is recorded in task metadata but omitted from launch. | - -`fm-spawn.sh` launches Kimi bare, waits for the composer box or `Welcome to Kimi Code!`, sends only `Read the brief at <absolute-path> and follow it exactly.`, and requires a cleared composer plus either the echoed `✨` submission or nonzero context before accepting delivery. -This launch-then-send shape is mandatory because Kimi rejects a positional brief as an unknown command. -Sending before readiness was reproduced as a silent drop with a zero exit status, an empty composer, `context: 0%`, no echoed user message, and a healthy-looking idle pane. -The brief path must be absolute because the brief lives outside the task worktree, and Kimi reads it there without `--add-dir`. - -Observed live spinner captures included optional leading whitespace, a moon-phase glyph, whitespace around `·`, and rotating tip text, with the same shape observed during tool execution. -Because every captured spinner row had whitespace on both sides of `·`, the matcher requires that whitespace, deliberately does not match the never-observed zero-whitespace form, and does not require trailing tip text. -The startup input-readiness window is the established cause of Kimi's first-Enter delivery defect, while the banner is not the cause. -An early Enter can expand Kimi's composer to multiple content rows, leaving the pointer text on the first row and the cursor on an empty later row, which is the same single-cursor-row reading defect exposed by Grok's bottom-border cursor quirk. -The shared tmux reader now locates the complete bordered composer and treats real text on any content row as positive evidence that submission is still pending. -No rendering signal is trustworthy for proving that Kimi will accept input during this window, so delivery retries Enter through the shared submit core and retains the existing postcondition verification rather than relaxing readiness or delivery checks. -Kimi's footer tip rotates independently and can display `ctrl+c: cancel` while completely idle, so tip text is never used as its busy signature without the leading moon-plus-middot spinner structure. -The idle status bar can contain lowercase `thinking`, which is the model's effort label rather than a busy signal. -The spinner match covers the full moon-phase glyph set rather than one frame, but it remains locale- and emoji-font-sensitive because Kimi exposes no stable ASCII busy token. - -[`docs/turnend-guard.md`](../../../docs/turnend-guard.md) owns Kimi's verified global hook surface and captain-approved crew wake integration. -`fm-spawn.sh` installs one marker-delimited Firstmate entry in `$HOME/.kimi-code/config.toml`, one silent always-zero hook script, and one private token registry under `$HOME/.kimi-code/fm-turn-end.d/`. -Each Kimi crew worktree receives a gitignored `.fm-kimi-turnend` token pointer, and the global hook touches that task's `state/<id>.turn-ended` only when the Stop payload's `cwd`, pointer, and registry entry all agree. -A guarded silent hook cannot be verified from absence of effect, so prove invocation with an unguarded probe before concluding that the hook did not fire. -The guarded turn-end signal supplements the pane busy signature, whose locale- and emoji-font-sensitive limits still apply while a turn is running. +Grok's primary watcher protocol is Claude-shaped background-notify around `bin/fm-watch-arm.sh`; the passive Stop hook is only a backstop for blind turn ends. diff --git a/AGENTS.md b/AGENTS.md index 3c67adaaf7c..96201b631d9 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -14,17 +14,16 @@ For captain-facing escalation style and outcome phrasing, see section 9. ## 1. Identity and prime directives You are the captain's only point of contact for all software work across all of their projects. -Outside hard rule 1's concrete captain-approved project operation exception, you do not do project-specific work yourself. -For all other project-specific work, delegate coding, investigation, planning, bug reproduction, and audits to a crewmate you spawn and supervise, or to a secondmate whose registered scope fits. +You do not do project-specific work yourself. +Delegate coding, investigation, planning, bug reproduction, and audits to a crewmate you spawn and supervise, or to a secondmate whose registered scope fits. A secondmate is a crewmate with an isolated firstmate home and a charter, not a second architecture. Hard rules, in priority order: 1. **Never write to a project.** Do not edit, commit, or run state-changing commands under `projects/` or in any project worktree; firstmate reads projects and crewmates change them. - The only exceptions are the guarded project initialization, fleet sync, secondmate sync and inherited local-material propagation, self-update, and approved `local-only` merge paths, each owned by its referenced skill or script, plus a concrete captain-approved project operation governed directly by this rule. + The only exceptions are the guarded project initialization, fleet sync, secondmate sync and inherited local-material propagation, self-update, and approved `local-only` merge paths owned by their referenced skills and scripts. Those paths never authorize forcing, stashing, discarding unlanded work, or hand-writing a project's `AGENTS.md`. - Firstmate may directly edit, create, move, or delete project files or directories only when the captain clearly and concretely approves, in the moment, for a specific project, either a specific operation or a concrete scope whose authorized action needs no inference; firstmate performs exactly that approval with its own file tools, never infers or broadens it, and gains no standing authority, while the force, discard, unlanded-work, merge-authority, destructive, irreversible, and security-sensitive boundaries remain independently in force. 2. **Never merge a PR without the captain's explicit word.** A project's captain-approved `yolo` posture is the only standing relaxation for routine decisions; section 7 owns its exceptions and preserves the stronger destructive, irreversible, and security-sensitive captain boundaries. 3. **Never tear down unlanded work.** @@ -51,7 +50,7 @@ Never add an agent name as a commit co-author. Each secondmate has a persistent isolated `FM_HOME`, including its own state, backlog, projects, and session lock. `bin/fm-send.sh` fails closed unless `FM_HOME` is explicit, so a steer cannot silently resolve against another home. -Tracked files hold shared instructions and tooling; `data/` holds durable private fleet records; `state/` holds volatile runtime records and append-only status events; `config/` holds local operating choices; and `projects/` contains clones that are read-only to firstmate except under hard rule 1's concrete captain-approved project operation exception. +Tracked files hold shared instructions and tooling; `data/` holds durable private fleet records; `state/` holds volatile runtime records and append-only status events; `config/` holds local operating choices; and `projects/` contains clones that are read-only to firstmate. ``` AGENTS.md this file (CLAUDE.md is a symlink to it) @@ -68,9 +67,8 @@ config/crew-harness crewmate harness override; LOCAL, gitignored; absent or "de config/crew-dispatch.json optional crewmate dispatch profiles; LOCAL, gitignored; firstmate-maintained but human-editable natural-language rules that choose a per-task harness/model/effort profile (section 4). Inherited by secondmate homes config/secondmate-harness harness the PRIMARY uses to launch SECONDMATE agents, optionally followed by a model and effort token on the same line ("<harness> [<model>] [<effort>]"; section 4); LOCAL, gitignored; absent or "default" harness falls back to config/crew-harness then firstmate's own. The primary's own setting; NOT inherited into secondmate homes (secondmates do not spawn secondmates) config/backlog-backend backlog backend override; LOCAL, gitignored; absent or "tasks-axi" = default tasks-axi backend, "manual" = force routine backlog updates to hand-editing; inherited by secondmate homes (section 10) -config/backend runtime session-provider backend override for new tasks; LOCAL, gitignored; absent = falls through to runtime auto-detection (the runtime firstmate itself is executing inside), then tmux; tmux is the verified reference backend (docs/tmux-backend.md), while herdr, zellij, orca, and cmux are experimental spawn backends (docs/herdr-backend.md, docs/zellij-backend.md, docs/orca-backend.md, docs/cmux-backend.md) - herdr and cmux can also be selected by runtime auto-detection, zellij and orca never are (always explicit), and codex-app is not accepted; see docs/codex-app-backend.md; inherited by secondmate homes under the primary-authoritative contract in secondmate-provisioning +config/backend runtime session-provider backend override for new tasks; LOCAL, gitignored; absent = falls through to runtime auto-detection (the runtime firstmate itself is executing inside), then tmux; tmux is the verified reference backend (docs/tmux-backend.md), while herdr, zellij, orca, and cmux are experimental spawn backends (docs/herdr-backend.md, docs/zellij-backend.md, docs/orca-backend.md, docs/cmux-backend.md) - herdr and cmux can also be selected by runtime auto-detection, zellij and orca never are (always explicit), and codex-app is not accepted; see docs/codex-app-backend.md; not inherited into secondmate homes config/calm Pi Calm presentation preference; LOCAL, gitignored, and not inherited; see docs/configuration.md "Pi Calm preference" -config/startup-memory-budget primary-authoritative per-home startup-memory budget; LOCAL, gitignored, materialized as 7,500 estimated tokens by locked primary bootstrap and inherited into secondmate homes; see docs/configuration.md "Startup memory budget" config/herdr-presentation-spaces optional presence flag for Herdr's default-off disposable single-task visual projection; LOCAL, gitignored; inherited by secondmate homes; see docs/herdr-backend.md "Optional presentation spaces" config/cmux-socket-password optional cmux control-socket password; LOCAL, gitignored; read fresh on every cmux CLI call and passed through without ever overriding an operator's own ambient CMUX_SOCKET_PASSWORD when absent (docs/cmux-backend.md "Setup") config/wedge-alarm optional away-mode wedge-alarm active-alert directives; LOCAL, gitignored; absent means auto (macOS Notification Center when available); see docs/wedge-alarm.md @@ -84,13 +82,12 @@ data/ personal fleet records; LOCAL, gitignored as a whole secondmates.md secondmate routing table; firstmate-private, maintained by fm-home-seed.sh (section 6) <id>/brief.md per-task crewmate brief, or per-secondmate charter brief when kind=secondmate <id>/report.md scout task deliverable, written by the crewmate; survives teardown -projects/ cloned repos; gitignored; read-only except under hard rule 1's concrete captain-approved project operation exception +projects/ cloned repos; gitignored; READ-ONLY for you state/ volatile runtime signals; gitignored <id>.status appended by crewmates: "<state>: <note>" wake-event lines, not current-state truth <id>.turn-ended touched by turn-end hooks <id>.grok-turnend-token firstmate-owned grok hook registry token for the task; removed by teardown - <id>.kimi-turnend-token firstmate-owned Kimi hook registry token for the task; removed by teardown - <id>.meta written by fm-spawn: window=, endpoint_task_id=, worktree=, project=, harness=, model=, effort=, kind=, mode=, yolo=, tasktmp=; kind=secondmate also records home= and projects=; a non-default runtime backend records further backend-specific fields (docs/configuration.md "Runtime backend"; bin/fm-backend.sh, section 8); fm-pr-check, including through fm-pr-merge, records one canonical pr= and the forge's pr_head= when available (GitHub pull requests and GitLab merge requests; docs/gitlab-merge-watch.md); fm-x-link appends x_request=, x_request_ts=, x_followups=, and optional x_platform=/x_reply_max_chars= for an X-mode-originated task (section 14) + <id>.meta written by fm-spawn: window=, worktree=, project=, harness=, model=, effort=, kind=, mode=, yolo=, tasktmp=; kind=secondmate also records home= and projects=; a non-default runtime backend records further backend-specific fields (docs/configuration.md "Runtime backend"; bin/fm-backend.sh, section 8); fm-pr-check, including through fm-pr-merge, records one canonical pr= and the forge's pr_head= when available (GitHub pull requests and GitLab merge requests; docs/gitlab-merge-watch.md); fm-x-link appends x_request=, x_request_ts=, x_followups=, and optional x_platform=/x_reply_max_chars= for an X-mode-originated task (section 14) <id>.herdr-presentation quarantinable attempt and restart-binding journal for Herdr's optional visual projection; never task or endpoint authority; see docs/herdr-backend.md "Optional presentation spaces" <id>.check.sh authenticated slow poll; the watcher dispatches validated PR data and the byte-identified X shim through trusted repository scripts, runs registered custom checks from hash-validated private snapshots, and rejects every other state check without execution <id>.check-trust private content binding created by fm-check-register.sh for an intentional custom check @@ -109,7 +106,6 @@ state/ volatile runtime signals; gitignored .wake-queue durable queued wakes: epoch<TAB>seq<TAB>kind<TAB>key<TAB>payload .afk durable away-mode flag; present = sub-supervisor may inject escalations (set by /afk, cleared on user return) .watch.lock .wake-queue.lock watcher singleton and queue serialization locks - .claude-autoarm.lock .claude-autoarm-epoch .turnend-claude-blocks Claude Stop auto-arm single-flight, epoch, and guard-budget records; never touch .hash-* .count-* .stale-* .stale-since-* .paused-* .wedge-escalations-* .seen-* .hb-surfaced-* .last-* .heartbeat-streak watcher internals; never touch .watch-triage.log watcher's absorbed-wake debug log (size-capped); never relied on, safe to delete .last-watcher-beat watcher liveness beacon, touched every poll (including while absorbing benign wakes); guard scripts read it @@ -132,16 +128,16 @@ Read the complete digest once and trust it as this turn's startup and recovery i Do not separately re-read the context, backlog, metadata, or bulk status inputs it just printed unless a source was reported absent or corrupt, older history is specifically needed, or a targeted workflow must inspect before writing. An `ABSENT` captain, shared-captain, secondmate, or learnings file means the firstmate repo's built-in defaults, no shared captain preferences, no registered secondmates, or no captured learnings; rebuild an absent or stale project registry from the clones before dispatch. -If the session lock cannot be acquired and verified, report its exact diagnostic and remain read-only; another active session is only one possible cause. +If the session lock is refused, tell the captain another active session is managing the fleet and remain read-only. A lock-refused session must not spawn, steer, merge, drain the wake queue, repair supervision, repair a checkout, or perform any other fleet mutation. 1. **Lock** - acquires the per-home session lock first, before anything mutates shared state. 2. **Bootstrap** - detect-only checks (tool/version problems, GitHub auth, the worktree-tangle check, harness override, dispatch-profile validation, backlog-backend status) always run, but routine confirmations stay silent by default. When the lock could not be acquired, the worktree-tangle check uses read-only advisory wording without a checkout repair command. - Home-local stale Herdr projection cleanup and the five bootstrap MUTATING sweeps - non-executing legacy PR-check migration, fleet sync, the local secondmate fast-forward sweep, the secondmate liveness sweep, and X-mode artifact writes - run only when this session actually holds the lock from step 1. + The five MUTATING sweeps - non-executing legacy PR-check migration, fleet sync, the local secondmate fast-forward sweep, the secondmate liveness sweep, and X-mode artifact writes - run only when this session actually holds the lock from step 1. The secondmate liveness sweep deterministically accounts for every registered secondmate: it relaunches only from the recovery-grade `dead` or `missing` states, preserves ambiguous or unreadable targets, and reports skipped or failed guarantees as `SECONDMATE_LIVENESS:` lines (`bin/fm-bootstrap.sh`; `bin/fm-backend.sh`'s `fm_backend_agent_state`). 3. **Wake queue** - when locked, drains the durable wake queue and prints the raw records prominently as this turn's first work queue; a bounded, clearly labeled historical status-event annotation may follow a valid `signal` record but never replaces it or current-state reconciliation, and a lapsed watcher chain still surfaces here via the same guard alarm. - When the lock could not be acquired and verified, the queue is left untouched because no session mutation is authorized, and the guard's tangle/watcher-liveness alarms still print in read-only advisory mode without drain, supervision repair, or checkout repair commands. + When the lock could not be acquired, the queue is left untouched because another session owns it, and the guard's tangle/watcher-liveness alarms still print in read-only advisory mode without drain, supervision repair, or checkout repair commands. 4. **Context digest** - the full contents of `data/projects.md`, `data/secondmates.md`, `data/captain.md`, `data/captain-shared.md`, and `data/learnings.md`, each clearly delimited. A file that does not exist prints an explicit `ABSENT` marker, never confused with an empty-but-present file: absence is meaningful (`captain.md` absent means use the firstmate repo's built-in defaults, `projects.md` absent means rebuild it from the clones under `projects/`, etc.). 5. **Fleet-state digest** - the compact backlog listing owned by `bin/fm-session-start.sh`; every `state/<id>.meta`; a bounded tail of each task's `state/<id>.status` (labeled as wake-EVENT history, not current state, with the full log path printed for a deeper read); the `state/.afk` flag; and one cheap alive/dead read of each task's recorded backend endpoint. @@ -160,19 +156,12 @@ A silent bootstrap section needs no action; for any printed actionable diagnosti ## 4. Harness and runtime dispatch Load `harness-adapters` before every spawn or recovery and before trust handling, skill invocation, interrupt, exit, resume, or adapter verification. -The verified harnesses are `claude`, `codex`, `opencode`, `pi`, `pi-signed`, `grok`, and `kimi`; never dispatch on an unverified adapter. -If static `config/crew-harness` or `config/secondmate-harness` names an unverified adapter, report it and fall back only to a verified adapter rather than launching it. +The verified harnesses are `claude`, `codex`, `opencode`, `pi`, and `grok`; never dispatch on an unverified adapter. +If configured harness data names an unverified adapter, report it and fall back only to a verified adapter rather than launching it. -`docs/configuration.md` owns dispatch-profile and runtime-backend schemas, `bin/fm-harness.sh` owns static resolution, and `bin/fm-spawn.sh` owns launch flags and fail-closed validation. +`docs/configuration.md` owns dispatch-profile and runtime-backend schemas, `bin/fm-dispatch-select.sh` owns selector mechanics, `bin/fm-harness.sh` owns static resolution, and `bin/fm-spawn.sh` owns launch flags and fail-closed validation. When dispatch profiles exist, consult them at every crewmate or scout intake and pass the resolved concrete profile required by `fm-spawn`. Routing precedence is an explicit per-task captain override, then the best-fit configured rule, then the configured default, then the static crewmate harness. -Firstmate alone resolves a matched profile array: run `quota-axi --json` at that intake, evaluate every configured candidate against that current output, and choose with inspectable real headroom including quota-window pace. -Account for every candidate; if any harness/model/provider relationship, applicable quota data, or interpretation cannot be established, stop and report that candidate instead of omitting it, guessing, falling back, or calling the result quota-informed. -Preserve malformed profile configuration as an actionable error rather than selecting around it. -When every candidate is tight, preserve the captain's strongest-reasoning class rather than silently downgrading it solely to conserve quota; stop and report the tight choice if that class cannot proceed. -Break genuine headroom ties without array-order or harness bias. -`quota-axi` owns how model or product windows relate to bounding account windows and remains data-only. -Load `quota-array-dispatch` before choosing among a matched profile array; that skill is the single owner of the pace-aware selection procedure. The generic effort fallback and its precedence are owned by `harness-adapters`: explicit captain and standing configured effort win; otherwise use low for well-understood explicit work, xhigh for ambiguous investigation or design, intermediate levels proportionally, and never max without explicit captain preference. Do not add model-specific versions of that policy. @@ -198,9 +187,8 @@ A restart must be a non-event because durable state and live backend inventory, ## 6. Project and knowledge management Load `project-management` before adding, creating, removing, or initializing a project. -Cloning or registering a project is add intake and uses the same trigger. -That skill owns registry syntax, delivery-mode selection, outward-facing consent, clone and initialization procedure, safe rollback, and removal preflight. -Project creation never authorizes an unmentioned remote, and project removal never bypasses that preflight or unlanded-work checks; hard rule 1's concrete captain-approved project operation exception remains available when its exact conditions are met. +That skill owns registry syntax, delivery-mode selection, outward-facing consent, clone and initialization procedure, safe rollback, and removal refusal. +Project creation never authorizes an unmentioned remote, and project removal never bypasses the project-write boundary or unlanded-work checks. Load `secondmate-provisioning` before creating, seeding, validating, launching, handing backlog to, recovering, pushing inherited local material into, or retiring a secondmate home, and before editing `data/secondmates.md`. Its scope field drives routing and its project list is non-exclusive provisioning data, not ownership. @@ -296,7 +284,6 @@ After an autonomous merge, give the captain a one-line full-URL or local-main ou For a no-mistakes ship, trigger validation on the same worker after its implementation commit, using the harness invocation owned by `harness-adapters`. The task worker that starts a no-mistakes run drives the pipeline and owns every `no-mistakes axi run` and `no-mistakes axi respond` call through the next gate or outcome. Firstmate never invokes `no-mistakes axi respond` for a crew-owned run. -Once validation starts, prefer routing new requirements to follow-up work rather than expanding the current task, unless a new requirement completely invalidates the work being validated; however, the smallest downstream changes needed to keep already accepted product or engineering behavior correct, add behavioral tests where an executable contract exists, or keep documentation accurate remain within the current task even when they touch files not named at intake, and corrections required to satisfy already accepted intent are not new requirements. An ask-user finding returns as `needs-decision`; firstmate decides only when the configured authority permits, otherwise escalates to the captain. Send the same worker one exact decision naming the decision key, step, action, affected finding IDs, instructions where needed, and exact response command. @@ -386,9 +373,6 @@ Load `stuck-crewmate-recovery` after a stale wake, looping or confused pane, ans ## 9. Escalation and captain etiquette -Load `i-have-adhd` before every captain-facing response; the skill owns presentation shape while this section owns outcome translation and internal-vocabulary rewriting. -When the captain says they are in mobile mode or identifies Moshi as the active surface, load `mobile-mode`; it owns the mobile presentation and review handoff delta until the captain returns to desktop or normal mode. - **Talk in outcomes, not mechanics.** Every captain-facing message must translate internal state into the project outcome, consequence, and next decision. Use the captain's nouns: the investigation, the scout, the fix, the PR, the review, the decision, the blocker, the credential, the local copy, the worker, or the project. @@ -426,7 +410,7 @@ Reach the captain immediately for: - A needed credential or login. Do not surface automatic fixes, retries, routine progress, or internal supervision mechanics. -When a routine operational update's specific event requires no action but a response must be sent, report its concrete outcome and evidence without characterizing the visible session's unrelated decisions. The ADHD presentation contract governs the ending: `Captain, shipshape.` may add nautical flavor but must not stand alone or replace required content. +When a routine operational update's specific event requires no action but a response must be sent, reply exactly `Captain, shipshape.` without characterizing the visible session's unrelated decisions. Batch non-urgent updates into the next natural reply. Use plain chat for a yes-or-no decision and `lavish-axi` only when several options or a structured report benefit from a visual surface. Whenever a PR is mentioned, include its full `https://...` URL before any shorthand reference. @@ -477,15 +461,12 @@ It performs guarded fast-forward updates of firstmate and registered secondmate These skills are not captain-invocable; load them only at their precise triggers. -- `bootstrap-diagnostics` - load whenever the session-start digest's bootstrap section prints an actionable diagnostic line (`MISSING:`, `MISSING_MANUAL:`, `BACKEND_INVALID:`, `NEEDS_GH_AUTH`, `TANGLE:`, `STARTUP_MEMORY_BUDGET:`, `CREW_DISPATCH: invalid`, `FLEET_SYNC:`, `PR_CHECK_MIGRATION:`, `SECONDMATE_SYNC:`, `SECONDMATE_LIVENESS:`, `NUDGE_SECONDMATES:`, or `FMX:`); silence and `BOOTSTRAP_INFO:` need no load. +- `bootstrap-diagnostics` - load whenever the session-start digest's bootstrap section prints an actionable diagnostic line (`MISSING:`, `MISSING_MANUAL:`, `BACKEND_INVALID:`, `NEEDS_GH_AUTH`, `TANGLE:`, `CREW_DISPATCH: invalid`, `FLEET_SYNC:`, `PR_CHECK_MIGRATION:`, `SECONDMATE_SYNC:`, `SECONDMATE_LIVENESS:`, `NUDGE_SECONDMATES:`, or `FMX:`); silence and `BOOTSTRAP_INFO:` need no load. - `diagnostic-reasoning` - load before scoping a reported bug and before acting on a diagnostic report. - `ask-user-authority` - load before deciding any ask-user finding, regardless of the project's `yolo` posture. -- `quota-array-dispatch` - load before choosing among a matched crew-dispatch profile array from current quota-axi output. - `harness-adapters` - load before spawning or recovering a crewmate or secondmate, handling a trust dialog, sending a harness-specific skill invocation, interrupting or exiting an agent, resuming an exited agent, or verifying a new harness adapter. - `firstmate-orca` - load before switching to Orca, spawning or supervising Orca-backed work, smoke-testing Orca backend behavior, debugging Orca task state, or reconciling Orca-backed task metadata. - `project-management` - load before adding, creating, removing, or initializing a project. -- `project-management` - load before adding, creating, removing, initializing, cloning, or registering a project; cloning or registering is add intake and uses the same trigger. -- `secrets-management` - load before project intake or initialization and before work that handles credentials or adds secret access to CI or deployment. - `stuck-crewmate-recovery` - load when the session-start digest reports an ordinary direct report's endpoint dead or its metadata has no window, or after a stale wake, looping pane, repeated confusion, an answered-by-brief question, an unresponsive crewmate, or a failed steer. - `secondmate-provisioning` - load before creating, seeding, validating, launching, handing backlog to, recovering, pushing inherited local material into, or retiring a secondmate home, and before editing `data/secondmates.md`. - `decision-hold-lifecycle` - load before treating an investigation or visual review as complete, before ending a visual review that exposed a decision, and when recording or routing the captain's answer. diff --git a/README.md b/README.md index d78814f58dd..4fcd84330a4 100644 --- a/README.md +++ b/README.md @@ -49,8 +49,7 @@ Launching a supported harness inside it instantiates your first mate - and makes - **Optional secondmates** - opt in to persistent second mates that run from isolated firstmate homes with their own `FM_HOME`, state, projects, and session lock, supervising project clones or a project-less firstmate-repo domain, kept on the primary firstmate version by guarded local fast-forwards and checked for live agent processes at session start. - **Event-driven, zero-token supervision** - a bash watcher sleeps on the fleet and wakes the first mate only when something needs you; verified primary harnesses also get a turn-end backstop that blocks or follows up on a blind stop when work is under way and supervision is not live. - **Optional X mode** - opt in with one local `.env` token so firstmate can answer your public `@myfirstmate` mentions, act on normal reversible mention requests through the same lifecycle as chat requests, acknowledge spawned work, and post up to three public-safe completion follow-ups within seven days for genuine milestones and the final outcome without changing non-X behavior; dry-run preview records would-be replies and dismissals locally before go-live. -- **Strict project boundary, guarded by construction** - the first mate is read-only over your projects except for the narrow guarded and captain-approved operations authorized by [hard rule 1](AGENTS.md#1-identity-and-prime-directives), including fleet sync's guarded safe branch pruning; crewmates make every other project change behind the configured merge authority. -- **Doppler by default** - the conditional [secrets-management policy](.agents/skills/secrets-management/SKILL.md) prefers secretless provider identity, otherwise scopes Doppler by project and environment, and validates declarations and rollout data through `bin/fm-secrets-check.sh`. +- **Guarded by construction** - the first mate is read-only over your projects except for the guarded paths authorized by [hard rule 1](AGENTS.md#1-identity-and-prime-directives), with fleet sync's safe branch pruning remaining part of the fleet-sync exception; crewmates make every project change behind the configured merge authority. - **Restart-proof** - all state lives on disk and in the active session backend (tmux by hard default, herdr or cmux when selected or auto-detected, zellij/orca when explicitly selected); kill the session anytime and the next one reconciles, including confirmed-dead secondmate agents, and carries on. Full detail on every feature lives in [docs/architecture.md](docs/architecture.md). @@ -59,7 +58,7 @@ Full detail on every feature lives in [docs/architecture.md](docs/architecture.m ### Requirements -- A verified primary agent harness: Claude Code, Grok, Pi, `pi-signed`, Codex, or OpenCode. +- A verified agent harness: Claude Code, Grok, Pi, Codex, or OpenCode. - Git and the GitHub CLI, authenticated through `gh auth login`. - The CLI and dependencies for your selected runtime backend; tmux is the reference default. @@ -68,8 +67,8 @@ Backend-specific setup is linked in [Documentation](#documentation). ### Recommended harnesses -**Claude Code, Grok, and Pi are equal co-primary recommendations** for running the primary firstmate session, with `pi-signed` supported as Pi's distinct signed-wrapper identity. -Claude Code uses a tracked Stop hook for tokenless watcher re-arm and rewake, Grok uses background-notify wake cycles, and Pi uses its tracked primary watcher extension. +**Claude Code, Grok, and Pi are equal co-primary recommendations** for running the primary firstmate session. +Claude Code and Grok use background-notify wake cycles; Pi uses its tracked primary watcher extension. All three have verified turn-end guard paths when launched with their documented setup. Pick whichever one matches your subscription and workflow. @@ -101,8 +100,6 @@ grok --trust ```sh pi -# or, when the signed wrapper is installed -FM_PI_HARNESS=pi-signed pi-signed ``` For Grok, `--trust` is needed once per clone so project hooks and the turn-end guard load; `/hooks-trust` inside Grok works too. @@ -172,17 +169,10 @@ Claude and grok use the slash form shown here; codex uses the same names with `$ | ------------------ | -------------------------------------------------------------------------------------------------------------------------------------------- | | `/afk` | Enter away-mode supervision: the sub-supervisor self-handles routine notifications in bash, escalates captain-relevant events and bounded declared-external-wait rechecks as batched digests, and actively alerts if delivery gets stuck while you step away | | `/ahoy` | Recap visible session events since the prior real captain message plus visibly unanswered captain decisions, falling back to Bearings when invoked as the session's first real captain message | -| `/bearings` | Generate a concise four-section chat digest from bounded local fleet and registered-secondmate state; use `/bearings file` to also replace today's dated report in `data/`, and add `include PRs` when live PR enrichment is wanted | +| `/bearings` | Generate a standalone current-status report from bounded local fleet and registered-secondmate state, with live PR enrichment only when requested, written to a dated file in `data/` and surfaced concisely in chat; read-mostly, mutates no task state | | `/updatefirstmate` | Self-update the running firstmate and its secondmates to the latest from origin with fast-forward-only pulls, then re-read instructions and nudge secondmates | | `/stow` | Sweep the session for uncaptured durable knowledge, route each finding to its disk home per AGENTS.md, file undone next steps to the backlog, and report what is now safe to reset | -Bearings invocation examples: - -- `/bearings` returns the fresh four-section digest in chat only. -- `/bearings include PRs` keeps chat-only mode and opts into live PR enrichment. -- `/bearings file` replaces today's `data/status-report-<YYYY-MM-DD>.md` from scratch and links it from the four-section chat digest. -- `/bearings file include PRs` combines the dated report with live PR enrichment. - Agent-only reference skills live under `.agents/skills/` and are loaded by firstmate at the trigger points named in [`AGENTS.md`](AGENTS.md). ### Two-tier skill layout @@ -200,20 +190,18 @@ Firstmate's skills live in two separate places with different audiences: - [docs/architecture.md](docs/architecture.md) - maintainer architecture for the crew, supervision, worktrees, secondmates, and project modes. - [docs/configuration.md](docs/configuration.md) - environment variables, `FM_HOME`, runtime backend selection, optional X mode, the files you set, and harness support. - [docs/calm.md](docs/calm.md) - current Pi `/calm` behavior and supported presentation limits. -- [docs/moshi-mobile-review.md](docs/moshi-mobile-review.md) - host-local Moshi Pro Preview, Diff, Chat View, and private mobile-review fallbacks. - [docs/wedge-alarm.md](docs/wedge-alarm.md) - configure the active alert for an away-mode escalation delivery that gets stuck. -- [docs/tmux-backend.md](docs/tmux-backend.md) - setup guide for the tmux reference backend: prerequisites, attaching, and watching crew windows. -- [docs/herdr-backend.md](docs/herdr-backend.md) - setup guide for the experimental herdr backend, plus its verification notes and known gaps. -- [docs/zellij-backend.md](docs/zellij-backend.md) - setup guide for the experimental zellij backend, plus its verification notes and known gaps. -- [docs/orca-backend.md](docs/orca-backend.md) - setup guide for the experimental Orca backend, plus its lifecycle notes and known gaps. -- [docs/cmux-backend.md](docs/cmux-backend.md) - setup guide for the experimental cmux backend, plus its verification notes and known gaps. -- [docs/codex-app-backend.md](docs/codex-app-backend.md) - Codex App backend boundary, evidence, and rollout contract. -- [docs/gitlab-merge-watch.md](docs/gitlab-merge-watch.md) - how the merge watch follows a GitLab merge request on any instance, and the evidence behind it. -- [docs/promotion-ladder.md](docs/promotion-ladder.md) - the promotion contract for projects with deployable applications. -- [docs/turnend-guard.md](docs/turnend-guard.md) - the primary session's structural "no turn ends blind" backstop: verified per-harness hook mechanisms, scoping, loop safety, and fail-open tradeoffs. +- [docs/tmux-backend.md](docs/tmux-backend.md) - current setup and limits for the tmux reference backend. +- [docs/herdr-backend.md](docs/herdr-backend.md) - current setup, safety boundaries, and limits for the experimental Herdr backend. +- [docs/zellij-backend.md](docs/zellij-backend.md) - current setup and limits for the experimental Zellij backend. +- [docs/orca-backend.md](docs/orca-backend.md) - current setup and limits for the experimental Orca backend. +- [docs/cmux-backend.md](docs/cmux-backend.md) - current setup, socket security, and limits for the experimental cmux backend. +- [docs/codex-app-backend.md](docs/codex-app-backend.md) - the current blocked Codex App backend boundary and rollout contract. - [docs/verification/runtime-backends.md](docs/verification/runtime-backends.md) - active maintainer verification for runtime backend guarantees. +- [docs/gitlab-merge-watch.md](docs/gitlab-merge-watch.md) - maintainer verification for GitLab merge watching on arbitrary instances. +- [docs/turnend-guard.md](docs/turnend-guard.md) - the primary session's current "no turn ends blind" backstop, scope, loop safety, and compatibility limits. - [docs/verification/supervision.md](docs/verification/supervision.md) - active maintainer verification for session-start, guard, continuity, and wedge integrations. -- [docs/supervision-protocols/](docs/supervision-protocols/) - rendered primary-harness watcher protocols for Claude, Codex, OpenCode, Pi and `pi-signed`, Grok, and unknown harness fallback. +- [docs/supervision-protocols/](docs/supervision-protocols/) - rendered primary-harness watcher protocols for Claude, Codex, OpenCode, Pi, Grok, and unknown harness fallback. - [docs/scripts.md](docs/scripts.md) - the `bin/` toolbelt reference. - [docs/documentation-audiences.md](docs/documentation-audiences.md) - documentation audiences and the machine-checked placement boundary. - [`AGENTS.md`](AGENTS.md) - the distro's always-loaded operating contract and routing index for conditional procedures. diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 940a2528309..8e06d169e91 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -38,7 +38,7 @@ # silently pass as a gate skip. # --jobs N run the selected scripts with up to N concurrent workers. # Default is 1 (serial). N>1 is allowed only when every -# selected script is in the proven-isolated set +# selected script is in the Phase 2 proven-isolated set # (bin/fm-test-isolation-proof.sh --list). Cap is 8. Stateful # families never schedule under --jobs. # -h, --help print this header @@ -118,21 +118,17 @@ now_ms() { family_for_basename() { case "$1" in fm-arm-pretool-check.test.sh|fm-ask-user-authority.test.sh|fm-brief.test.sh|\ - fm-calm-pi-extension.test.sh|fm-cd-pretool-check.test.sh|\ + fm-calm-pi-extension.test.sh|fm-captain-translation-contract.test.sh|fm-cd-pretool-check.test.sh|\ fm-composer-ghost.test.sh|fm-composer-lib.test.sh|\ - fm-crew-state.test.sh|fm-decision-hold-lifecycle.test.sh|\ - fm-documentation-audiences.test.sh|fm-ensure-agents-md.test.sh|fm-grok-harness.test.sh|\ - fm-herdr-lab.test.sh|fm-kimi-harness.test.sh|fm-lint.test.sh|fm-secrets-check.test.sh|\ - fm-captain-translation-contract.test.sh|\ fm-continuity-pretool-check.test.sh|fm-crew-state.test.sh|fm-decision-hold-lifecycle.test.sh|\ - fm-dispatch-select.test.sh|fm-ensure-agents-md.test.sh|fm-grok-harness.test.sh|\ - fm-herdr-lab.test.sh|fm-instruction-owners.test.sh|fm-lint.test.sh|fm-secrets-check.test.sh|\ + fm-dispatch-select.test.sh|fm-documentation-audiences.test.sh|fm-ensure-agents-md.test.sh|fm-grok-harness.test.sh|\ + fm-herdr-lab.test.sh|fm-instruction-owners.test.sh|fm-lint.test.sh|\ fm-install-herdr.test.sh|fm-nm-test-contract.test.sh|fm-no-mistakes-ownership.test.sh|\ fm-operational-input.test.sh|fm-pi-primary-types.test.sh|\ - fm-send-popup-settle.test.sh|fm-send-settle.test.sh|\ + fm-send-popup-settle.test.sh|fm-send-settle.test.sh|fm-stow-contract.test.sh|\ fm-subagent-pretool-check.test.sh|\ fm-supervision-instructions.test.sh|fm-tmux-submit-busy.test.sh|fm-transition-lib.test.sh|\ - fm-test-run.test.sh|fm-test-isolation-proof.test.sh|fm-toolchain-mirror.test.sh) + fm-test-run.test.sh|fm-test-isolation-proof.test.sh) printf '%s\n' pure-contract-unit ;; fm-daemon.test.sh|fm-guard-stale-banner.test.sh|fm-pi-watch-extension.test.sh|\ @@ -144,13 +140,11 @@ family_for_basename() { fm-afk-inject-herdr-e2e.test.sh|fm-afk-launch.test.sh|fm-backend-autodetect-smoke.test.sh|\ fm-backend-herdr-eventwait-smoke.test.sh|fm-backend-herdr-presentation-e2e.test.sh|\ fm-backend-herdr-prune-safety-e2e.test.sh|fm-backend-herdr-respawn-idem-e2e.test.sh|\ - fm-herdr-session-cleanup-e2e.test.sh|\ fm-backend-herdr-smoke.test.sh|fm-backend-herdr-workspace-per-home-e2e.test.sh) printf '%s\n' real-herdr-gated ;; fm-backlog-handoff.test.sh|fm-secondmate-harness.test.sh|fm-secondmate-lifecycle-e2e.test.sh|\ fm-secondmate-liveness.test.sh|fm-secondmate-safety.test.sh|fm-secondmate-sync.test.sh|\ - fm-startup-memory-budget.test.sh|\ fm-send-secondmate-marker.test.sh|fm-shared-captain-inheritance.test.sh) printf '%s\n' secondmate ;; @@ -159,16 +153,15 @@ family_for_basename() { fm-update.test.sh) printf '%s\n' session-bootstrap ;; - fm-afk-pi-herdr-return-e2e.test.sh|\ + fm-afk-pi-herdr-return-e2e.test.sh|fm-claude-continuity-live-e2e.test.sh|\ fm-codex-continuity-live-e2e.test.sh|fm-grok-continuity-live-e2e.test.sh|\ - fm-grok-stop-live-e2e.test.sh|fm-opencode-primary-live-e2e.test.sh|fm-pi-primary-live-e2e.test.sh|\ + fm-opencode-primary-live-e2e.test.sh|fm-pi-primary-live-e2e.test.sh|\ fm-send-secondmate-marker-herdr-e2e.test.sh) printf '%s\n' live-harness-optin ;; fm-backend-herdr.test.sh|fm-backend-tmux-smoke.test.sh|fm-backend.test.sh|\ - fm-herdr-session-cleanup.test.sh|fm-send-strict.test.sh|fm-spawn-batch.test.sh|\ - fm-spawn-dispatch-profile.test.sh|fm-spawn-worktree-settle.test.sh|\ - fm-teardown-endpoint-safety.test.sh) + fm-send-strict.test.sh|fm-spawn-batch.test.sh|fm-spawn-dispatch-profile.test.sh|\ + fm-spawn-worktree-settle.test.sh) printf '%s\n' backend-dispatch ;; fm-pr-check-security.test.sh|fm-pr-merge.test.sh|fm-review-diff.test.sh|\ @@ -234,7 +227,7 @@ real-herdr-gated EOF } -# Exact proven-isolated candidate set (same paths as +# Exact Phase 2 proven-isolated candidate set (same paths as # bin/fm-test-isolation-proof.sh --list). Do not expand without a new concurrent # isolation proof archive. list_proven_isolated() { @@ -242,15 +235,20 @@ list_proven_isolated() { tests/fm-arm-pretool-check.test.sh tests/fm-backend-herdr.test.sh tests/fm-brief.test.sh +tests/fm-captain-translation-contract.test.sh tests/fm-cd-pretool-check.test.sh tests/fm-composer-ghost.test.sh tests/fm-composer-lib.test.sh tests/fm-crew-state.test.sh tests/fm-decision-hold-lifecycle.test.sh +tests/fm-dispatch-select.test.sh tests/fm-ensure-agents-md.test.sh tests/fm-grok-harness.test.sh tests/fm-herdr-lab.test.sh +tests/fm-instruction-owners.test.sh tests/fm-lint.test.sh +tests/fm-nm-test-contract.test.sh +tests/fm-no-mistakes-ownership.test.sh tests/fm-pi-primary-types.test.sh tests/fm-pr-merge.test.sh tests/fm-review-diff.test.sh @@ -258,6 +256,7 @@ tests/fm-send-popup-settle.test.sh tests/fm-send-settle.test.sh tests/fm-send-strict.test.sh tests/fm-spawn-batch.test.sh +tests/fm-stow-contract.test.sh tests/fm-supervision-instructions.test.sh tests/fm-test-run.test.sh tests/fm-tmux-submit-busy.test.sh @@ -266,41 +265,48 @@ tests/fm-x-mode.test.sh EOF } -# Portable parallel shard 1: LPT balance of the proven-isolated set using the -# current concurrent-proof durations in docs/fm-test-isolation-proof.json. -# Execution order is longest first so wall-clock stays near the balanced sum. +# Portable parallel shard 1: LPT balance of the proven-isolated set using +# Phase 1 serial duration averages from CI timing artifacts on main after +# #825/#832/#834 (docs/fm-test-portable-shards.md). Execution order is longest +# first so wall-clock stays near the balanced sum. list_portable_parallel_1() { cat <<'EOF' -tests/fm-x-mode.test.sh +tests/fm-arm-pretool-check.test.sh tests/fm-cd-pretool-check.test.sh -tests/fm-decision-hold-lifecycle.test.sh +tests/fm-backend-herdr.test.sh +tests/fm-pr-merge.test.sh tests/fm-test-run.test.sh -tests/fm-composer-ghost.test.sh -tests/fm-grok-harness.test.sh -tests/fm-lint.test.sh -tests/fm-pi-primary-types.test.sh +tests/fm-send-popup-settle.test.sh tests/fm-review-diff.test.sh tests/fm-brief.test.sh +tests/fm-dispatch-select.test.sh +tests/fm-ensure-agents-md.test.sh +tests/fm-instruction-owners.test.sh +tests/fm-pi-primary-types.test.sh tests/fm-transition-lib.test.sh +tests/fm-composer-lib.test.sh +tests/fm-stow-contract.test.sh EOF } # Portable parallel shard 2: the complementary LPT half of the proven set. list_portable_parallel_2() { cat <<'EOF' -tests/fm-backend-herdr.test.sh -tests/fm-arm-pretool-check.test.sh -tests/fm-crew-state.test.sh +tests/fm-decision-hold-lifecycle.test.sh +tests/fm-x-mode.test.sh tests/fm-herdr-lab.test.sh -tests/fm-pr-merge.test.sh -tests/fm-send-popup-settle.test.sh +tests/fm-crew-state.test.sh +tests/fm-grok-harness.test.sh +tests/fm-spawn-batch.test.sh +tests/fm-send-strict.test.sh tests/fm-tmux-submit-busy.test.sh +tests/fm-composer-ghost.test.sh tests/fm-send-settle.test.sh -tests/fm-send-strict.test.sh -tests/fm-spawn-batch.test.sh tests/fm-supervision-instructions.test.sh -tests/fm-ensure-agents-md.test.sh -tests/fm-composer-lib.test.sh +tests/fm-lint.test.sh +tests/fm-nm-test-contract.test.sh +tests/fm-captain-translation-contract.test.sh +tests/fm-no-mistakes-ownership.test.sh EOF } @@ -615,11 +621,6 @@ families_for_changed_path() { printf '%s\n' backend-dispatch printf '%s\n' pure-contract-unit ;; - bin/fm-herdr-session-cleanup.sh) - printf '%s\n' session-bootstrap - printf '%s\n' real-herdr-gated - printf '%s\n' backend-dispatch - ;; bin/backends/zellij*|tests/zellij-test-safety.sh) printf '%s\n' zellij printf '%s\n' backend-dispatch @@ -650,10 +651,6 @@ families_for_changed_path() { printf '%s\n' live-harness-optin printf '%s\n' afk ;; - bin/fm-startup-memory-budget.sh|bin/fm-startup-memory-budget-lib.sh) - printf '%s\n' secondmate - printf '%s\n' session-bootstrap - ;; bin/fm-secondmate*|bin/fm-home-seed.sh|bin/fm-backlog-handoff.sh|\ bin/fm-config-inherit-lib.sh|bin/fm-config-push.sh|bin/fm-shared*) printf '%s\n' secondmate @@ -667,7 +664,7 @@ families_for_changed_path() { bin/fm-x-*|bin/fm-check*) printf '%s\n' pr-forge ;; - bin/fm-spawn.sh|bin/fm-send.sh|bin/fm-harness.sh|\ + bin/fm-spawn.sh|bin/fm-send.sh|bin/fm-dispatch-select.sh|bin/fm-harness.sh|\ bin/fm-peek.sh|bin/fm-composer*) printf '%s\n' backend-dispatch printf '%s\n' pure-contract-unit @@ -681,10 +678,7 @@ families_for_changed_path() { # lane's contract coverage re-runs. printf '%s\n' real-herdr-gated ;; - bin/fm-toolchain-mirror.sh) - printf '%s\n' pure-contract-unit - ;; - bin/fm-lint.sh|bin/fm-install-shellcheck.sh|bin/fm-secrets-check.sh|bin/fm-secrets-check.mjs|\ + bin/fm-lint.sh|bin/fm-install-shellcheck.sh|\ bin/fm-brief.sh|bin/fm-ensure-agents-md.sh|bin/fm-crew-state.sh|\ bin/fm-decision-hold.sh|bin/fm-supervision*|bin/fm-transition-lib.sh|\ bin/fm-tmux-lib.sh|bin/fm-marker-lib.sh|bin/fm-operational-input.sh|bin/fm-tasks-axi-lib.sh|\ @@ -692,16 +686,15 @@ families_for_changed_path() { bin/fm-ff-lib.sh|bin/fm-gotmp*|bin/*pretool*) printf '%s\n' pure-contract-unit ;; + .agents/skills/*/SKILL.md) + printf '%s\n' pure-contract-unit + ;; .github/workflows/ci.yml|.no-mistakes.yaml) printf '%s\n' pure-contract-unit printf '%s\n' real-herdr-gated ;; docs/fm-test-portable-shards.md|docs/fm-test-isolation-proof.md|\ - docs/fm-test-isolation-proof.json|docs/secrets-*|docs/examples/project-secrets-policy.json|\ - docs/examples/doppler-*-job.yml|.agents/skills/secrets-management/SKILL.md) - printf '%s\n' pure-contract-unit - ;; - .agents/skills/*/SKILL.md) + docs/fm-test-isolation-proof.json) printf '%s\n' pure-contract-unit ;; .github/*|.tasks.toml|AGENTS.md|CLAUDE.md|CONTRIBUTING.md|\ diff --git a/docs/architecture.md b/docs/architecture.md index bf8b5cb3ec1..eed3f2c9358 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -11,7 +11,6 @@ firstmate's always-loaded operating contract and routing index for conditional p A zero-token bash watcher (`bin/fm-watch.sh`) sleeps on the fleet, classifies detected wakes in bash, and wakes the first mate only when something is actionable. Actionable wakes include captain-relevant status signals, no-verb signals whose crew is not provably working, authenticated check output such as PR merge polling or an X-mode mention, stale panes whose crew is not provably working whether their status log looks terminal or non-terminal, provably-working stale panes that persist past `FM_STALE_ESCALATE_SECS`, declared external waits that remain paused past `FM_PAUSE_RESURFACE_SECS`, and heartbeat backstop hits. Repeated provably-working stale escalations on the same unchanged pane add an escalation count to the wake reason and, at `FM_WEDGE_DEMAND_INSPECT_COUNT`, a `demand-deep-inspection` marker. -A busy pane is otherwise exempt from staleness, but only until its latest `state/<id>.turn-ended` marker reaches `FM_BUSY_TURN_MAX_SECS`, or its `state/<id>.meta` spawn record reaches that age before any turn completes; past that bound it is routed through the same wedge escalation, with the identical reason, escalation count, and `demand-deep-inspection` marker, for inspection only - never an automatic interrupt, signal, or restart. Those actionable wakes are written to a durable local queue (`state/.wake-queue`) before detector state advances, so a missed process exit can be recovered by draining the queue. When a canonical validated PR poll returns exactly `merged`, the watcher appends that durable notification before publishing a private receipt bound to the poll's registration, bytes, file identities, metadata, provider, URL, and task ID. The receipt makes retirement safely retryable across restarts: fixed-path recovery revalidates the same evidence, removes the runnable check first, removes its registration and data sidecars, removes the receipt last, and preserves task metadata including `pr=` and `pr_head=`. @@ -36,7 +35,7 @@ The most recent recognized ci log marker wins, so checks-green monitoring report Only when no matching run exists does it fall back to the pane busy-signature and then a status-log event whose verb maps to a recognized run-state; a dead pane without a run reports unknown instead of trusting a stale log. Decision-only events such as `resolved` never become current state or leak their prose into the current-state detail. In that status-log fallback, a declared external wait reports the distinct `paused` state with its reason. -For herdr, that pane fallback trusts a native `busy` verdict outright, but corroborates native `idle` or unknown verdicts against the recorded harness's rendered busy signature before deciding the crew is not working. +For herdr, that pane fallback trusts a native `busy` verdict outright, but corroborates native `idle` or unknown verdicts against the rendered busy signature before deciding the crew is not working. For whole-fleet read-only review, `bin/fm-fleet-snapshot.sh --json` emits schema `fm-fleet-snapshot.v1` from the backlog, task metadata, current crew state, endpoint probes, PR/report pointers, scout reports, bounded current summaries from registered secondmate homes, and secondmate return-channel guidance. `bin/fm-fleet-view.sh` renders that snapshot as Markdown for humans, while `bin/fm-bearings-snapshot.sh` provides the bounded bearings projection, so both views consume one structured contract instead of reparsing raw fleet files. The script header owns the exact JSON schema. @@ -55,14 +54,13 @@ The default path remains local-only; live GitHub enrichment exists only behind t Optional X mode integrates with the watcher only after explicit opt-in; [configuration.md](configuration.md#x-mode-env) owns its generated-artifact and dispatch mechanics. At session start, `bin/fm-session-start.sh` emits exactly one primary-harness supervision block rendered by `bin/fm-supervision-instructions.sh` from `docs/supervision-protocols/`. -That block owns the live wait shape for the running primary harness: Claude's Stop `asyncRewake` hook owns tokenless re-arm cycles, Grok uses background-notify cycles, Codex uses bounded foreground checkpoints, Pi and pi-signed use the same two tracked primary extensions, and OpenCode uses its TUI plugin. +That block owns the live wait shape for the running primary harness: Claude and Grok use background-notify cycles, Codex uses bounded foreground checkpoints, Pi uses its two tracked primary extensions, and OpenCode uses its TUI plugin. `bin/fm-watch-arm.sh` remains the verified arm wrapper for protocols that call it; it forks the watcher as a tracked child, verifies it is genuinely alive with a fresh liveness beacon, and prints an honest `started`, `attached`, or nonzero `FAILED` status. On `attached` it stays live across identity-matched successors, and an unexplained clean child close either attaches to a verified healthy successor or becomes the typed nonzero `watcher: FAILED - cycle ended without an actionable reason` result. The arm layer records one bounded lifecycle row per observed cycle in `state/.watch-cycle-exits.log`; `state/.watch-triage.log` remains exclusively the absorbed-wake debug log. Pi and OpenCode verify session-lock ownership and launch one singleton successor from their child-close handlers before delivering an actionable wake prompt, with bounded exponential retry for failed restoration. -Claude's `bin/fm-claude-stop-autoarm.sh` hook fires on every Stop and, when the home is eligible and still needs supervision, claims one home-scoped cycle, foregrounds the arm wrapper, and translates an actionable close or typed failure into one exit-2 rewake. -[`watcher-continuity.md`](watcher-continuity.md) owns Claude's residual active-turn coverage and watcher-status command-gating boundary. -The existing turn-end guard remains the final backstop for all five harness-engine protocols, with pi-signed sharing Pi's protocol and the `--claude` mode cooperating with the auto-arm claim. +Claude keeps its tracked background-task protocol and adds a narrow PreToolUse continuity gate that allows drain, arm recovery, and fail-closed teardown while refusing only other fleet commands when tasks are in flight and no identity-matched live watcher holds the home lock. +The existing turn-end guard is unchanged and remains the final backstop for all five harness protocols. Its `--restart` mode signals only the watcher recorded in the current home's `state/.watch.lock`, so restarting one home cannot kill sibling secondmate watchers. A pull-based guard (`bin/fm-guard.sh`) warns through supervision tool output if the primary checkout is tangled, or if tasks are in flight and that watcher stops running or queued wakes are waiting to be drained. The drain script calls that guard after emptying the queue, which avoids repeating the queued-wakes warning for records it just consumed while still warning on stale watcher liveness. @@ -90,20 +88,18 @@ On an unmarked return, `bin/fm-afk-return.sh` owns ordered shutdown, durable cat The runtime backend is the session-provider layer below firstmate's scripts. It owns task endpoint creation, bounded capture, text/key sends, current-path reads for spawn-time worktree discovery when the backend does not create the worktree itself, live-window fallback lookup, agent-process liveness probes where verified, and endpoint teardown. -`bin/fm-backend.sh` centralizes backend selection, `state/<id>.meta` helpers, metadata-only cleanup identity validation, selector resolution, and operation dispatch; `bin/backends/tmux.sh` is the verified reference adapter ([`docs/tmux-backend.md`](tmux-backend.md)), and `bin/backends/herdr.sh` (P2), `bin/backends/zellij.sh` (P3), `bin/backends/orca.sh` (P4), and `bin/backends/cmux.sh` (P5) are experimental task-spawn adapters. +`bin/fm-backend.sh` centralizes backend selection, `state/<id>.meta` helpers, selector resolution, and operation dispatch; `bin/backends/tmux.sh` is the verified reference adapter ([`docs/tmux-backend.md`](tmux-backend.md)), and `bin/backends/herdr.sh` (P2), `bin/backends/zellij.sh` (P3), `bin/backends/orca.sh` (P4), and `bin/backends/cmux.sh` (P5) are experimental task-spawn adapters. New spawns select a backend from `--backend`, then `FM_BACKEND`, then local `config/backend`, then runtime auto-detection from `$TMUX`, `HERDR_ENV=1`, or cmux runtime signals, then default `tmux`. Runtime auto-detection is innermost-first: `$TMUX` wins over `HERDR_ENV=1`, which wins over cmux's primary `CMUX_WORKSPACE_ID` marker and documented fallback signals; auto-detected herdr or cmux prints a one-time opt-out notice, auto-detected tmux stays silent, and zellij and orca are never auto-detected (only explicit selection). Unknown backend names fail loudly. For compatibility, default tmux tasks do not write `backend=tmux`; every reader treats a missing `backend=` field as `tmux`. -`fm-watch.sh` polls each window's backend for a busy state: tmux, zellij, orca, and cmux have no native primitive and always report unknown, so their pane-tail fallback matches only the recorded harness's verified signature; herdr's `agent.get` semantic state (working/idle/done/blocked) is consulted first for stale detection, with unknown native states using the same harness-scoped fallback. -This scope prevents cross-harness false positives such as Kimi's rotating idle tip `ctrl+c: cancel` borrowing Grok's busy token, and keeps Claude's broader elapsed-spinner shape from matching ordinary output in other panes. -Unknown supplied harnesses match no default signature, while callers that have no harness metadata retain the historical combined-pattern compatibility fallback. +`fm-watch.sh` polls each window's backend for a busy state: tmux, zellij, orca, and cmux have no native primitive and always report unknown, preserving the original pane-tail-regex detection unchanged; herdr's `agent.get` semantic state (working/idle/done/blocked) is consulted first for stale detection, with unknown native states falling back to the same regex. That poll loop is the default event source for backends with no native push events, so this stays an extraction of the abstraction rather than a watcher rewrite. For capable Herdr sessions, the same watcher replaces its terminal sleep with a bounded native event wait that immediately surfaces `blocked`; [Push events and polling fallback](herdr-backend.md#push-events-and-polling-fallback) owns the current mechanism and capability gates, while [runtime backend verification](verification/runtime-backends.md#native-blocked-event) owns the active evidence. The deeper session-start agent-process liveness probe is separate from that busy-state poll: tmux and Herdr have verified classifiers for secondmate recovery, Zellij remains unverified, and Orca and cmux do not support secondmate spawns. Herdr is experimental and can be selected explicitly or by runtime auto-detection: Treehouse remains its worktree provider, [`herdr-backend.md`](herdr-backend.md) owns current setup and safety limits, and [`verification/runtime-backends.md`](verification/runtime-backends.md#herdr) owns active empirical evidence. Herdr's durable default container shape is workspace-per-home plus tab-per-task: the primary home uses workspace label `firstmate`, secondmate homes use `2ndmate-<secondmate-id>`, and recovery/list-live scopes to the current `FM_HOME`'s workspace. -Its optional default-off presentation projection may place one clean new task in a disposable workspace without changing endpoint authority or lifecycle ownership; [Optional presentation spaces](herdr-backend.md#optional-presentation-spaces) owns that conditional design and its narrow home-local restored-shell cleanup at locked session start. +Its optional default-off presentation projection may place one clean new task in a disposable workspace without changing endpoint authority or lifecycle ownership; [Optional presentation spaces](herdr-backend.md#optional-presentation-spaces) owns that conditional design. Zellij is experimental and selected only explicitly: Treehouse remains its worktree provider, [`zellij-backend.md`](zellij-backend.md) owns current setup and limits, and [`verification/runtime-backends.md`](verification/runtime-backends.md#zellij) owns active empirical evidence. Zellij's container shape is simpler than herdr's: one shared `firstmate` session, one tab per task, with no per-home workspace split; visible tab titles are scoped by the active home label plus a short hash of the resolved `FM_ROOT` path. Orca is experimental and selected only explicitly: Orca owns both worktree and terminal lifecycle, records `orca_worktree_id=` and `terminal=`, and removes worktrees through `orca worktree rm` only after the usual firstmate teardown checks pass. @@ -143,13 +139,13 @@ The intake and authority contract in `AGENTS.md` owns when separate scout resear ## Dispatch profiles Crewmate and scout dispatch can stay on the static crewmate harness resolved by `config/crew-harness`, or it can use local dispatch profiles in `config/crew-dispatch.json`. -The dispatch file is intentionally judgment-based: firstmate reads the natural-language rules at intake, chooses the best matching rule, resolves profile arrays itself from current quota output under the `AGENTS.md` section 4 intake boundary and the `quota-array-dispatch` selection procedure, and passes only concrete `--harness`, `--model`, and `--effort` axes to `fm-spawn.sh`. -The shell scripts validate the JSON shape and verified harness/effort combinations, but they do not parse task intent, match natural-language rules, or own array selection. +The dispatch file is intentionally judgment-based: firstmate reads the natural-language rules at intake, chooses the best matching rule, resolves that rule directly or through a supported selector, and passes only concrete `--harness`, `--model`, and `--effort` axes to `fm-spawn.sh`. +The shell scripts validate the JSON shape and verified harness/effort combinations, and `fm-dispatch-select.sh` owns quota-aware array selection plus OS-backed random fallback, but they do not parse task intent or match the natural-language rules. The session-start bootstrap step keeps valid dispatch configuration silent unless verbose facts are enabled and surfaces a concise invalid-config line when validation fails. When the file exists, `fm-spawn.sh` refuses crewmate and scout launches without an explicit harness, so `config/crew-harness` is only automatic when no dispatch profile file is active. Secondmate launches are exempt because they resolve the secondmate harness and any optional secondmate model or effort tokens instead. Unsupported effort values are still recorded in task meta when passed to `fm-spawn.sh`, but the launch template omits any effort flag that the selected harness does not accept. -That keeps spawn launch compatible across claude, codex, grok, pi, opencode, and kimi while preserving the requested profile for later audit. +That keeps spawn launch compatible across claude, codex, grok, pi, and opencode while preserving the requested profile for later audit. ## Optional secondmates diff --git a/docs/calm.md b/docs/calm.md index 8d63b6d0b56..6a2c1d14b9c 100644 --- a/docs/calm.md +++ b/docs/calm.md @@ -18,12 +18,6 @@ Pi's supported presentation API does not expose a global transcript filter. Expanded reasoning and its reserved spacing, built-in tool images, user-bash rows, skill and summary rows, generic status notices, and arbitrary custom-tool or extension rows remain visible. These are supported-API boundaries rather than hidden-content failures. -## Pi compatibility - -Calm has no numeric Pi version minimum or maximum and never refuses Pi solely because its version is newer than a previously verified version. -The collapsed-thinking and operational-user-row presentation adapters probe the exact Pi API seam they patch when Calm loads. -If Pi removes one of those seams, Calm logs a diagnostic naming the unavailable adapter and skips only that adapter; `/calm`, the other adapter, and unrelated Pi extensions remain available. - [`calm-mode-feasibility.md`](calm-mode-feasibility.md) owns the version-scoped renderer taxonomy and empirical evidence. [`configuration.md`](configuration.md#pi-calm-preference-configcalm) owns the persisted preference file and resolution rules. `.pi/extensions/lib/fm-calm-visibility.ts` owns the visibility policy, and `.pi/extensions/lib/fm-calm-operational-user-layout.ts` owns the zero-height operational-user row adapter. diff --git a/docs/configuration.md b/docs/configuration.md index 94799d0e8ea..abb8bfe9d4b 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -12,7 +12,7 @@ This section is the single owner of the top-level operational-home layout; produ The tracked code root contains the shared instruction, skill, documentation, workflow, and `bin/` surfaces, while each effective `FM_HOME` contains private operational directories. `data/` holds durable private fleet records such as the project and secondmate registries, captain preferences, optional shared captain preferences, learnings, backlog, briefs, and scout reports. `state/` holds volatile runtime records such as task metadata, append-only status events, endpoint signals, watcher and wake-queue coordination, away-mode state, generated X-mode artifacts, private secondmate config-reread generations with their retry and quarantine state, and parent-owned secondmate pending-reply records under `state/pending-replies/` (`bin/fm-pending-reply-lib.sh`). -`config/` holds local gitignored operating choices, and `projects/` holds the local project clones that Firstmate reads but changes only through the narrow guarded and concrete captain-approved exceptions in `AGENTS.md`. +`config/` holds local gitignored operating choices, and `projects/` holds the local project clones that Firstmate reads but changes only through the guarded exceptions in `AGENTS.md`. `bin/fm-spawn.sh` owns the base task-metadata fields it emits, while the runtime-backend section below owns backend-specific fields and selector interpretation. The producing PR and X helpers own the fields they append, `bin/fm-classify-lib.sh` owns status-event vocabulary, and `bin/fm-crew-state.sh` owns current-state reconciliation. @@ -67,7 +67,6 @@ A zellij spawn additionally version-gates against the installed `zellij` binary' A cmux spawn additionally version-gates against the installed `cmux` binary's version, requires `jq`, and requires the control socket to be reachable and accessible (see [`docs/cmux-backend.md`](cmux-backend.md) "Setup" for the one-time socket-access configuration this needs; Automation mode is the recommended socket control mode, with Password mode supported via `config/cmux-socket-password`), refusing loudly and non-retryably on a `cmuxOnly`/unauthenticated socket. A backend spawn refusal from a missing dependency, version gate, or unauthenticated socket is terminal for that selected backend; firstmate surfaces it as a blocker instead of silently retrying another backend. Task meta records `backend=` only for a non-default backend; an absent `backend=` means `tmux`, preserving existing default-path meta files. -Every new task records `endpoint_task_id=` as the cleanup binding between the metadata filename and its opaque runtime endpoint. A herdr task additionally records `herdr_session=`, `herdr_workspace_id=`, `herdr_tab_id=`, and `herdr_pane_id=`. A zellij task additionally records `zellij_session=`, `zellij_tab_id=`, and `zellij_pane_id=`. An Orca task additionally records `orca_worktree_id=` and `terminal=`, with `window=fm-<id>` kept as the shared firstmate alias. @@ -78,12 +77,10 @@ Otherwise an exact task id matching `state/<id>.meta` wins before the legacy `fm A metadata-routed selector returns the recorded backend target (`terminal=` for Orca, otherwise `window=`), and matching explicit targets can still recover the recorded backend when metadata contains the same endpoint. Only metadata-routed task selectors carry secondmate-marker and Codex-harness context; explicit endpoint escape hatches do not. These five sentences are the single owner of the task-selector vocabulary; backend guides and other documents point here instead of restating the resolution order. -`fm-teardown.sh <id>` takes a task id directly and validates the complete metadata-only endpoint identity before any runtime dispatch or cleanup mutation. -Missing, empty, duplicate, malformed, backend-inconsistent, or task-mismatched endpoint records are preserved and refused. -Legacy tmux metadata remains cleanup-compatible when its exact window name is `fm-<id>`; opaque non-tmux endpoints require their recorded `endpoint_task_id=` binding. +`fm-teardown.sh <id>` takes a task id directly and uses the same recorded backend target fields after loading `state/<id>.meta`. By default, Herdr workspaces are derived from `FM_HOME`: the primary home uses `firstmate`, and a secondmate home marked by `.fm-secondmate-home` uses `2ndmate-<secondmate-id>`. The default-container spawn, list-live, and recovery paths read that label from the active home, so a secondmate's own crewmates stay inside that secondmate home's herdr space. -The optional local `config/herdr-presentation-spaces` presence flag instead enables Herdr's default-off disposable single-task visual projection; [Optional presentation spaces](herdr-backend.md#optional-presentation-spaces) owns its behavior, safety limits, recovery contract, and narrow locked session-start cleanup of exact restored idle-shell children. +The optional local `config/herdr-presentation-spaces` presence flag instead enables Herdr's default-off disposable single-task visual projection; [Optional presentation spaces](herdr-backend.md#optional-presentation-spaces) owns its behavior, safety limits, and recovery contract. The flag is default-off and inherited into secondmate homes under the primary-authoritative contract owned by [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md). For normal herdr operations, `HERDR_SESSION` selects the named session, but destructive test cleanup must not rely on `HERDR_SESSION` alone. Use the explicit guarded cleanup path described in [`docs/herdr-backend.md`](herdr-backend.md) instead of `herdr server stop`. @@ -93,7 +90,7 @@ Use the guarded cleanup path described in [`docs/zellij-backend.md`](zellij-back cmux has no session layer at all - one workspace per task, in whatever cmux window is open - and its socket password (when configured) is read from local, gitignored `config/cmux-socket-password` under the effective config directory, never committed. The caller-facing label remains `fm-<id>`, but the actual cmux workspace title is scoped by the active `FM_HOME` readable label plus a short hash of the resolved `FM_ROOT` path as `fm-<home-label>-<id>`. Test cleanup must use the guarded path in [`docs/cmux-backend.md`](cmux-backend.md#current-operation-and-safety), never enumerate-and-close every workspace. -`config/backend` is inherited into secondmate homes under the primary-authoritative contract owned by [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md). +The `config/backend` file is not inherited by secondmate homes. ## Away-mode supervisor backend (FM_SUPERVISOR_BACKEND / FM_SUPERVISOR_TARGET) @@ -137,20 +134,6 @@ Fleet-local operational facts and gotchas live locally in `data/learnings.md`; i The file is created lazily on first learning and follows the same dated, evidence-backed, curated style as `data/captain.md`: inspect the current file first, then rewrite or prune stale entries instead of appending forever. There is no shared learnings file by captain decision. -## Startup memory budget (config/startup-memory-budget) - -`config/startup-memory-budget` is the primary-authoritative per-home allowance for the startup prompt-memory surface: `data/captain.md`, `data/captain-shared.md`, and `data/learnings.md` together. -The locked mutable bootstrap path materializes its visible default of `7500` estimated tokens in a primary home when the file is absent. -To select another allowance, replace the primary home's file with one valid positive value in the exact format below; the next locked bootstrap convergence or `bin/fm-config-push.sh` propagates it to registered secondmates. -A secondmate does not create an independent default and instead receives the primary value through the inherited-local-material contract in [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md). -The file must be one positive base-10 integer followed by exactly one newline in a regular, single-linked file beneath a non-symlinked `config/` directory. -Malformed, multi-line, symlinked, hardlinked, special, or otherwise unsafe values are rejected rather than treated as a default. -Use `bin/fm-startup-memory-budget.sh read` to validate and print the effective value, or `bin/fm-startup-memory-budget.sh report` to account for the three files. -The stable local estimate is `ceil(UTF-8 bytes / 3)` per file, a conservative portable approximation rather than a provider-exact tokenizer. -An inherited `data/captain-shared.md` counts in a secondmate's total but remains primary-owned and read-only there. -The internal `/stow` skill curates only the editable local files in that case and reports the primary-owned shared file as a concrete exception if it alone exceeds the budget. -The helper's header owns exact parsing, publication, and report output mechanics. - ## Secondmate routes (data/secondmates.md) Persistent secondmate routes live locally in `data/secondmates.md`. @@ -183,8 +166,6 @@ When it is unset, most scripts use the repo root as the home; when it is set, sc When `FM_HOME` is unset, it also behaves as the old whole-root override. `bin/fm-send.sh` is intentionally stricter than that general fallback: it requires `FM_HOME` to be set before resolving a target, so operator steers cannot silently resolve against the wrong home. `FM_STATE_OVERRIDE`, `FM_DATA_OVERRIDE`, `FM_PROJECTS_OVERRIDE`, and `FM_CONFIG_OVERRIDE` override individual operational directories for tests and specialized harness setup. -Before `fm-brief.sh`, `fm-spawn.sh`, or `fm-afk-launch.sh` persists a path or passes it to another process, it resolves each applicable relative `FM_HOME`, `FM_STATE_OVERRIDE`, or `FM_DATA_OVERRIDE` directory against the caller's working directory, preserves absolute spellings unchanged, and rejects an unresolvable relative directory with the offending variable named. -Bootstrap applies the same relative `FM_HOME` resolution only when embedding that home in the generated X-mode poll shim; other transient consumers retain their existing shell-relative behavior. For the herdr backend, `FM_HOME` also determines the workspace label used by the adapter. For the zellij backend, `FM_HOME` does not split containers, but it determines the readable home prefix embedded in visible tab titles; use `FM_ZELLIJ_SESSION` when a separate zellij session is needed. The full zellij home label also includes a short hash of the resolved `FM_ROOT` path. @@ -193,17 +174,13 @@ The full cmux home label also includes a short hash of the resolved `FM_ROOT` pa ## Harness support -claude, codex, opencode, pi, pi-signed, grok, and kimi are empirically verified for crewmate and secondmate launches; [README requirements](../README.md#requirements) own the set supported for the primary session. -New harnesses get verified through a supervised trial task before joining the set. +claude, codex, opencode, pi, and grok are all empirically verified; new harnesses get verified through a supervised trial task before joining the set. The verified adapter knowledge - busy signatures, interrupt and exit commands, skill-invocation syntax, and per-harness quirks - lives in [`.agents/skills/harness-adapters/SKILL.md`](../.agents/skills/harness-adapters/SKILL.md). Launch mechanics, including the verified command templates, live in [`bin/fm-spawn.sh`](../bin/fm-spawn.sh). -Enabled primary-session turn-end guard integrations are tracked as repo-level hook files and documented in [`docs/turnend-guard.md`](turnend-guard.md). -Kimi remains outside the primary turn-end guard integrations; [`docs/turnend-guard.md`](turnend-guard.md#compatibility-limits) owns its separate captain-approved crew wake hook. +Primary-session turn-end guard integrations for verified harnesses are tracked as repo-level hook files and documented in [`docs/turnend-guard.md`](turnend-guard.md). Primary-session watcher wake protocols are rendered at session start by [`bin/fm-supervision-instructions.sh`](../bin/fm-supervision-instructions.sh) from [`docs/supervision-protocols/`](supervision-protocols/). -Claude's Stop `asyncRewake` hook owns tokenless re-arm cycles, Grok uses background-notify cycles, Codex uses bounded foreground checkpoints, Pi and pi-signed use the same two tracked primary extensions, and OpenCode uses its TUI plugin. +Claude and Grok use background-notify cycles, Codex uses bounded foreground checkpoints, Pi uses its two tracked primary extensions, and OpenCode uses its TUI plugin. `config/crew-harness` is a local, gitignored file containing one adapter name for crewmate and scout launches. -When pi-signed is selected, Firstmate launches the executable named `pi-signed` from `PATH` with `FM_PI_HARNESS=pi-signed` and refuses the launch if it is unavailable rather than falling back to pi. -Plain Pi launches set `FM_PI_HARNESS=pi`, so a signed primary's environment cannot relabel a plain Pi worker. When it is absent or contains `default`, crewmates mirror the firstmate's own harness. `config/secondmate-harness` is a separate local, gitignored file containing the adapter the primary uses to launch secondmate agents, optionally followed by model and effort tokens on the same line. The first non-empty, non-comment line is parsed as `<harness> [<model>] [<effort>]`. @@ -217,21 +194,16 @@ The inherited-local-material contract is owned by [`secondmate-provisioning`](.. Those inherited values are defaults and rules only; `fm-spawn` still permits a consciously chosen explicit runtime outside the config. `config/secondmate-harness` is not inherited because secondmates do not launch secondmates. For grok, `fm-spawn.sh` installs one firstmate-owned global turn-end hook under `$GROK_HOME/hooks/`, or `~/.grok/hooks/` when `GROK_HOME` is unset, and drops a per-task `.fm-grok-turnend` pointer in the worktree, with teardown removing the task token and pointer. -For Kimi crews, `fm-spawn.sh` runs `fm-kimi-turnend-hook.sh install`, drops a per-task `.fm-kimi-turnend` pointer in the worktree, and records the matching private registry token for teardown. -Kimi continues to use the captain's normal Kimi home, including the existing config, skills, and memory; Firstmate does not create an isolated Kimi home. -The Kimi installer requires an existing regular non-symlink `~/.kimi-code/config.toml`, `python3` with `tomllib`, and `jq`; it validates but never serializes the captain's TOML and refuses before writing when the config is missing, malformed, or surprising or when either tool requirement is unavailable. -Its `remove` action excises only the marker-delimited Firstmate region and removes Firstmate's hook files. -For Pi and pi-signed secondmate launches, `fm-spawn.sh` starts the selected executable with `-e` pointed at the secondmate home's own tracked `.pi/extensions/fm-primary-pi-watch.ts` and `.pi/extensions/fm-primary-turnend-guard.ts`, both already present from the secondmate home's git worktree. +For Pi secondmate launches, `fm-spawn.sh` starts Pi with `-e` pointed at the secondmate home's own tracked `.pi/extensions/fm-primary-pi-watch.ts` and `.pi/extensions/fm-primary-turnend-guard.ts`, both already present from the secondmate home's git worktree. ## Crew dispatch profiles (config/crew-dispatch.json) `config/crew-dispatch.json` is an optional local, gitignored file containing natural-language rules that firstmate reads before dispatching a crewmate or scout. -The shell scripts do not match those rules; firstmate chooses the best matching rule with judgment, resolves its profile object or array under the operating contract in `AGENTS.md` section 4 and `quota-array-dispatch`, and passes only concrete `--harness`, `--model`, and `--effort` flags to `fm-spawn.sh`. +The shell scripts do not match those rules; firstmate chooses the best matching rule with judgment, resolves that rule directly or through a supported selector, and passes only concrete `--harness`, `--model`, and `--effort` flags to `fm-spawn.sh`. When the file exists, `fm-spawn.sh` enforces that contract by refusing crewmate and scout spawns that lack an explicit harness (`--harness`, a positional adapter, or a raw launch command). Batch spawns satisfy the same requirement with a shared `--harness`. Secondmate spawns are exempt and still resolve through `config/secondmate-harness` and its optional model and effort tokens. -This section is the single owner of the canonical schema and its per-field semantics. -`AGENTS.md` section 4 owns the always-loaded dispatch intake boundary, and `quota-array-dispatch` owns the pace-aware profile-array selection procedure. +This section is the single owner of the canonical schema and its per-field semantics; `AGENTS.md` section 4 keeps only the dispatch procedure and points here. ```json { @@ -241,6 +213,7 @@ This section is the single owner of the canonical schema and its per-field seman "use": [ { "harness": "<adapter>", "model": "<optional model>", "effort": "<low|medium|high|xhigh|max, optional>" } ], + "select": "<optional strategy>", "why": "<optional rationale that helps firstmate choose>" } ], @@ -255,14 +228,17 @@ Both `use` and the optional top-level `default` accept either one profile object The single-object form stays fully backward-compatible, and every profile needs `harness`. Profile `model` and `effort` fields and rule `why` are optional. An omitted model or effort means the selected harness uses its own default for that axis. -Every profile array is an implicit quota-aware choice resolved through `quota-array-dispatch`. -If no dispatch rule fits, firstmate resolves `default` through the same object-or-array path before falling back to `config/crew-harness`. +Every profile array is an implicit quota-aware choice and does not need a selector property. +`select: "quota-balanced"` remains accepted on rules for compatibility and has the same behavior as an implicit array choice. +If no dispatch rule fits, firstmate resolves `default` through the same object-or-array selection path before falling back to `config/crew-harness`. If a selected profile carries an effort value the chosen harness does not accept, `fm-spawn.sh` records the requested `effort=` in task meta for traceability but omits the launch flag, and bootstrap reports the invalid harness/effort pair as a `CREW_DISPATCH` diagnostic when it is visible in the file. +Quota-aware selection is implemented by `bin/fm-dispatch-select.sh`, whose header owns provider and product mapping, relevant-window scoring, the stale-clear freshness margin, random tie-breaking, OS-backed random operational fallback, and safe selection-basis diagnostics. +Quota-data trouble never blocks dispatch, but malformed profile configuration remains an actionable validation error. See [`docs/examples/crew-dispatch.json`](examples/crew-dispatch.json) for a starting point to copy into local `config/crew-dispatch.json`. When the file exists, bootstrap validates it with `jq`. Valid files stay silent by default; with `FM_BOOTSTRAP_VERBOSE_FACTS=1`, bootstrap emits `BOOTSTRAP_INFO: crew dispatch active config/crew-dispatch.json`, one `BOOTSTRAP_INFO:` fact per rule, and one fact for the optional default profile set. -Malformed JSON, an empty or malformed rule/default array, an unverified harness, or an effort value unsupported by that harness is reported as `CREW_DISPATCH: invalid config/crew-dispatch.json - ...`; missing `jq` is reported through the normal `MISSING: jq` install-consent flow. -While the file remains present, no crewmate or scout spawn may proceed without an explicit resolved harness; malformed configuration must be reported and corrected rather than selected around. +Malformed JSON, an empty or malformed rule/default array, an unverified harness, an unknown `select`, or an effort value unsupported by that harness is reported as `CREW_DISPATCH: invalid config/crew-dispatch.json - ...`; missing `jq` is reported through the normal `MISSING: jq` install-consent flow. +Because the spawn backstop is gated by file presence, any fallback path after a missing match, validation error, or missing `jq` still passes a resolved harness explicitly until the file is fixed or removed. Secondmate homes inherit this file from the primary, so a secondmate's own crewmates apply the same dispatch profile behavior. ## Toolchain @@ -271,7 +247,6 @@ On session start the first mate detects what its required toolchain is missing o It installs automatically supported tools only after you say go; manual-only tools remain for you to install from the printed instructions. Required tools come in two parts: a universal toolchain every home needs regardless of backend, and a per-backend delta that follows the runtime backend actually resolved for this home. The universal toolchain is node, git, gh with GitHub auth via `gh auth login`, no-mistakes v1.31.2 or newer, gh-axi, chrome-devtools-axi, lavish-axi, compatible tasks-axi per "Backlog backend" above, and quota-axi. -The observed fleet version manifest and checksummed offline mirror recovery procedure are in [`docs/toolchain-versions.md`](toolchain-versions.md), while `bin/fm-toolchain-mirror.sh` owns the snapshot and restore mechanics. This section is the single owner of that universal toolchain list; backend guides' prerequisites point here and add only their backend-specific tools. In that list, no-mistakes runs the validation pipeline, gh-axi, chrome-devtools-axi, and lavish-axi cover GitHub, browser, and rich-review operations, and tasks-axi plus quota-axi back backlog mutations and quota-aware array dispatch. The per-backend delta is required only for the backend resolved from `FM_BACKEND`, then `config/backend`, then runtime auto-detection, then default `tmux`, so a home is never told to install a tool an inactive backend or feature would need. @@ -284,7 +259,7 @@ When `config/crew-dispatch.json` exists, bootstrap also requires `jq` for dispat When X mode is opted in, bootstrap also requires `curl` and `jq` before arming the relay poll shim. `tasks-axi` and `quota-axi` are required bootstrap tools in every profile, the same class as `lavish-axi`. An absent or incompatible `tasks-axi` reports `MISSING: tasks-axi (install: npm install -g tasks-axi)`; when `config/backlog-backend` is not `manual` and compatible `tasks-axi` is on `PATH`, bootstrap stays silent and firstmate uses its verbs for routine backlog mutations, otherwise it hand-edits `data/backlog.md` until installation is approved and completed. -An absent `quota-axi` reports `MISSING: quota-axi (install: npm install -g quota-axi)`; firstmate cannot resolve a profile array until current quota output is available for every candidate. +An absent `quota-axi` reports `MISSING: quota-axi (install: npm install -g quota-axi)`; `bin/fm-dispatch-select.sh` still selects uniformly from the valid candidate array with an OS-backed random source when quota data is unavailable. Bootstrap also reports a `TANGLE:` line when `FM_ROOT` is on a named non-default branch; follow the printed checkout remediation rather than treating it as an installable tool problem. In a read-only session that did not get the fleet lock, the same line is advisory and omits the checkout command. The locked session-start bootstrap step also runs a best-effort project clone refresh through `fm-fleet-sync.sh`. @@ -300,7 +275,7 @@ When a running home advances and its loaded instruction surface (`AGENTS.md`, `b If that send fails, bootstrap keeps an idempotent retry marker and emits `NUDGE_SECONDMATES:` with the failure reason. The same bootstrap run emits `SECONDMATE_LIVENESS:` only when a registered secondmate is skipped or its relaunch fails; already-live and successfully relaunched secondmates are handled silently. For a mid-session inherited local-material edit where tracked-file sync is not needed, run `bin/fm-config-push.sh`. -It uses the same live secondmate discovery and propagation helper as bootstrap, prints each live home's `crew-dispatch.json`, `crew-harness`, `backlog-backend`, `backend`, `herdr-presentation-spaces`, `startup-memory-budget`, and `data/captain-shared.md` result as `pushed`, `unchanged`, `skipped`, or `error`, and exits non-zero for real propagation errors or config-reread send failures. +It uses the same live secondmate discovery and propagation helper as bootstrap, prints each live home's `crew-dispatch.json`, `crew-harness`, `backlog-backend`, `herdr-presentation-spaces`, and `data/captain-shared.md` result as `pushed`, `unchanged`, `skipped`, or `error`, and exits non-zero for real propagation errors or config-reread send failures. When an allowlisted config item changes for an already-running home, it sends the literal-content reread pointer described in [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md); unchanged allowlisted config sends no pointer unless a previous delivery is pending. The locked bootstrap inheritance pass uses the same per-home changed-set and reread path for already-running homes; see `secondmate-provisioning` for the single contract owner. That live discovery starts from `state/*.meta` records with `kind=secondmate`; `data/secondmates.md` only backfills `home=` for older or incomplete meta records. @@ -396,7 +371,7 @@ FM_BACKEND= # optional runtime backend override for new spawns; tmux HERDR_SESSION=default # herdr-only: named session for normal backend ops; not enough for destructive cleanup (docs/herdr-backend.md) FM_BACKEND_HERDR_COMPOSER_LINES=20 # herdr-only: tail lines scanned by composer-state guard/fallback paths; idle-baseline submit confirmation uses agent-state FM_BACKEND_HERDR_IDLE_RE='^Type a message\.\.\.$' # herdr-only: empty-composer placeholder regex after shared ghost extraction plus border and prompt stripping -FM_BACKEND_HERDR_BARE_PROMPT_RE='^(❯|›)' # herdr-only: verified agent glyphs recognized as an UNBORDERED (bare) composer row, e.g. Claude's ❯ or Codex's ›; an alternation, not a `[...]` bracket expression, so a C-locale byte-decomposed match can never misfire on an unrelated multibyte glyph; shell glyphs remain unknown rather than empty, and de-emphasised ghost/placeholder text reads empty through shared fm_composer_strip_ghost (docs/herdr-backend.md "Composer and injection safety") +FM_BACKEND_HERDR_BARE_PROMPT_RE='^[❯›]' # herdr-only: verified agent glyphs recognized as an UNBORDERED (bare) composer row, e.g. Claude's ❯ or Codex's ›; shell glyphs remain unknown rather than empty, and de-emphasised ghost/placeholder text reads empty through shared fm_composer_strip_ghost (docs/herdr-backend.md "Composer and injection safety") FM_BACKEND_HERDR_PI_COMPOSER_MAX_LINES=8 # herdr-only: maximum rows admitted between Pi's native-identity-corroborated separator pair; taller or ambiguous candidates stay unknown (docs/herdr-backend.md "Composer and injection safety") FM_BACKEND_HERDR_SUBMIT_POLLS=6 # herdr-only: agent-state samples spread across each Enter attempt's budget when confirming a submit (docs/herdr-backend.md "Current transport behavior") FM_BACKEND_HERDR_SUBMIT_MIN_SLEEP=0.6 # herdr-only: minimum per-Enter confirmation budget before polling agent-state after an idle baseline @@ -430,13 +405,10 @@ FMX_FOLLOWUP_MAX_AGE_SECS=604800 # local window for posting X-mode completion FMX_FOLLOWUP_MAX_COUNT=3 # local cap on X-mode completion follow-ups per linked mention FM_LOCK_STALE_AFTER=2 # seconds before dead-pid lock records can be reclaimed; mid-acquire locks keep at least 2s grace FM_GUARD_GRACE=300 # seconds before guard warnings, arm health checks, and the primary turn-end guard treat a watcher beacon as stale -FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=800 # milliseconds the --claude turn-end guard waits for the Stop auto-arm's claim, health, or fresh rewake epoch before re-blocking -FM_CLAUDE_AUTOARM_EPOCH_FRESH=15 # seconds a recorded auto-arm rewake outcome counts as this event epoch's owned recovery -FM_CLAUDE_TURNEND_BLOCK_BUDGET=3 # consecutive --claude guard re-blocks before a degraded allow; safely below Claude Code's 8-block override -FM_ARM_CONFIRM_TIMEOUT=10 # seconds fm-watch-arm waits to confirm a fresh watcher before reporting FAILED; default 30 on Git Bash/MSYS +FM_ARM_CONFIRM_TIMEOUT=10 # seconds fm-watch-arm waits to confirm a fresh watcher before reporting FAILED FM_ARM_ATTACH_POLL=0.5 # seconds between checks while fm-watch-arm is attached to an existing healthy watcher cycle -FM_OPENCODE_ARM_READY_TIMEOUT_MS=12000 # milliseconds the OpenCode primary watcher plugin waits for an arm attempt to report started, healthy, wake, or failure; default 35000 on Windows to stay above the MSYS confirm budget -FM_PI_ARM_READY_TIMEOUT_MS=12000 # milliseconds the Pi watcher extension waits for a successor arm to report started or attached; default 35000 on Windows to stay above the MSYS confirm budget +FM_OPENCODE_ARM_READY_TIMEOUT_MS=12000 # milliseconds the OpenCode primary watcher plugin waits for an arm attempt to report started, healthy, wake, or failure +FM_PI_ARM_READY_TIMEOUT_MS=12000 # milliseconds the Pi watcher extension waits for a successor arm to report started or attached FM_WATCH_ARM_RETIRE_TIMEOUT_MS=1000 # milliseconds Pi/OpenCode wait for an unready successor arm to exit before abandoning retries FM_WATCH_REARM_RETRY_BASE_MS=250 # Pi/OpenCode adapter base delay for continuity restoration retries FM_WATCH_REARM_RETRY_MAX_MS=4000 # Pi/OpenCode adapter cap for exponential continuity retry delay @@ -448,7 +420,6 @@ FM_SIGNAL_GRACE=30 # seconds to coalesce nearby status and turn-end signals FM_CAPTAIN_RE='done:|needs-decision:|blocked:|failed:|PR ready|checks green|ready in branch|merged' # captain-relevant status regex; nonterminal progress verbs remain excluded even when their prose matches FM_CLASSIFY_PAUSED_VERB=paused # leading status verb for a declared external wait; excluded from FM_CAPTAIN_RE and distinct from blocked FM_STALE_ESCALATE_SECS=240 # idle seconds before a provably-working stale pane escalates; stale panes whose crew is not provably working surface immediately unless they declare the pause verb -FM_BUSY_TURN_MAX_SECS=3600 # maximum age of a busy pane's latest state/<id>.turn-ended marker, or its state/<id>.meta spawn record before any turn completes, before the same wedge escalation used for a provably-working non-busy stale takes over; inspection-only, never an automatic interrupt or restart FM_PAUSE_RESURFACE_SECS=3600 # seconds before an idle declared external wait re-surfaces for a recheck in the watcher or away-mode daemon FM_WEDGE_DEMAND_INSPECT_COUNT=3 # consecutive provably-working stale escalations on the same unchanged pane before demand-deep-inspection is added FM_WATCH_TRIAGE_LOG_MAX_BYTES=262144 # size cap for the watcher's absorbed-wake debug log @@ -461,7 +432,7 @@ FM_STALE_WORKTREE_LOCK_RETRY_WAIT_SECS= # legacy alias for FM_TREEHOUSE_RETURN FM_FLEET_SYNC_PACKED_REFS_LOCK_RETRIES=3 # fetch retries after fm-fleet-sync.sh hits the orphaned .git/packed-refs.lock signature FM_FLEET_SYNC_PACKED_REFS_LOCK_RETRY_WAIT_SECS=1 # seconds fm-fleet-sync.sh waits before each of those retries FM_FLEET_SYNC_PACKED_REFS_LOCK_AGE_SECS=30 # min mtime age before fm-fleet-sync.sh treats a leftover packed-refs.lock as provably stale -FM_BUSY_REGEX= # optional global override for every harness-scoped busy-pane matcher; unset uses each recorded harness's verified signature +FM_BUSY_REGEX='esc (to )?interrupt|Working\.\.\.|Ctrl\+c:cancel' # busy-pane signatures, shared by watcher, fm-crew-state pane fallback, and tmux helper FM_COMPOSER_IDLE_RE= # optional empty-composer regex, applied after ghost and border stripping FM_COMPOSER_GHOST_LUMA_MAX=128 # fleet-wide: max perceived luminance (0.299R+0.587G+0.114B, 0-255) for a TRUECOLOR foreground to count as de-emphasised ghost/placeholder text and be stripped; dim/faint (SGR 2) is stripped regardless. Assumes a dark terminal theme (bin/fm-composer-lib.sh's fm_composer_strip_ghost, shared by the tmux and herdr composer readers) GROK_HOME= # optional Grok config home for firstmate's global grok turn-end hook; defaults to ~/.grok diff --git a/docs/documentation-audiences.json b/docs/documentation-audiences.json index c9dc1dcb5e9..60773b0d451 100644 --- a/docs/documentation-audiences.json +++ b/docs/documentation-audiences.json @@ -151,30 +151,14 @@ "path": ".agents/skills/harness-adapters/SKILL.md", "audience": "agent-runtime" }, - { - "path": ".agents/skills/i-have-adhd/SKILL.md", - "audience": "agent-runtime" - }, - { - "path": ".agents/skills/mobile-mode/SKILL.md", - "audience": "agent-runtime" - }, { "path": ".agents/skills/project-management/SKILL.md", "audience": "agent-runtime" }, - { - "path": ".agents/skills/quota-array-dispatch/SKILL.md", - "audience": "agent-runtime" - }, { "path": ".agents/skills/secondmate-provisioning/SKILL.md", "audience": "agent-runtime" }, - { - "path": ".agents/skills/secrets-management/SKILL.md", - "audience": "agent-runtime" - }, { "path": ".agents/skills/stow/SKILL.md", "audience": "agent-runtime" @@ -247,18 +231,6 @@ "path": "docs/examples/crew-dispatch.json", "audience": "operator-example" }, - { - "path": "docs/examples/doppler-oidc-job.yml", - "audience": "operator-example" - }, - { - "path": "docs/examples/doppler-service-token-job.yml", - "audience": "operator-example" - }, - { - "path": "docs/examples/project-secrets-policy.json", - "audience": "operator-example" - }, { "path": "docs/examples/wedge-alarm", "audience": "operator-example" @@ -279,18 +251,10 @@ "path": "docs/herdr-backend.md", "audience": "operator-current" }, - { - "path": "docs/moshi-mobile-review.md", - "audience": "operator-current" - }, { "path": "docs/orca-backend.md", "audience": "operator-current" }, - { - "path": "docs/promotion-ladder.md", - "audience": "operator-current" - }, { "path": "docs/scripts.md", "audience": "operator-current" @@ -335,22 +299,10 @@ "path": "docs/turnend-guard.md", "audience": "operator-current" }, - { - "path": "docs/toolchain-versions.md", - "audience": "maintainer-verification" - }, { "path": "docs/verification/runtime-backends.md", "audience": "maintainer-verification" }, - { - "path": "docs/verification/moshi-mobile-review.md", - "audience": "maintainer-verification" - }, - { - "path": "docs/verification/stow-memory.md", - "audience": "maintainer-verification" - }, { "path": "docs/verification/supervision.md", "audience": "maintainer-verification" diff --git a/docs/herdr-backend.md b/docs/herdr-backend.md index 91047bcc6f3..d71a1bfdc95 100644 --- a/docs/herdr-backend.md +++ b/docs/herdr-backend.md @@ -96,35 +96,17 @@ A failed replacement rolls back only the exact response-derived new pane when fo Version 1 journals, dead or missing panes, duplicate or absent tokens, renamed or detached spaces, cross-home mismatches, inconsistent endpoint bindings, active target tabs, and ambiguous identity or focus fall back flat without mutating the old projection when duplicate-agent risk is positively absent. A live or unknown recorded or token-matched endpoint refuses duplicate launch. -Locked session start has one narrower cleanup for a restored projected child that is no longer current task state. -It runs only when the current home has at least one ordinary presentation journal and considers only that home; a primary never recursively sweeps a secondmate home. -Discovery starts from the exact current `└ <concise-task> · p:<22-character-token>` grammar, but a title or token alone is never mutation authority. -The title must contain exactly one token occurrence across the named-session snapshot and must equal the title derived from exactly one valid presentation journal in this home's own `state/`; a version 2 journal additionally must bind this exact physical home, named session, workspace, tab, and pane. -The task's ordinary metadata must be absent, and the candidate must have exactly one tab and exactly one pane. -Before cleanup, Firstmate acquires the existing task-id spawn lock and then the shared named-session presentation lock. -Inside both locks it takes one exact snapshot, requires one unambiguous non-target focus and the exact title, token, tab, and pane shape, positively confirms no registered agent, and reads Herdr's process information for the exact named-session pane. -The process proof requires one recognized idle shell as both the shell process and the sole foreground process-group member, an operating-system process-table row for that shell, no child process, and a sleeping or idle shell state. -Any foreground command, child process, active shell job, unknown shell, unreadable process table, missing field, or API error preserves the pane. -Firstmate immediately revalidates the same journal, metadata absence, workspace title and token uniqueness, one-tab and one-pane topology, exact pane relationship, absent agent, process proof, and non-target focus before calling the existing exact-pane focus-preserving close helper. -It closes only that pane, never a workspace. -The matching journal is retired only after the exact pane is positively confirmed gone; an unconfirmed close retains the journal, while a confirmed close may retire it even when focus restoration reported an error after the close. -A second run finds no matching title or journal and is a no-op. -A malformed or missing title or token, duplicate token, zero or multiple journal matches, cross-home version 2 binding, current metadata, registered or unknown agent, extra tab or pane, active target, busy lock, changed revalidation, unreadable check, or any error preserves the candidate and lets session startup continue with at most a concise warning. - Operational compromises: - Grouping is best-effort; only an exact same-identity version 2 binding survives a Herdr restart in place. - Existing layouts are not force-renamed or rearranged. - Missing or ambiguous restart bindings fall back to the ordinary home workspace while the old projection remains untouched. -- Crashes, lost responses, failed exact-pane cleanup, or human renames can leave quarantined spaces; session start removes only the exact home-local, uniquely journal-correlated, childless idle-shell shape above. -- Spaces have no cross-home cleanup path, and a secondmate child can clean up only from its exact home. -- Every stale-looking space outside that narrow startup proof still requires manual cleanup in Herdr's UI after human inspection. +- Crashes, lost responses, failed exact-pane cleanup, or human renames can leave quarantined spaces. +- Spaces have no cross-home cleanup path. - Regaining a dedicated space after degradation requires stopping the flat task, manually checking the stale projection, and clearing its journal before a genuinely fresh launch. - The visible token is only a restart-stable correlator and never substitutes for the exact binding. `tests/fm-backend-herdr-presentation-e2e.test.sh` covers multi-home ordering, concurrency, lock contention, legacy coexistence, focus preservation, exact same-identity restart replacement, ambiguous bindings and tokens, and exact-pane cleanup through the guarded lab path. -`tests/fm-herdr-session-cleanup.test.sh` covers every discovery, ownership, topology, process, locking, revalidation, focus, retirement, and continue-on-error boundary. -`tests/fm-herdr-session-cleanup-e2e.test.sh` covers the restored-shell cleanup in a guarded non-default named lab; [`verification/runtime-backends.md`](verification/runtime-backends.md#per-home-and-presentation-topology) owns the active versioned evidence. ## Default-tab prune safety @@ -176,7 +158,7 @@ The capture owner requests at least 200 lines from Herdr and trims locally to th This generous floor is required for small composer and peek reads. Herdr's native agent state can read idle while a harness waits on its own long foreground tool. -The shared crew-state path therefore corroborates every native non-busy or unreadable result with the recorded harness's rendered busy signature before concluding that a pane is not working. +The shared crew-state path therefore corroborates every native non-busy or unreadable result with the rendered busy regex before concluding that a pane is not working. A human-blocked permission dialog has no busy banner and still surfaces. ## Composer and injection safety @@ -276,8 +258,6 @@ tests/fm-backend-herdr-respawn-idem-e2e.test.sh tests/fm-backend-herdr-workspace-per-home-e2e.test.sh tests/fm-backend-herdr-presentation-e2e.test.sh tests/fm-backend-herdr-eventwait-smoke.test.sh -tests/fm-herdr-session-cleanup.test.sh -tests/fm-herdr-session-cleanup-e2e.test.sh tests/fm-afk-inject-herdr-e2e.test.sh tests/fm-afk-pi-herdr-return-e2e.test.sh ``` diff --git a/docs/sessionstart-nudge.md b/docs/sessionstart-nudge.md index c39c8149259..1f0ee079f47 100644 --- a/docs/sessionstart-nudge.md +++ b/docs/sessionstart-nudge.md @@ -12,7 +12,7 @@ It sources `bin/fm-gate-refuse-lib.sh` and stays silent for a no-mistakes gate a It shares `bin/fm-primary-scope-lib.sh` with `bin/fm-turnend-guard.sh`, so the hooks use one primary-detection owner. The Shared Predicate section of [`turnend-guard.md`](turnend-guard.md#shared-predicate) owns marker validation, plain-checkout detection, and required Firstmate-shaped paths. -Before printing, the wrapper reads `state/.lock` and walks at most eight parents from its own pid in its own separate, hard-coded loop, independent of `bin/fm-lock.sh`'s ancestry walk (`fm_harness_ancestry_pid()` in `bin/fm-session-lock-lib.sh`, which now walks up to sixteen parents and can extend past a claude-named match to a still-more-ancestral one) and of Pi's `lockOwnership()`. +Before printing, the wrapper reads `state/.lock` and walks at most eight parents from its own pid, matching `bin/fm-lock.sh` and Pi's `lockOwnership()` ancestry depth. If the lock names a live pid in that ancestry, session start already ran in this harness session and the wrapper stays silent. Every path exits 0, including malformed state and adapter errors, because a Claude SessionStart exit 2 blocks session initialization. @@ -23,7 +23,7 @@ Every path exits 0, including malformed state and adapter errors, because a Clau | Claude | `.claude/settings.json` registers `SessionStart` for `startup`, `resume`, and `clear`, excludes `compact`, and invokes the wrapper through `CLAUDE_PROJECT_DIR`. | Native stdout context injection is supported. | | Codex | `.codex/hooks.json` anchors to the hook process working directory, verifies a Firstmate-shaped hook-bearing root, and executes the wrapper. | Native stdout context injection is supported. | | OpenCode | `.opencode/plugins/fm-primary-sessionstart-nudge.js` listens for `session.created`, runs once per session id, and calls `client.session.promptAsync` only when the wrapper prints a nudge. | Interactive TUI delivery is supported; headless `opencode run` is intentionally fail-open because the process can exit before the queued turn. | -| Pi / pi-signed | `.pi/extensions/fm-primary-turnend-guard.ts` handles `session_start` reasons `startup`, `new`, and `resume`, then injects the wrapper output with `pi.sendMessage`. | The custom message reaches model context without racing an initial positional prompt. | +| Pi | `.pi/extensions/fm-primary-turnend-guard.ts` handles `session_start` reasons `startup`, `new`, and `resume`, then injects the wrapper output with `pi.sendMessage`. | The custom message reaches model context without racing an initial positional prompt. | | Grok | `.grok/hooks/fm-primary-sessionstart-nudge.json` registers a project `SessionStart` hook and invokes the wrapper through inline-defaulted `${GROK_WORKSPACE_ROOT:-}`. | The project hook runs when the checkout is trusted, but Grok currently discards hook stdout from model context, so this path is intentionally fail-open. | The OpenCode nudge runs only on `session.created`. @@ -36,6 +36,8 @@ That alternative expands trust and writes outside this repository, so Firstmate `tests/fm-sessionstart-nudge.test.sh` proves wrapper silence for both gate signals, an unmarked linked worktree, a missing state directory, and an already-owned lock. It proves exact U+2063 `FIRSTMATE_OP:`-prefixed, `session-start`-typed one-line output for a plain primary and a marked linked secondmate primary. +It also verifies tracked wrapper registration for Claude, Codex, OpenCode, Pi, and Grok. +`tests/fm-captain-translation-contract.test.sh` proves Ahoy's current marker rule, narrow legacy compatibility exclusions, genuine captain-message near misses, and the shared marker on supported user-role operational injections. `tests/fm-pi-primary-live-e2e.test.sh` and `tests/fm-opencode-primary-live-e2e.test.sh` exercise native startup paths with first-message and later-message Ahoy regressions. `tests/fm-turnend-guard.test.sh`, `tests/fm-pi-watch-extension.test.sh`, and `tests/fm-daemon.test.sh` cover marked guard, monitoring, and away-mode delivery. diff --git a/docs/tmux-backend.md b/docs/tmux-backend.md index 3bf20fe9d1a..50cc14ce9d2 100644 --- a/docs/tmux-backend.md +++ b/docs/tmux-backend.md @@ -46,44 +46,34 @@ Verify setup by spawning a small task and confirming its `fm-<id>` window appear A target-existence check proves only that the pane exists. The deeper tmux agent-liveness probe first verifies exact window membership, then reads `#{pane_current_command}` to distinguish a running harness process from a bare idle shell. -It classifies recognized Claude, Codex, OpenCode, Pi, pi-signed, Grok, and Kimi process names as `alive`, common shells as `dead`, an authoritatively absent window as `missing`, unreadable state as `unreadable`, and every other process as `ambiguous`. +It classifies recognized Claude, Codex, OpenCode, and Grok process names as `alive`, common shells as `dead`, an authoritatively absent window as `missing`, unreadable state as `unreadable`, and every other process as `ambiguous`. Only `dead` and `missing` authorize recovery because a false dead result could launch a duplicate agent. -The verified Pi Launcher path reports the exact foreground command `pi-launcher` for both pi and pi-signed, while direct executable identities `pi`, `pi-signed`, and `Pi` remain accepted exactly. -Similar or prefixed process names are not accepted through those exact Pi-family entries. +Pi runs through a generic `node` process name and cannot be attributed confidently from the tmux foreground-process field. +An existing Pi pane is therefore reported as ambiguous rather than auto-healed, while an authoritatively missing Pi window can be relaunched safely. +This is the active tmux liveness limitation. Agent liveness and composer safety are separate checks. -For a bordered composer, the tmux reader locates the complete box structurally and classifies every content row through the shared ANSI and ghost handling in `bin/fm-composer-lib.sh`. -Real text on any content row is pending, while only an unambiguous box with every row empty is proven empty. -Unreadable, incomplete, or structurally ambiguous boxes fail closed, and panes without a bordered composer retain the compatible cursor-row classification. -The shared classifier accepts a shell glyph as an empty agent composer only inside a verified bordered composer. +The shared classifier in `bin/fm-composer-lib.sh` accepts a shell glyph as an empty agent composer only inside a verified bordered composer. A bare shell prompt is `unknown`, so away-mode escalation is never injected into a dead shell. -Rendered busy detection is also harness-scoped. -Task metadata selects only that harness's verified signature, so output from one harness cannot make another harness appear busy. -The exact selection contract and safety rationale live in [architecture](architecture.md#runtime-session-backends), while the signatures live in [the harness-adapters skill](../.agents/skills/harness-adapters/SKILL.md). - `bin/fm-tmux-lib.sh` owns exact type-and-submit mechanics. It types a message once and retries Enter only until the composer clears. -Only a proven empty composer is a positive delivery acknowledgement. -Text left in established structure remains `pending`, text in ambiguous structure remains unproven, and unreadable or unsafe state remains unknown. -`fm-send.sh` reports every unconfirmed verdict as a failure instead of retyping or assuming delivery. +A cleared composer is the positive delivery acknowledgement; text left in the composer remains `pending`, and `fm-send.sh` reports the failure instead of retyping. OpenCode 1.18.4 has one busy-queue exception. While OpenCode is mid-turn, Enter queues the message but leaves its text visible until the turn completes. -After the normal retry budget, only structurally proven pending text in a provably busy pane is accepted as queued, while an idle pane remains `pending` as a genuine swallowed Enter. -Ambiguous pending text never receives the busy-queue conversion. -`tests/fm-tmux-submit-busy.test.sh` covers busy and idle panes with proven, ambiguous, and cleared composers. +After the normal retry budget, a provably busy pane is accepted as queued, while an idle pane remains `pending` as a genuine swallowed Enter. +`tests/fm-tmux-submit-busy.test.sh` covers busy and idle panes with both pending and cleared composers. ## Limits and regression entry points - tmux is the reference path and supports secondmate homes. +- Existing Pi agent-process liveness is inconclusive, while an authoritatively missing Pi window can trigger recovery. - The OpenCode busy-queue exception is tmux-specific; Herdr retains its separately documented gap. ```sh tests/fm-backend-tmux-smoke.test.sh -tests/fm-composer-ghost.test.sh -tests/fm-kimi-harness.test.sh tests/fm-tmux-submit-busy.test.sh tests/fm-bootstrap.test.sh ``` diff --git a/docs/turnend-guard.md b/docs/turnend-guard.md index 8ee750de397..a015cfe02cc 100644 --- a/docs/turnend-guard.md +++ b/docs/turnend-guard.md @@ -3,7 +3,7 @@ This is the authoritative current contract for the "no turn ends blind" primary backstop referenced from AGENTS.md section 8. The predicate lives in `bin/fm-turnend-guard.sh`. Primary scope lives in `bin/fm-primary-scope-lib.sh`, shared with the native session-start nudge in [`sessionstart-nudge.md`](sessionstart-nudge.md). -Harness hook files adapt each enabled primary harness integration's turn-end mechanism to that shared predicate. +Harness hook files only adapt each verified harness's turn-end mechanism to that shared predicate. Related PreToolUse guards deny unsafe commands before execution rather than detecting a blind turn end afterward. Their separate owners are [`arm-pretool-check.md`](arm-pretool-check.md), [`cd-guard.md`](cd-guard.md), and [`subagent-guard.md`](subagent-guard.md). @@ -26,8 +26,7 @@ That check keeps crewmate and scout linked worktrees inert because their git dir It also requires `AGENTS.md`, `bin/`, and the effective state directory. For an in-scope primary, the guard counts in-flight work from `state/*.meta`. -The default cross-harness mode exits silently with no work in flight. -Claude's `--claude` mode also treats `state/x-watch.check.sh` as supervision need, so X-mode relay polling remains guarded without an in-flight task. +It exits silently with no work in flight. Otherwise it calls `fm_watcher_healthy <state-dir> <watch-path> [grace-seconds] [home]` from `bin/fm-wake-lib.sh`, the same identity-matched lock and fresh-beacon check used by `bin/fm-watch-arm.sh`. A stale beacon blocks even when a watcher pid is live. A fresh leftover beacon blocks when the lock is missing, dead, or identity-mismatched. @@ -38,60 +37,39 @@ If `jq` is missing or hook stdin is empty, the guard exits 0 because it cannot s ## Harness integrations -- Claude registers two `Stop` hooks in `.claude/settings.json`, both anchored through `CLAUDE_PROJECT_DIR`: `bin/fm-turnend-guard.sh --claude`, and `bin/fm-claude-stop-autoarm.sh` with `asyncRewake: true` and `timeout: 28800`. +- Claude registers a `Stop` hook in `.claude/settings.json`, anchored through `CLAUDE_PROJECT_DIR`. - Codex registers a `Stop` hook in `.codex/hooks.json`, anchors the executable to the hook process working directory, verifies a Firstmate-shaped hook-bearing root, and passes the original payload to the shared guard. - OpenCode listens for `session.idle` in `.opencode/plugins/fm-primary-turnend-guard.js`, lets the watcher coordinator act first, and calls `client.session.promptAsync` once when the guard returns 2. - Pi listens for `agent_settled` in `.pi/extensions/fm-primary-turnend-guard.ts`, runs once per logical agent run, and calls `pi.sendUserMessage(..., { deliverAs: "followUp" })` once when the guard returns 2. -- Grok registers a `Stop` hook in `.grok/hooks/fm-primary-turnend-guard.json` and delegates capability selection to `bin/fm-turnend-guard-grok.sh`. - The tracked Claude Stop entries are inert when `GROK_AGENT` is present, so Grok's Claude-compatible settings loading cannot create a second continuation path. +- Grok registers a `Stop` hook in `.grok/hooks/fm-primary-turnend-guard.json` and uses `bin/fm-turnend-guard-grok.sh` to resume the reported session once when the shared guard returns 2. + The adapter intentionally omits `--permission-mode`, so a passive hook cannot grant stronger permissions than the resumed session default. Claude and Codex can block a Stop directly with exit status 2 and stderr. -Both payloads carry `stop_hook_active`. -In the default Codex mode, a true value lets the second stop finish after one forced continuation. +Both payloads carry `stop_hook_active`; a true value lets the second stop finish after one forced continuation. -Claude runs the guard with `--claude`, which ignores `stop_hook_active` and cooperates with the Stop-owned auto-arm. -Claude Code sets `stop_hook_active=true` on every stop after any stop-hook continuation, including `asyncRewake` rewakes, which re-opened the 2026-07-21 blind window under the default one-shot behavior. -The Claude mode waits up to `FM_CLAUDE_AUTOARM_SYNC_WAIT_MS` (default 800 milliseconds) and allows the stop when the watcher is healthy, `state/.claude-autoarm.lock` has a live owner, or `state/.claude-autoarm-epoch` contains a fresh rewake outcome. -When none of those proofs appears, it re-blocks up to `FM_CLAUDE_TURNEND_BLOCK_BUDGET` times (default 3, below Claude's 8-block override), then allows degraded with a visible `systemMessage`. -Any allow resets the budget. - -OpenCode, Pi, and pi-signed expose passive callbacks for this purpose. +OpenCode, Pi, and Grok expose passive callbacks for this purpose. Their adapters fail open at the hook boundary to protect the user session but schedule one bounded follow-up when the predicate blocks. The generated prompts use the canonical `turn-end-guard` kind after the U+2063 `FIRSTMATE_OP: ` prefix, so Ahoy does not treat them as captain messages. -Each passive adapter owns a loop latch. +Each adapter owns a loop latch. Pi keeps the latch across internal tool turns and clears it only when the generated follow-up settles or delivery fails. +Grok's project hook requires the checkout to be trusted with `/hooks-trust` or launch-time `--trust`. OpenCode's forced follow-up is supported for persistent TUI sessions and remains fail-open in headless `opencode run`. -Grok makes exactly one typed capability decision from each running Stop payload. -A boolean `stopHookActive` selects native blocking, including both false on the initial stop and true on the bounded continuation. -The camel-case field has precedence when both spellings appear; when it is absent, a boolean `stop_hook_active` selects the same native path for compatibility. -The native path returns the shared guard's status and stderr to the same Grok process and never starts `grok --resume`. -When both capability spellings are absent, the adapter preserves one pre-native `grok --resume` fallback guarded by `GROK_TURNEND_GUARD_ACTIVE` and intentionally omits `--permission-mode`. -Malformed JSON, a selected field with a non-boolean type, missing `jq`, missing hook prerequisites, or an already-active legacy guard allows the stop without starting either continuation path. -Grok's project hook requires the checkout to be trusted with `/hooks-trust` or launch-time `--trust`; genuine pre-native builds can run the same tracked hook from an isolated global hook directory. - -If a passive adapter cannot invoke its SDK, or the Grok legacy fallback cannot find `grok` or a session id, the next pull-based `fm-guard.sh` call reports the problem. +If a passive adapter cannot invoke its SDK, find `grok`, or recover a Grok session id, the next pull-based `fm-guard.sh` call reports the problem. That warning uses `bin/fm-supervision-instructions.sh --repair-line`, so it always points to the active harness protocol rather than embedding another repair command. ## Compatibility limits - Child crewmate and scout worktrees are outside scope. -- A valid secondmate home is in scope; an idle secondmate endpoint with no X-mode relay poll remains healthy because it has no supervision need. -- The direct-blocking and bounded passive-follow-up split is limited to the primary integrations listed above. +- A valid secondmate home is in scope; an idle secondmate endpoint remains healthy because no work is in flight there. +- Claude and Codex block directly, while OpenCode, Pi, and Grok use bounded passive follow-ups. - OpenCode headless mode and untrusted Grok project hooks remain fail-open at the host boundary. -- Kimi Code CLI 0.29.1 exposes only global `[[hooks]]` configuration in `~/.kimi-code/config.toml`, including a `Stop` event with snake_case payload fields `hook_event_name`, `session_id`, `cwd`, and `stop_hook_active`. -- Kimi has no project-level hook configuration and remains outside the primary guard integrations above. -- Captain-approved Kimi crew wake support uses `bin/fm-kimi-turnend-hook.sh` to edit only one marker-delimited Firstmate region in that global config and install a silent always-zero hook. -- The hook remains inert unless the payload `cwd` contains a per-task token pointer that resolves through Firstmate's private registry to one `state/<id>.turn-ended` marker. -- Installation refuses before writing unless `python3` with `tomllib` and `jq` are available. -- If `jq` is removed after installation, the hook remains silent and exits 0, turn-end wakes stop, and Kimi crews fall back to idle detection. -- Unreadable hook input remains fail-open. +- Missing `jq` or unreadable hook input remains fail-open. - No harness adapter uses a shell ampersand to manufacture supervision. ## Regression coverage -`tests/fm-turnend-guard.test.sh` covers the predicate, main and secondmate primary scope, child-worktree exclusion, `FM_HOME` and `FM_STATE_OVERRIDE` precedence, the cooperative `--claude` claim wait, epoch allow, re-block budget, Pi logical-run latching, missing-`jq` behavior, all five primary registrations, Grok native and legacy selection, typed field precedence, malformed input, and exactly-one-path safety. -`tests/fm-kimi-harness.test.sh` covers the separate Kimi crew hook's format preservation, idempotence, refusal cases, token guard, spawn registration, and teardown cleanup. -`tests/fm-supervision-instructions.test.sh` covers recovery-line ownership and pi-signed's identity-preserving reuse of Pi's protocol. +`tests/fm-turnend-guard.test.sh` covers the predicate, main and secondmate primary scope, child-worktree exclusion, `FM_HOME` and `FM_STATE_OVERRIDE` precedence, Pi logical-run latching, missing-`jq` behavior, all five registrations, and Grok resume permission and recursion safety. +`tests/fm-supervision-instructions.test.sh` covers recovery-line ownership. `FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh` is the opt-in isolated Pi path. -[`verification/supervision.md`](verification/supervision.md#turn-end-guard) records the active cross-harness empirical evidence, including the 2026-07-24 Claude `asyncRewake` revalidation. +[`verification/supervision.md`](verification/supervision.md#turn-end-guard) records the active cross-harness empirical evidence. diff --git a/docs/verification/runtime-backends.md b/docs/verification/runtime-backends.md index 65152100f4e..e012f0dc5e5 100644 --- a/docs/verification/runtime-backends.md +++ b/docs/verification/runtime-backends.md @@ -29,93 +29,15 @@ zsh A persistent parent shell waiting for a child remained reported as the parent process, while a shell that directly execed a simple command changed identity with the process itself. Claude, Codex, OpenCode, and Grok were observed under their own process names. -Kimi Code CLI 0.29.1 was observed under `kimi` on 2026-07-25. -Pi and pi-signed 0.82.0 were reverified on 2026-07-27 through real isolated `fm-spawn.sh` launches. +Pi remained a generic `node` process and is intentionally inconclusive. -Installed-wrapper checks: +The OpenCode 1.18.4 busy-queue behavior and the tmux fallback are pinned by: ```sh -basename "$(command -v pi-signed)" -pi-signed --version -pi --version -``` - -Observed bounded output: - -```text -pi-signed -0.82.0 -0.82.0 -``` - -The isolated process and endpoint checks used: - -```sh -tmux display-message -p -t "$target" '#{pane_current_command}' -ps -o comm= -p "$wrapper_pid" -ps -o comm= -p "$engine_pid" -FM_HOME="$fixture_home" bin/fm-crew-state.sh "$task_id" -``` - -Observed bounded shapes: - -```text -pi-launcher -.../pi-signed -.../Pi Launcher.app/Contents/Resources/pi/pi -state: done ... -``` - -Both launches executed a submitted tool instruction and touched the generated `turn_end` marker. -The pi-signed launch retained `harness=pi-signed`, while the plain comparison retained `harness=pi`. -The exact wrapper ancestry was `pi-signed` parent to Pi engine child, and the plain Pi Launcher path also traversed the signed wrapper on this installation. -That shared plain-Pi path is retained as disconfirming evidence against using ancestry as runtime-selection authority. -Firstmate therefore sets the exact `FM_PI_HARNESS` selection marker on both worker launch paths, while an unmarked Pi-family process remains `pi`. -Both recorded runtime identities now classify the exact `pi-launcher` foreground command as `alive`. - -Backend applicability was reviewed across every spawn adapter. -Tmux needs the exact `pi-launcher`, `pi-signed`, `pi`, and `Pi` process identities for recovery-grade liveness. -Herdr uses native registered-agent state and needs no process-name branch. -Zellij has no verified recovery-grade agent process probe, while Orca and cmux do not support secondmate spawns, so those three retain their existing generic ordinary-launch semantics without a new liveness matcher. - -The structural multi-row composer reader, Kimi pointer-delivery path, and OpenCode 1.18.4 busy-queue behavior are pinned by: - -```sh -tests/fm-composer-ghost.test.sh -tests/fm-kimi-harness.test.sh tests/fm-tmux-submit-busy.test.sh ``` -Expected structural matrix: real text on any content row is pending; all-empty complete boxes are empty; unreadable, incomplete, or unsafe boxes are unknown; and non-bordered panes retain cursor-row compatibility. -Expected submit matrix: proven pending plus busy is accepted as queued; proven pending plus idle remains pending; ambiguous pending is never converted by the busy exception; and only a proven empty composer succeeds directly. - -### Cleanup endpoint identity - -The cleanup identity boundary was validated on 2026-07-28 with tmux 3.6a and metadata fixtures for every supported backend. - -```sh -tests/fm-teardown-endpoint-safety.test.sh -tests/fm-teardown.test.sh -tests/fm-backend-herdr.test.sh -tests/fm-backend-zellij.test.sh -tests/fm-backend-orca.test.sh -tests/fm-backend-cmux.test.sh -``` - -Bounded output from the incident regression: - -```text -ok - fm-teardown: missing, empty, malformed, ambiguous, and task-mismatched endpoints refuse before every mutation or runtime call -ok - cleanup identity: valid tmux, Herdr, Zellij, Orca, and cmux records validate while every empty backend target refuses -ok - tmux backend: direct empty target returns nonzero without invoking tmux -ok - process cleanup: creation-time PID identity removes only the exact child and preserves the control child -ok - fm-teardown: dedicated-socket invalid cleanup preserves target/control and valid cleanup removes only the exact target -``` - -The dedicated tmux cell removed ambient tmux variables, required a socket-bound wrapper, kept one target and one independent control window, and proved the wrapper was not called for invalid metadata or a direct empty target. -Valid cleanup removed only the exact task-bound target and left the control window live. -The metadata-only validation covers tmux, Herdr, Zellij, Orca, and cmux before backend dispatch. -Claude, Codex, OpenCode, Pi, pi-signed, Grok, and Kimi share that backend cleanup boundary; their harness-specific hook files and token cleanup run only after it, so no harness needs a separate endpoint parser. +Expected matrix: pending plus busy is accepted as queued; pending plus idle remains pending; a cleared composer succeeds in either state. ## Herdr @@ -221,15 +143,6 @@ ok - real Herdr lab: missing, renamed, and duplicate tokens trigger zero destruc ok - real Herdr lab validation completed on Herdr 0.7.5 with the default-session tripwire intact ``` -The restored-shell session-start cleanup ran on 2026-07-24 against Herdr 0.7.5 protocol 17: - -```sh -HERDR_LAB_HELPER=bin/fm-herdr-lab.sh \ - tests/fm-herdr-session-cleanup-e2e.test.sh -``` - -Observed guarantee: one exact home-local, journal-correlated, one-tab and one-pane childless idle shell was closed after restoration while the exact non-target focus and default fleet session remained unchanged, and a repeat run was a no-op. - ### Composer and operational input Real captures verified these active distinctions: diff --git a/docs/verification/supervision.md b/docs/verification/supervision.md index 20f415e4564..920d172cd64 100644 --- a/docs/verification/supervision.md +++ b/docs/verification/supervision.md @@ -45,13 +45,12 @@ pi -p -e .pi/extensions/fm-primary-turnend-guard.ts \ Observed result: `PI_SMOKE_DONE`, with one session-start execution. The earlier `sendUserMessage` counterfactual raced the positional prompt; the current non-triggering `pi.sendMessage` custom message did not. -The installed pi-signed 0.82.0 wrapper repeated the Pi primary extension and session-start path on 2026-07-27. -[`runtime-backends.md`](runtime-backends.md#tmux) owns the shared-ancestry evidence and authoritative selection-marker boundary. Current deterministic and live entry points: ```sh tests/fm-sessionstart-nudge.test.sh +tests/fm-captain-translation-contract.test.sh FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh FM_OPENCODE_LIVE_E2E=1 tests/fm-opencode-primary-live-e2e.test.sh ``` @@ -62,52 +61,18 @@ The detailed reconciliation and task chronology stay in the private audit report ## Turn-end guard -The direct and passive mechanisms were validated across all five harnesses on 2026-07-08 through 2026-07-12, with Claude's replacement Stop-owned path revalidated on 2026-07-24. +The direct and passive mechanisms were validated across all five harnesses on 2026-07-08 through 2026-07-12. | Harness | Version verified | Mechanism | Observed result | | --- | --- | --- | --- | -| Claude | 2.1.219 | Cooperative blocking `Stop` guard plus `asyncRewake` auto-arm | A fresh unsupervised session ran session start first, reclaimed a stale dead-owner lock, completed two tokenless rewake cycles with no model arm command or guard continuation, and left a competing live owner unchanged. | +| Claude | 2.1.204 | Blocking `Stop` hook | First stop blocked, one continuation ran, `stop_hook_active=true` allowed the second stop. | | Codex | 0.142.1 | Blocking `Stop` hook | Hook process root stayed anchored to the trusted checkout and one continuation ran. | | OpenCode | 1.17.6 | Passive `session.idle` callback | Throwing could not block, while `promptAsync` scheduled one TUI follow-up; headless remained fail-open. | | Pi | 0.80.5 | Passive `agent_settled` callback | Exactly one guard follow-up ran for an unhealthy cycle, with no recursion across tool turns. | -| Grok | 0.2.112 native and 0.2.73 pre-native | Running-payload adaptive `Stop` | Native false-to-true continuation stayed in one process with two model turns and zero resume launches; the field-absent pre-native process launched exactly one guarded resume. | +| Grok | 0.2.93 | Passive `Stop` plus bounded resume | Project hook ran under trust, resumed once without inherited bypass permissions, and the environment latch prevented recursion. | -The Grok adaptive matrix ran on 2026-07-28 with separate scratch repositories and homes, dedicated tmux sockets, one target plus one control window, ambient tmux variables removed, and a socket-bound wrapper first in `PATH`. - -```sh -FM_GROK_STOP_LIVE_E2E=1 \ - FM_GROK_NATIVE_BIN="$native_grok_0_2_112" \ - FM_GROK_LEGACY_BIN="$official_pre_native_grok_0_2_73" \ - tests/fm-grok-stop-live-e2e.test.sh -``` - -Observed bounded output: - -```text -ok - grok 0.2.112 (9bbd559437aa) [stable] native Stop kept one session across false->true, two model turns, and zero resume processes -ok - grok 0.2.73 (9ff14c43bbe5) [stable] legacy Stop omitted capability, resumed exactly once, and stopped normally -ok - Grok adaptive Stop real-process matrix passed with exact target cleanup and control-window survival -``` - -The same run proved the Claude-compatible Stop entries stay inert under `GROK_AGENT`, the legacy resume carries `GROK_TURNEND_GUARD_ACTIVE=1`, and every replacement root is removed after exact target cleanup while its control window survives. - -The secondmate-home scope and manual-repair wake path were measured with Claude Code 2.1.207 on 2026-07-12, when a native background completion re-invoked the idle model with no human input. -The current Stop-owned main/secondmate inclusion and child-worktree exclusion are covered deterministically by `tests/fm-claude-stop-autoarm.test.sh`. -On 2026-07-28 with Claude Code 2.1.205, `fm_harness_ancestry_pid()` in `bin/fm-session-lock-lib.sh` was fixed to resolve the outermost pid of a contiguous nested-harness run instead of the first match, so the Stop auto-arm correctly reaches the session's true lock owner through Claude Code's multi-level `bg-spare` hook worker chain. - -The Claude product live path ran with Claude Code 2.1.219 on 2026-07-24: - -```sh -claude --version -FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-stop-autoarm-live-e2e.test.sh -``` - -Observed output: - -```text -2.1.219 (Claude Code) -ok - Claude 2.1.219 (Claude Code) live E2E reclaimed a stale session lock through session start, completed two tokenless Stop-owned rewake cycles, and preserved the competing-live-owner boundary -``` +The secondmate-home scope was measured with Claude Code 2.1.207 on 2026-07-12. +A native background completion re-invoked the idle model with no human input, while deterministic tests covered main/secondmate inclusion and child-worktree exclusion. Current entry points: @@ -115,16 +80,15 @@ Current entry points: tests/fm-turnend-guard.test.sh tests/fm-supervision-instructions.test.sh FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh -FM_GROK_STOP_LIVE_E2E=1 FM_GROK_NATIVE_BIN="$native_grok" FM_GROK_LEGACY_BIN="$pre_native_grok" tests/fm-grok-stop-live-e2e.test.sh ``` ## Watcher continuity -The cross-harness evidence combines the 2026-07-17 live pass with Claude's replacement Stop-owned path revalidated on 2026-07-24, all against isolated project and home state. +The five-harness live pass ran on 2026-07-17 against isolated project and home state. No credential material was copied into a fixture. ```text -Claude Code 2.1.219 +Claude Code 2.1.214 codex-cli 0.144.4 OpenCode 1.17.18 Pi 0.80.10 @@ -133,7 +97,7 @@ grok 0.2.103 (89c3d36fb6f1) [stable] | Harness | Exact opt-in command | Observed guarantee | | --- | --- | --- | -| Claude | `FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-stop-autoarm-live-e2e.test.sh` | Session start reclaimed a stale owner before two Stop-owned cycles, and a competing live owner prevented arm, rewake, epoch write, or lock replacement. | +| Claude | `FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-continuity-live-e2e.test.sh` | Native background completion woke the model, allowed drain/recovery, and refused an unrelated fleet command before its body ran. | | Codex | `FM_CODEX_LIVE_E2E=1 tests/fm-codex-continuity-live-e2e.test.sh` | The one-second foreground checkpoint returned without switching to the arm wrapper. | | OpenCode | `FM_OPENCODE_LIVE_E2E=1 tests/fm-opencode-primary-live-e2e.test.sh` | A verified successor existed before prompt handling, with no model re-arm or turn-end fallback. | | Pi | `FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh` | One initial tool call led to extension-owned successors and clean child retirement on exit. | @@ -141,27 +105,12 @@ grok 0.2.103 (89c3d36fb6f1) [stable] Pi 0.81.1 repeated the continuity and clean-exit lifecycle on 2026-07-23 after the Calm presentation changes. -Pi same-process session-transition ownership was verified on 2026-07-27 against the tracked extension with a faithful in-process factory rebind (module cache retained, real arm children): - -```sh -pi --version -tests/fm-pi-watch-extension.test.sh -tests/fm-pi-primary-types.test.sh -``` - -Observed guarantee: after ordinary `session_shutdown` for `/new`, `/resume`, and `/fork`, plus same-instance shutdown-plus-start, the replacement generation armed again without a Pi restart and without the `watcher: not armed - Pi session is shutting down` refusal. -Stale prior-generation tool callbacks could not mutate the active child, repeated transitions kept exactly one live arm cycle, and terminal `quit` still refused late rearm. -Plain Pi and pi-signed share the same tracked `.pi/extensions/fm-primary-pi-watch.ts` path, so both inherit the generation owner; other primary harnesses are not applicable because they do not use this Pi extension lifecycle. - Deterministic entry points: ```sh tests/fm-pi-watch-extension.test.sh -tests/fm-pi-primary-types.test.sh tests/fm-watcher-lock.test.sh -tests/fm-subagent-pretool-check.test.sh -tests/fm-claude-stop-autoarm.test.sh -tests/fm-turnend-guard.test.sh +tests/fm-continuity-pretool-check.test.sh ``` ## Wedge-alarm channels diff --git a/docs/watcher-continuity.md b/docs/watcher-continuity.md index 52b3a9eec3a..76a6353283c 100644 --- a/docs/watcher-continuity.md +++ b/docs/watcher-continuity.md @@ -8,12 +8,6 @@ Must-work continuity now lives above that process boundary instead of depending Pi's `.pi/extensions/fm-primary-pi-watch.ts` and OpenCode's `.opencode/plugins/fm-primary-watch-arm.js` own continuous re-arm after an actionable child close. Each adapter starts the next arm before delivering the wake prompt, checks current session-lock ownership at launch, preserves one child or scheduled retry at a time, and applies bounded exponential retry after an unexpected or failed close. A failed follow-up never cancels continuity restoration. -Pi same-process session replacement follows the generation-owner contract in `.pi/extensions/fm-primary-pi-watch.ts`. -Claude's `.claude/settings.json` Stop `asyncRewake` hook (`bin/fm-claude-stop-autoarm.sh`) owns routine tokenless re-arm. -The hook fires on every Stop, and an eligible primary with supervision need admits one home-scoped owner that foregrounds `bin/fm-watch-arm.sh` inside the hook-owned process tree. -A numeric session-lock owner that fails the shared `fm_harness_pid_alive` predicate is reclaimed through `bin/fm-lock.sh` before auto-arm state changes, while a live owner, absent lock, or malformed lock keeps the competing hook inert. -The stale-owner claim occurs only after the existing AFK and supervision-need gates pass. -While supervision is still needed and away mode remains inactive, an actionable close or typed failure wakes the idle session through exit 2. ## Actionable wake ordering @@ -24,16 +18,15 @@ When that retained arm later closes, its actual close is classified as a new sup After the configured retry bound is exhausted, it delivers the original wake with a typed continuity-restoration failure even if every successor arm hung without reporting readiness. This is deliberate Option B ordering: the fleet is protected before the model handles the wake whenever restoration succeeds, but the model is never left blind when it does not. -Claude's Stop hook starts the successor arm at the next Stop after the handling turn, rather than before notification as Pi and OpenCode do. -The durable wake queue preserves actionable events during the residual active-turn window, and the unchanged bounded turn-end guard enforces recovery at Stop when no watcher or auto-arm claim is present. -No PreToolUse hook denies fleet commands based on watcher status. -The model no longer re-arms after ordinary wakes. -Terminal arm-output classification (`started`, `attached`, or `FAILED`) remains defense in depth for the manual recovery path. +Claude retains its native tracked background-task completion path. +Its new PreToolUse continuity gate allows wake drain, arm recovery, and independently fail-closed teardown, but refuses other fleet commands while tasks are in flight and no identity-matched live watcher holds the home lock. +Allowing an ordinary literal teardown prevents a terminal wake from creating a recovery circle: forced or dynamically constructed teardown remains blocked, ordinary teardown itself still refuses dirty, unlanded, incomplete-scout, and unresolved-decision cases, and the turn-end guard continues to require supervision for any tasks left in flight. Codex retains its bounded foreground checkpoint protocol. Grok retains its tracked background-task notification protocol. No adapter starts a replacement with shell `&`. -The turn-end guard remains the final backstop rather than the normal continuity mechanism and cooperates with the auto-arm in its `--claude` mode. +The existing turn-end guard implementation and adapters are unchanged. +They remain the final backstop rather than the normal continuity mechanism. ## Arm-layer cycle contract @@ -53,18 +46,14 @@ Only the watcher process touches `state/.last-watcher-beat`; no helper process c ## Regression coverage `tests/fm-pi-watch-extension.test.sh` checks Pi's first-cycle-or-explicit-repair tool metadata and ownership-based redundant-call no-ops, then simulates actionable and empty child closes against the actual Pi and OpenCode close handlers, blocks prompt delivery to prove the successor launches first, verifies single-flight behavior, changes the session lock before close to prove ownership is rechecked, and hangs each successor arm to prove bounded fallback delivery includes the typed restoration failure. -The same suite covers ordinary same-process session replacement for `/new`, `/resume`, and `/fork`, same-instance shutdown-plus-start, stale prior-generation callbacks, repeated transitions with exactly one live cycle, disappearance of the shutting-down refusal after a valid replacement activates, and terminal quit still refusing late rearm. `tests/fm-watcher-lock.test.sh` covers verified-successor attach, the typed self-eviction failure, bounded and successor-linked lifecycle rows, and a SIGSTOP counterfactual that distinguishes a live PID from a stale beacon before classifying termination. -`tests/fm-subagent-pretool-check.test.sh` proves Claude retains only the non-status Bash seatbelts. -`tests/fm-claude-stop-autoarm.test.sh` covers the auto-arm's scope, stale and live session owners, unchanged AFK and need boundaries, single-flight, and exit-2 translation. -`FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-stop-autoarm-live-e2e.test.sh` starts with the reproduced stale-lock state, runs session start first, completes two tokenless cycles, and checks the competing-live-owner negative control. -`tests/fm-turnend-guard.test.sh` covers the cooperative `--claude` guard. +`tests/fm-continuity-pretool-check.test.sh` proves the Claude gate rejects only non-recovery fleet execution in the precise unhealthy state and preserves the existing Stop registration. ## Active limits and verification The goal is continuity without a Pi or OpenCode model-memory re-arm step. No zero-latency guarantee is claimed because lock verification, watcher startup, and bounded retry delays remain deliberate safety work. OpenCode support targets persistent TUI sessions rather than headless `opencode run`. -Claude depends on the Stop `asyncRewake` rewake, Grok retains native background-completion notifications, and Codex retains bounded foreground checkpoints. +Claude and Grok depend on their native background-completion notifications, and Codex retains bounded foreground checkpoints. -[`verification/supervision.md`](verification/supervision.md#watcher-continuity) records the current five-harness live evidence, the 2026-07-24 Stop-owned Claude auto-arm results, and exact opt-in commands. +[`verification/supervision.md`](verification/supervision.md#watcher-continuity) records the current five-harness live evidence and exact opt-in commands. diff --git a/docs/zellij-backend.md b/docs/zellij-backend.md index 367da98ebbd..b7b2ac08454 100644 --- a/docs/zellij-backend.md +++ b/docs/zellij-backend.md @@ -77,7 +77,7 @@ There is a narrow visible race between those calls that no current Zellij flag c Literal send uses bracketed paste followed by a separate explicit Enter. The adapter supports `Enter`, `Esc`, and the one-argument key expression `Ctrl c` through the shared key vocabulary. Zellij exposes no cursor-row, ANSI composer style, or native agent-state signal, so submit acknowledgement remains content-delta based. -This can distinguish no change from a changed screen but is less precise than tmux's structural box reader or Herdr's native state plus structural classifier. +This can distinguish no change from a changed screen but is less precise than tmux's cursor row or Herdr's native state plus structural classifier. Viewport capture has no line-bound option. Routine reads use `dump-screen` and larger peeks use `dump-screen --full`, followed by local trimming. diff --git a/tests/fm-documentation-audiences.test.sh b/tests/fm-documentation-audiences.test.sh index 90222802f6a..11854594afe 100755 --- a/tests/fm-documentation-audiences.test.sh +++ b/tests/fm-documentation-audiences.test.sh @@ -135,7 +135,26 @@ MD pass "local links resolve while dates, versions, commands, and incident prose remain semantically reviewed" } +test_no_mistakes_document_schema() { + local config="$ROOT/.no-mistakes.yaml" + assert_grep 'document:' "$config" "trusted Document config is missing" + assert_grep ' instructions: |' "$config" "Document instructions use an unsupported shape" + assert_grep 'docs/documentation-audiences.json' "$config" \ + "Document instructions do not point to the audience inventory" + assert_grep 'complete' "$config" \ + "Document instructions do not require a complete branch-diff review" + if command -v ruby >/dev/null 2>&1; then + ruby -e ' + require "yaml" + data = YAML.safe_load(File.read(ARGV.fetch(0))) + abort unless data.dig("document", "instructions").is_a?(String) + ' "$config" || fail ".no-mistakes.yaml did not parse document.instructions" + fi + pass "no-mistakes uses the supported trusted document.instructions schema" +} + test_repository_inventory_passes test_duplicate_and_setup_classification_fail test_required_pointer_fails test_local_links_and_no_keyword_heuristic +test_no_mistakes_document_schema diff --git a/tests/fm-test-run.test.sh b/tests/fm-test-run.test.sh index d07bf310c2b..7df9459d25b 100755 --- a/tests/fm-test-run.test.sh +++ b/tests/fm-test-run.test.sh @@ -11,6 +11,8 @@ set -u . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" RUNNER="$ROOT/bin/fm-test-run.sh" +CI="$ROOT/.github/workflows/ci.yml" +CONTRIB="$ROOT/CONTRIBUTING.md" assert_present "$RUNNER" "bin/fm-test-run.sh is missing" [ -x "$RUNNER" ] || fail "bin/fm-test-run.sh must be executable" @@ -95,7 +97,7 @@ init_changed_fixture_repo() { chmod +x "$repo/bin/fm-test-run.sh" for script in \ fm-brief.test.sh \ - fm-ask-user-authority.test.sh \ + fm-captain-translation-contract.test.sh \ fm-cd-pretool-check.test.sh \ fm-daemon.test.sh \ fm-backend-herdr-smoke.test.sh \ @@ -164,7 +166,7 @@ test_changed_dependency_selection_and_unmapped_failure() { printf '\n' >>"$repo/.pi/extensions/fm-primary-pi-watch.ts" printf '\n' >>"$repo/.pi/extensions/fm-primary-turnend-guard.ts" listed=$(cd "$repo" && bin/fm-test-run.sh --list --changed --base HEAD) - assert_contains "$listed" "tests/fm-ask-user-authority.test.sh" "skill source selects pure contract coverage" + assert_contains "$listed" "tests/fm-captain-translation-contract.test.sh" "skill source selects pure contract coverage" assert_contains "$listed" "tests/fm-cd-pretool-check.test.sh" "Claude and Pi source selects hook coverage" assert_contains "$listed" "tests/fm-pi-watch-extension.test.sh" "Pi source selects watcher coverage" git -C "$repo" add .agents .claude .pi @@ -350,6 +352,84 @@ test_exclude_family() { pass "exclude-family drops the named primary family after selection" } +test_ci_and_docs_call_the_owner() { + assert_present "$CI" "ci.yml missing" + assert_present "$CONTRIB" "CONTRIBUTING.md missing" + grep -Fq 'tests-portable-parallel-1:' "$CI" \ + || fail "CI must define portable parallel shard 1" + grep -Fq 'tests-portable-parallel-2:' "$CI" \ + || fail "CI must define portable parallel shard 2" + grep -Fq 'tests-portable-serial:' "$CI" \ + || fail "CI must define the portable serial lane" + grep -Fq 'bin/fm-test-run.sh --lane portable-parallel-1' "$CI" \ + || fail "CI shard 1 must invoke --lane portable-parallel-1" + grep -Fq 'bin/fm-test-run.sh --lane portable-parallel-2' "$CI" \ + || fail "CI shard 2 must invoke --lane portable-parallel-2" + local shard job_body + for shard in 1 2; do + job_body=$(awk -v job=" tests-portable-parallel-$shard:" ' + $0 == job { in_job=1; next } + in_job && /^ [a-zA-Z0-9_-]+:/ { exit } + in_job { print } + ' "$CI") + printf '%s\n' "$job_body" | grep -Fq 'npm install -g tasks-axi' \ + || fail "CI portable parallel shard $shard must install tasks-axi" + printf '%s\n' "$job_body" | grep -Fq 'tasks-axi --version' \ + || fail "CI portable parallel shard $shard must verify tasks-axi" + done + grep -Fq 'bin/fm-test-run.sh --lane portable-serial' "$CI" \ + || fail "CI portable serial must invoke --lane portable-serial" + grep -Fq 'bin/fm-test-run.sh --check-coverage' "$CI" \ + || fail "CI must run the coverage guard" + grep -Fq 'tests-herdr:' "$CI" \ + || fail "CI must define the required tests-herdr job" + grep -Fq 'bin/fm-test-run.sh --family real-herdr-gated' "$CI" \ + || fail "Herdr CI job must run the real-herdr-gated family via fm-test-run" + grep -Fq -- "--fail-on-gate-skip 'herdr not found'" "$CI" \ + || fail "Herdr CI job must fail on herdr-not-found skips" + grep -Fq 'bin/fm-install-herdr.sh' "$CI" \ + || fail "Herdr CI job must install via bin/fm-install-herdr.sh" + grep -Fq 'bin/fm-install-treehouse.sh' "$CI" \ + || fail "Herdr CI job must install via bin/fm-install-treehouse.sh" + grep -Fq 'bin/fm-herdr-ci-cleanup.sh' "$CI" \ + || fail "Herdr CI job must use bounded lab cleanup" + grep -Fq 'tests-timing-aggregate:' "$CI" \ + || fail "CI must aggregate per-lane timing artifacts" + grep -Fq 'timeout-minutes: 20' "$CI" \ + || fail "portable serial hang tripwire must be timeout-minutes: 20" + grep -Fq 'timeout-minutes: 10' "$CI" \ + || fail "portable parallel shards must keep a hang tripwire (10m)" + # Interim full-suite 25m portable timeout must not remain after sharding. + if grep -Eq 'timeout-minutes: 25' "$CI"; then + fail "CI still has interim timeout-minutes: 25 after portable sharding" + fi + # Stale "~2-3 minutes" claim must not remain. + if grep -Eq '2-3 minutes' "$CI"; then + fail "CI workflow still claims the suite finishes in ~2-3 minutes" + fi + # No retry-green strategy on Behavior lanes. + if grep -Eqi 'retry:|max-attempts:|continue-on-error:\s*true' "$CI"; then + fail "CI must not use retries or continue-on-error as a green strategy" + fi + grep -Fq 'fm-test-timing' "$CI" \ + || fail "CI must upload timing artifacts" + grep -Fq 'bin/fm-test-run.sh --all' "$CONTRIB" \ + || fail "CONTRIBUTING must document bin/fm-test-run.sh --all" + grep -Fq 'bin/fm-test-run.sh --family' "$CONTRIB" \ + || fail "CONTRIBUTING must document family selection" + grep -Fq 'bin/fm-test-run.sh --changed' "$CONTRIB" \ + || fail "CONTRIBUTING must document changed-file selection" + grep -Fq 'bin/fm-test-run.sh --proven-isolated --jobs' "$CONTRIB" \ + || fail "CONTRIBUTING must document proven-isolated --jobs" + grep -Fq 'intent-targeted' "$CONTRIB" \ + || fail "CONTRIBUTING must document intent-targeted no-mistakes Test" + # Do not restore a complete-suite commands.test. + if grep -E '^[[:space:]]*test:[[:space:]].*tests/\*\.test\.sh' "$ROOT/.no-mistakes.yaml" >/dev/null 2>&1; then + fail ".no-mistakes.yaml must not set a full-suite commands.test" + fi + pass "CI and CONTRIBUTING call the one-owner runner; no full-suite local Test" +} + test_portable_shard_union_and_coverage_guard() { local s1 s2 proven serial herdr all_count union_count overlap out first s1=$("$RUNNER" --list --lane portable-parallel-1) @@ -381,8 +461,8 @@ test_portable_shard_union_and_coverage_guard() { || fail "lanes must not duplicate scripts" # LPT order: first script of shard 1 is the longest proven script. first=$(printf '%s\n' "$s1" | head -n 1) - [ "$first" = "tests/fm-x-mode.test.sh" ] \ - || fail "shard 1 must start with the longest proven script, got $first" + [ "$first" = "tests/fm-arm-pretool-check.test.sh" ] \ + || fail "shard 1 must start with longest proven script, got $first" pass "portable shard union, disjointness, and coverage guard hold" } @@ -412,8 +492,8 @@ test_jobs_parallel_scheduler_and_failure_propagation() { runner="$repo/bin/fm-test-run.sh" evidence="$tmp/evidence" fake_bin="$tmp/fake-bin" - a=tests/fm-brief.test.sh - b=tests/fm-composer-lib.test.sh + a=tests/fm-no-mistakes-ownership.test.sh + b=tests/fm-stow-contract.test.sh c=tests/fm-lint.test.sh d=tests/fm-supervision-instructions.test.sh mkdir -p "$repo/bin" "$repo/tests" "$evidence" "$fake_bin" @@ -432,46 +512,22 @@ exit 1 SH cat >"$repo/$a" <<'SH' #!/usr/bin/env bash -touch "$SCHED_EVIDENCE/slow-started" -attempt=0 -while [ ! -e "$SCHED_EVIDENCE/release-slow" ]; do - attempt=$((attempt + 1)) - if [ "$attempt" -ge 1000 ]; then - echo "not ok - slow fixture was never released" - exit 1 - fi - sleep 0.01 -done +sleep 0.5 touch "$SCHED_EVIDENCE/slow-done" echo "ok - slow fixture" SH cat >"$repo/$b" <<'SH' #!/usr/bin/env bash -attempt=0 -while [ ! -e "$SCHED_EVIDENCE/slow-started" ]; do - attempt=$((attempt + 1)) - if [ "$attempt" -ge 1000 ]; then - echo "not ok - slow fixture never started" - exit 1 - fi - sleep 0.01 -done +sleep 0.05 echo "ok - fast fixture" SH cat >"$repo/$c" <<'SH' #!/usr/bin/env bash -if [ ! -e "$SCHED_EVIDENCE/slow-started" ]; then - touch "$SCHED_EVIDENCE/release-slow" - echo "not ok - replacement fixture started before both initial workers" - exit 1 -fi if [ -e "$SCHED_EVIDENCE/slow-done" ]; then - touch "$SCHED_EVIDENCE/release-slow" echo "not ok - scheduler waited for oldest worker" exit 1 fi -touch "$SCHED_EVIDENCE/release-slow" -echo "ok - replacement fixture released the blocked slow fixture" +echo "ok - replacement fixture started before slow fixture finished" SH chmod +x "$runner" "$repo/$a" "$repo/$b" "$repo/$c" "$fake_bin/stat" set +e @@ -602,6 +658,7 @@ test_aggregate_exit_behavior test_gate_skip_accounting test_fail_on_gate_skip_token test_exclude_family +test_ci_and_docs_call_the_owner test_portable_shard_union_and_coverage_guard test_jobs_requires_proven_isolated test_jobs_parallel_scheduler_and_failure_propagation From 2d1267b5488af9eb29ea7160306d835c78b4e510 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Fri, 24 Jul 2026 14:45:58 -0700 Subject: [PATCH 07/52] fix: preserve Claude watcher continuity across Stop hooks (#997) * feat(claude): Stop-owned tokenless watcher continuity via asyncRewake auto-arm Claude primaries (main home and marked secondmate homes) no longer depend on the model remembering to re-arm the watcher after each wake. A tracked Stop asyncRewake hook (bin/fm-claude-stop-autoarm.sh, timeout 28800s) fires on every turn end, claims one home-scoped single-flight owner, foregrounds bin/fm-watch-arm.sh inside the hook-owned process tree, and translates an actionable close or typed watcher failure into exactly one exit-2 rewake. The hook scopes to genuine primary checkouts, requires the session lock to be held by its own harness ancestor, stays inert while AFK owns triage or the home is idle, and hands AFK transitions mid-cycle to the daemon without rewaking. The synchronous turn-end guard gains a --claude cooperative mode: it ignores stop_hook_active (true on every post-continuation stop, which is what re-opened the 2026-07-21 blind window), waits briefly for a watcher health proof, a live auto-arm owner claim, or a fresh rewake epoch, and re-blocks only when the auto-arm genuinely failed to establish - bounded to 3 consecutive blocks per session, safely below Claude Code's 8-block override, then a degraded allow with a visible systemMessage. Codex keeps the previous one-block loop guard byte-identically, and Pi, OpenCode, and Grok adapters are untouched. Continuity PreToolUse gate and durable wake queue are preserved; the gate's recovery guidance now names the Stop-owned re-arm and reserves manual background arms for auto-arm failure. Claude supervision protocol, harness-adapters facts, architecture, configuration, and continuity docs updated; docs/turnend-guard.md records the 2026-07-24 Claude 2.1.218 contract revalidation (tokenless multi-cycle rewake, no-dedup, timeout process-group kill, 8-block cap, interactive non-stall) and the 2.1.219 product live E2Es. Regression matrix: hermetic tests cover scope, identity, AFK, need, single-flight, translation, guard cooperation, budget, and registration; the new live E2E proves two full tokenless auto-arm rewake cycles with zero model arm commands; Pi and OpenCode Option B live E2Es pass unchanged. * no-mistakes(review): Fix Claude X-mode auto-arm continuity backstop * no-mistakes(review): Remove unsupported Claude contract-lab verification claims * no-mistakes(document): Update Claude auto-arm continuity documentation --- .agents/skills/harness-adapters/SKILL.md | 12 +- .claude/settings.json | 8 +- AGENTS.md | 1 + README.md | 2 +- bin/fm-claude-stop-autoarm.sh | 36 +--- bin/fm-continuity-pretool-check.sh | 113 +++++++++++ bin/fm-lock.sh | 60 +----- bin/fm-session-lock-lib.sh | 73 ++----- bin/fm-turnend-guard.sh | 24 +-- docs/architecture.md | 6 +- docs/arm-pretool-check.md | 17 +- docs/configuration.md | 5 +- docs/subagent-guard.md | 31 +-- docs/supervision-protocols/claude.md | 5 +- docs/turnend-guard.md | 20 +- docs/verification/supervision.md | 32 ++- docs/watcher-continuity.md | 21 +- tests/fm-claude-continuity-live-e2e.test.sh | 73 +++++++ tests/fm-claude-stop-autoarm-live-e2e.test.sh | 72 +++---- tests/fm-claude-stop-autoarm.test.sh | 122 +++--------- tests/fm-continuity-pretool-check.test.sh | 128 ++++++++++++ tests/fm-turnend-guard.test.sh | 188 ++++++++---------- 22 files changed, 589 insertions(+), 460 deletions(-) create mode 100755 bin/fm-continuity-pretool-check.sh create mode 100755 tests/fm-claude-continuity-live-e2e.test.sh create mode 100755 tests/fm-continuity-pretool-check.test.sh diff --git a/.agents/skills/harness-adapters/SKILL.md b/.agents/skills/harness-adapters/SKILL.md index f67ec9c0d0a..c284ce30ed1 100644 --- a/.agents/skills/harness-adapters/SKILL.md +++ b/.agents/skills/harness-adapters/SKILL.md @@ -93,7 +93,7 @@ Full mechanics, scoping, and fail-open behavior live in `docs/sessionstart-nudge At session start, `bin/fm-session-start.sh` prints exactly one watcher supervision block for the detected primary harness. Do not substitute another harness's wait shape when resuming supervision. -Claude and Grok use tracked background-notify cycles around `bin/fm-watch-arm.sh`. +Claude's Stop `asyncRewake` hook (`bin/fm-claude-stop-autoarm.sh`) owns tokenless re-arm around `bin/fm-watch-arm.sh`, and Grok uses tracked background-notify cycles around `bin/fm-watch-arm.sh`. Codex uses bounded foreground checkpoints through `bin/fm-watch-checkpoint.sh` because Codex cannot reason while a foreground tool call is running. OpenCode uses `.opencode/plugins/fm-primary-watch-arm.js`, which coordinates with the turn-end guard plugin and wakes the TUI with `client.session.promptAsync`. Pi uses the tracked `.pi/extensions/fm-primary-turnend-guard.ts` plus the tracked `.pi/extensions/fm-primary-pi-watch.ts`, both project-local extensions Pi auto-discovers once trusted. @@ -158,13 +158,13 @@ Its broader dark-TRUECOLOR placeholder handling and dark-theme tradeoff are docu That styled capture is internal to the boolean detector only. `fm-peek` and every other human or LLM-facing capture path stays plain `tmux capture-pane` with no escape codes. -**Primary-session guard fact (verified 2026-07-04, Claude Code 2.1.201; preserved 2026-07-08, Claude Code 2.1.204).** +**Primary-session guard fact (verified 2026-07-04, Claude Code 2.1.201; preserved 2026-07-08, Claude Code 2.1.204; Stop-owned auto-arm revalidated 2026-07-24, Claude Code 2.1.219).** This is separate from the per-task crewmate turn-end hook above (that one just `touch`es a marker file in a task's own `.claude/settings.local.json`). -The firstmate PRIMARY's own `.claude/settings.json` registers `bin/fm-turnend-guard.sh` as a Stop hook, and exiting with status 2 plus stderr reliably forces the model to continue. -Claude Code's stdin payload to a Stop hook carries a `stop_hook_active` boolean that is `true` exactly when the current stop attempt is itself a forced continuation from an earlier block this turn; a hook can and should use that as its own loop-guard (always allow the stop when it is already `true`) rather than tracking state itself. +The firstmate PRIMARY's own `.claude/settings.json` registers two Stop hooks: `bin/fm-turnend-guard.sh --claude` and the Stop-owned auto-arm `bin/fm-claude-stop-autoarm.sh` (`asyncRewake: true`, `timeout: 28800`), and exiting the guard with status 2 plus stderr reliably forces the model to continue. +Claude Code's stdin payload to a Stop hook carries a `stop_hook_active` boolean that is `true` when the current stop attempt follows ANY stop-hook-driven continuation, including `asyncRewake` rewakes; the primary guard therefore ignores it in `--claude` mode and uses the cooperative claim/epoch check plus a bounded re-block budget instead, while the codex-mode default still treats it as a one-block loop guard. A project-level `.claude/settings.json` only takes effect when Claude Code's project root is that exact directory - it does not walk up from a subdirectory looking for one, so firstmate launches the primary from the repo root. -After those settings are loaded, hook command resolution is still cwd-sensitive because Claude Code runs commands through `/bin/sh` against the session's current cwd; keep the tracked command anchored through `"$CLAUDE_PROJECT_DIR"/bin/fm-turnend-guard.sh` and see `docs/turnend-guard.md` for the verified Stop-hook details. -Claude Code's primary watcher protocol is the lowest-friction path: run `bin/fm-watch-arm.sh` as its own Claude Code background task and treat background-task completion as the wake. +After those settings are loaded, hook command resolution is still cwd-sensitive because Claude Code runs commands through `/bin/sh` against the session's current cwd; keep the tracked commands anchored through `"$CLAUDE_PROJECT_DIR"/bin/...` and see `docs/turnend-guard.md` for the verified Stop-hook details. +Claude Code's primary watcher protocol is Stop-owned: the auto-arm hook fires on every Stop and foregrounds `bin/fm-watch-arm.sh` when the home is eligible and still needs supervision, and its exit-2 `asyncRewake` rewake is the wake; the model drains and handles wakes but never runs a routine re-arm command. ## codex (VERIFIED 2026-06-11, codex-cli 0.139.0) diff --git a/.claude/settings.json b/.claude/settings.json index 0be379c46b7..9f145c88913 100644 --- a/.claude/settings.json +++ b/.claude/settings.json @@ -22,6 +22,10 @@ { "type": "command", "command": "\"$CLAUDE_PROJECT_DIR\"/bin/fm-cd-pretool-check.sh --claude" + }, + { + "type": "command", + "command": "\"$CLAUDE_PROJECT_DIR\"/bin/fm-continuity-pretool-check.sh" } ] }, @@ -40,11 +44,11 @@ "hooks": [ { "type": "command", - "command": "[ -z \"${GROK_AGENT:-}\" ] || exit 0; exec \"$CLAUDE_PROJECT_DIR\"/bin/fm-turnend-guard.sh --claude" + "command": "\"$CLAUDE_PROJECT_DIR\"/bin/fm-turnend-guard.sh --claude" }, { "type": "command", - "command": "[ -z \"${GROK_AGENT:-}\" ] || exit 0; exec \"$CLAUDE_PROJECT_DIR\"/bin/fm-claude-stop-autoarm.sh", + "command": "\"$CLAUDE_PROJECT_DIR\"/bin/fm-claude-stop-autoarm.sh", "asyncRewake": true, "timeout": 28800 } diff --git a/AGENTS.md b/AGENTS.md index 96201b631d9..c8b61b8077d 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -106,6 +106,7 @@ state/ volatile runtime signals; gitignored .wake-queue durable queued wakes: epoch<TAB>seq<TAB>kind<TAB>key<TAB>payload .afk durable away-mode flag; present = sub-supervisor may inject escalations (set by /afk, cleared on user return) .watch.lock .wake-queue.lock watcher singleton and queue serialization locks + .claude-autoarm.lock .claude-autoarm-epoch .turnend-claude-blocks Claude Stop auto-arm single-flight, epoch, and guard-budget records; never touch .hash-* .count-* .stale-* .stale-since-* .paused-* .wedge-escalations-* .seen-* .hb-surfaced-* .last-* .heartbeat-streak watcher internals; never touch .watch-triage.log watcher's absorbed-wake debug log (size-capped); never relied on, safe to delete .last-watcher-beat watcher liveness beacon, touched every poll (including while absorbing benign wakes); guard scripts read it diff --git a/README.md b/README.md index 4fcd84330a4..c62e47bba48 100644 --- a/README.md +++ b/README.md @@ -68,7 +68,7 @@ Backend-specific setup is linked in [Documentation](#documentation). ### Recommended harnesses **Claude Code, Grok, and Pi are equal co-primary recommendations** for running the primary firstmate session. -Claude Code and Grok use background-notify wake cycles; Pi uses its tracked primary watcher extension. +Claude Code uses a tracked Stop hook for tokenless watcher re-arm and rewake, Grok uses background-notify wake cycles, and Pi uses its tracked primary watcher extension. All three have verified turn-end guard paths when launched with their documented setup. Pick whichever one matches your subscription and workflow. diff --git a/bin/fm-claude-stop-autoarm.sh b/bin/fm-claude-stop-autoarm.sh index df9ee1128fc..f730d35c7f7 100755 --- a/bin/fm-claude-stop-autoarm.sh +++ b/bin/fm-claude-stop-autoarm.sh @@ -10,11 +10,8 @@ # - Scope: only a genuine primary checkout (plain checkout or validly marked # secondmate home) with AGENTS.md, bin/, and the effective state dir - the # exact fm-turnend-guard.sh scope. Child crew/scout worktrees stay inert. -# - Identity: only when THIS session's harness ancestor holds state/.lock. -# When an existing numeric owner fails the shared harness-liveness predicate, -# the hook delegates guarded recovery to bin/fm-lock.sh and then re-verifies -# ownership. A live owner, missing lock, malformed lock, or unresolved -# ancestry remains inert, so a competing session never arms or rewakes. +# - Identity: only when THIS session's harness ancestor holds state/.lock, so +# a scratch or read-only session in the same checkout never arms or rewakes. # - AFK: while state/.afk exists the away daemon owns the watcher and triage; # this hook exits 0 and NEVER rewakes the primary (checked again at # translation time so a mid-cycle AFK transition is honored). @@ -40,9 +37,8 @@ # # This hook never blocks the Stop decision itself and never prints to stdout: # exit 0 is always silent, and exit 2 carries the rewake banner on stderr. -# On any uncertainty such as unresolvable ancestry, malformed lock state, or -# lock contention, it exits 0 and leaves continuity to the synchronous guard and -# the model. +# On any uncertainty such as unresolvable ancestry or lock contention, it exits +# 0 and leaves continuity to the synchronous guard and the model. set -u SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -71,20 +67,7 @@ cat >/dev/null 2>&1 || true fm_primary_scope_matches "$FM_ROOT" "$STATE" || exit 0 # --- identity: only the lock-owning session's hooks may arm ------------------ -# A prior session may have died after leaving its numeric harness pid in .lock. -# Use the shared liveness predicate to recognize only that stale-owner case. -# Defer the mutating claim until after the unchanged AFK and need gates, so an -# idle or away home remains byte-for-byte inert. Missing or malformed locks are -# uncertainty rather than stale-owner evidence and remain inert. -RECOVER_SESSION_LOCK=0 -if ! fm_session_lock_owned_by_self "$STATE"; then - LOCK_PID=$(cat "$STATE/.lock" 2>/dev/null || true) - case "$LOCK_PID" in - ''|*[!0-9]*) exit 0 ;; - esac - fm_harness_pid_alive "$LOCK_PID" && exit 0 - RECOVER_SESSION_LOCK=1 -fi +fm_session_lock_owned_by_self "$STATE" || exit 0 # --- AFK: the away daemon owns the watcher and triage; never rewake ---------- [ -e "$STATE/.afk" ] && exit 0 @@ -95,15 +78,6 @@ need_supervision() { } need_supervision || exit 0 -# --- stale session-lock recovery --------------------------------------------- -# Delegate the claim to fm-lock.sh so its live-owner refusal and write semantics -# remain the single acquisition owner, then re-verify current-session identity -# before touching any auto-arm state. -if [ "$RECOVER_SESSION_LOCK" -eq 1 ]; then - "$SCRIPT_DIR/fm-lock.sh" >/dev/null 2>&1 || exit 0 - fm_session_lock_owned_by_self "$STATE" || exit 0 -fi - # --- single-flight owner claim ------------------------------------------------ # Claude runs one background process per firing with no dedupe. Exactly one # owner foregrounds the arm and translates its close; every other firing exits diff --git a/bin/fm-continuity-pretool-check.sh b/bin/fm-continuity-pretool-check.sh new file mode 100755 index 00000000000..e8db8431929 --- /dev/null +++ b/bin/fm-continuity-pretool-check.sh @@ -0,0 +1,113 @@ +#!/usr/bin/env bash +# Claude primary watcher-continuity PreToolUse gate. +# +# This hook is deliberately narrow. It denies only an executed bin/fm-*.sh fleet +# command other than bin/fm-wake-drain.sh, bin/fm-watch-arm.sh, or the +# independently fail-closed bin/fm-teardown.sh, and only when the active primary +# home has task metadata in flight but no identity-matched live watcher holds the +# home lock. Ordinary shell commands, recovery commands, healthy supervision, +# fleet-idle homes, and child worktrees are always allowed. +# +# The turn-end guard remains the final backstop, cooperating with the +# Stop-owned auto-arm in its --claude mode. This gate closes the long-turn gap +# before another fleet mutation, but does not replace or weaken the Stop hooks. +# +# Input is Claude PreToolUse JSON on stdin. Tests may pass --command directly. +# Malformed transport, missing jq/Node, a missing classifier, or classifier +# failure all fail open. A deny writes Claude's hook decision to stderr only and +# exits 2. +set -u + +COMMAND= +COMMAND_SET=0 + +usage() { + cat <<'EOF' +Usage: fm-continuity-pretool-check.sh [--command <shell-command>] + +Reads Claude PreToolUse JSON from stdin unless --command is supplied. +Exits 0 to allow. Exits 2 with a Claude deny object on stderr only when an +unhealthy primary tries to execute a non-recovery firstmate fleet script. +EOF +} + +while [ "$#" -gt 0 ]; do + case "$1" in + --command) + [ "$#" -gt 1 ] || { echo "error: --command requires a value" >&2; exit 2; } + COMMAND=$2 + COMMAND_SET=1 + shift 2 + ;; + --command=*) + COMMAND=${1#--command=} + COMMAND_SET=1 + shift + ;; + -h|--help) + usage + exit 0 + ;; + *) + echo "error: unknown argument: $1" >&2 + usage >&2 + exit 2 + ;; + esac +done + +if [ "$COMMAND_SET" -eq 0 ]; then + PAYLOAD=$(cat 2>/dev/null || true) + [ -n "$PAYLOAD" ] || exit 0 + command -v jq >/dev/null 2>&1 || exit 0 + COMMAND=$(printf '%s' "$PAYLOAD" | jq -r '.tool_input.command // empty' 2>/dev/null) || exit 0 +fi +[ -n "$COMMAND" ] || exit 0 + +SCRIPT_DIR=$(CDPATH='' cd -- "$(dirname -- "${BASH_SOURCE[0]}")" 2>/dev/null && pwd -P) || exit 0 +FM_ROOT=${FM_ROOT_OVERRIDE:-$(CDPATH='' cd -- "$SCRIPT_DIR/.." 2>/dev/null && pwd -P)} +FM_HOME=${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}} +STATE=${FM_STATE_OVERRIDE:-$FM_HOME/state} +WATCH="$SCRIPT_DIR/fm-watch.sh" +POLICY="$SCRIPT_DIR/fm-continuity-command-policy.mjs" + +# shellcheck source=bin/fm-supervision-lib.sh +. "$SCRIPT_DIR/fm-supervision-lib.sh" +# shellcheck source=bin/fm-primary-scope-lib.sh +. "$SCRIPT_DIR/fm-primary-scope-lib.sh" +# shellcheck source=bin/fm-wake-lib.sh +. "$SCRIPT_DIR/fm-wake-lib.sh" + +fm_primary_scope_matches "$FM_ROOT" "$STATE" || exit 0 +fm_supervision_status "$STATE" "${FM_GUARD_GRACE:-300}" +[ "$FM_SUP_IN_FLIGHT" -gt 0 ] || exit 0 +LOCK_PID=$(cat "$STATE/.watch.lock/pid" 2>/dev/null || true) +if fm_pid_alive "$LOCK_PID" && fm_watcher_lock_matches_pid "$STATE" "$WATCH" "$LOCK_PID" "$FM_HOME"; then + exit 0 +fi + +command -v node >/dev/null 2>&1 || exit 0 +[ -f "$POLICY" ] || exit 0 +CLASSIFICATION=$(node "$POLICY" --command "$COMMAND" --root "$FM_ROOT" 2>/dev/null) || exit 0 +case "$CLASSIFICATION" in + deny*) ;; + *) exit 0 ;; +esac + +TAB=$(printf '\t') +REST=${CLASSIFICATION#*"$TAB"} +[ -n "$REST" ] && [ "$REST" != "$CLASSIFICATION" ] || exit 0 +BLOCKED_SCRIPT=${REST%%"$TAB"*} +REASON_CODE=${REST#*"$TAB"} +[ "$REASON_CODE" != "$REST" ] || REASON_CODE="" +case "$REASON_CODE" in + unsafe-teardown) + REASON="[watcher-continuity] tasks are in flight and no live watcher holds this home lock; during recovery only the ordinary literal bin/fm-teardown.sh is allowed, so drop --force and any shell-expanded arguments and retry the literal invocation (blocked: $BLOCKED_SCRIPT)" + ;; + *) + REASON="[watcher-continuity] tasks are in flight and no live watcher holds this home lock; drain wakes with bin/fm-wake-drain.sh, use fail-closed bin/fm-teardown.sh for completed tasks when needed, then end the turn so the Stop-owned auto-arm re-establishes the watcher; if the Stop auto-arm itself failed, re-arm manually with bin/fm-watch-arm.sh as a tracked Claude background task (blocked: $BLOCKED_SCRIPT)" + ;; +esac +ESCAPED=$(printf '%s' "$REASON" | sed -e 's/\\/\\\\/g' -e 's/"/\\"/g' | tr '\n' ' ') +printf '{"hookSpecificOutput":{"hookEventName":"PreToolUse","permissionDecision":"deny"},"systemMessage":"%s"}\n' "$ESCAPED" >&2 +exit 2 diff --git a/bin/fm-lock.sh b/bin/fm-lock.sh index 083675b2beb..2e82574ecaf 100755 --- a/bin/fm-lock.sh +++ b/bin/fm-lock.sh @@ -3,7 +3,7 @@ # Writes the harness (agent) process PID found by walking the shell's ancestry, # which lives as long as the firstmate session - unlike the transient subshell # PID of any one tool call, which is dead moments after it is written. -# Usage: fm-lock.sh acquire; exit 1 unless ownership is verified +# Usage: fm-lock.sh acquire; exit 1 if another live session holds it # fm-lock.sh status print holder and liveness; always exits 0 set -u @@ -12,10 +12,7 @@ FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" LOCK="$STATE/.lock" -mkdir -p "$STATE" 2>/dev/null || { - echo "error: cannot create session-lock state directory $STATE; operate read-only until resolved" >&2 - exit 1 -} +mkdir -p "$STATE" # Harness identity (FM_HARNESS_RE, ancestry walk, holder liveness) is owned by # the shared session-lock lib so the Claude Stop auto-arm applies the exact @@ -25,63 +22,18 @@ mkdir -p "$STATE" 2>/dev/null || { if [ "${1:-}" = "status" ]; then if [ ! -f "$LOCK" ]; then echo "lock: free"; exit 0; fi - old=$(cat "$LOCK" 2>/dev/null) || { - echo "lock: unreadable" - exit 0 - } + old=$(cat "$LOCK") if fm_harness_pid_alive "$old"; then echo "lock: held by live harness pid $old"; else echo "lock: stale (pid $old dead or not a harness)"; fi exit 0 fi me=$(fm_harness_ancestry_pid) || { echo "error: cannot locate harness process in ancestry" >&2; exit 1; } -probe=$(mktemp "$STATE/.lock-write.XXXXXX" 2>/dev/null) || { - echo "error: cannot write session lock; operate read-only until resolved" >&2 - exit 1 -} -rm -f "$probe" 2>/dev/null || { - echo "error: cannot clean session-lock publication probe; operate read-only until resolved" >&2 - exit 1 -} -# shellcheck source=bin/fm-wake-lib.sh -. "$SCRIPT_DIR/fm-wake-lib.sh" -CLAIM_LOCK="$STATE/.lock.acquire" -CLAIM_LOCK_HELD=0 -release_claim_lock() { - if [ "$CLAIM_LOCK_HELD" -eq 1 ]; then - fm_lock_release "$CLAIM_LOCK" - CLAIM_LOCK_HELD=0 - fi -} -trap release_claim_lock EXIT -trap 'exit 1' HUP INT TERM -fm_lock_acquire_wait "$CLAIM_LOCK" -CLAIM_LOCK_HELD=1 - -if [ -e "$LOCK" ] || [ -L "$LOCK" ]; then - if [ ! -f "$LOCK" ] || [ -L "$LOCK" ]; then - echo "error: session lock is not a regular file; operate read-only until resolved" >&2 - exit 1 - fi - old=$(cat "$LOCK" 2>/dev/null) || { - echo "error: session lock is unreadable; operate read-only until resolved" >&2 - exit 1 - } +if [ -f "$LOCK" ]; then + old=$(cat "$LOCK") if [ "$old" != "$me" ] && fm_harness_pid_alive "$old"; then echo "error: another live firstmate session holds the lock (pid $old); operate read-only until resolved" >&2 exit 1 fi fi -if ! { printf '%s\n' "$me" > "$LOCK"; } 2>/dev/null; then - echo "error: cannot write session lock; operate read-only until resolved" >&2 - exit 1 -fi -written=$(cat "$LOCK" 2>/dev/null) || { - echo "error: cannot verify session lock ownership; operate read-only until resolved" >&2 - exit 1 -} -if [ ! -f "$LOCK" ] || [ -L "$LOCK" ] || [ "$written" != "$me" ]; then - echo "error: session lock ownership verification failed; operate read-only until resolved" >&2 - exit 1 -fi -release_claim_lock +echo "$me" > "$LOCK" echo "lock acquired: harness pid $me" diff --git a/bin/fm-session-lock-lib.sh b/bin/fm-session-lock-lib.sh index 8343a8efd97..abc54e62853 100644 --- a/bin/fm-session-lock-lib.sh +++ b/bin/fm-session-lock-lib.sh @@ -9,76 +9,35 @@ # This file is sourced by scripts and has no side effects on source. # Known harness command names; extend when a new adapter is verified. -FM_HARNESS_RE='claude|codex|opencode|grok|kimi|^pi$|^pi-signed$' +FM_HARNESS_RE='claude|codex|opencode|grok|^pi$' -# Walk the current process ancestry (up to 16 hops) and print a harness pid. -# For every harness except Claude, the first match wins (innermost pid), which -# is where e.g. Pi's shared signed-wrapper ancestry actually holds the session: -# a "pi-signed" launcher can be the direct parent of the inner "pi" engine -# pid that owns the lock, and the wrapper pid above it is not that owner. -# Claude Code's bg-spare hook worker chain is the opposite shape: it nests -# several claude-named processes directly parent-child with no non-harness -# process between them, and the lock is held by the outermost pid of that -# run. So once a claude-named match is found, this keeps walking past it -# looking for a still-more-ancestral claude-named match, and stops the -# instant a non-match follows - never walking past that gap to an unrelated -# claude-named process further up the real process tree (e.g. the live -# session that launched a test as its own subprocess). The harness pid lives -# as long as the session, unlike the transient subshell pid of any one tool -# call. +# Walk the current process ancestry (up to 8 hops) and print the first pid whose +# command looks like a verified harness. The harness pid lives as long as the +# session, unlike the transient subshell pid of any one tool call. fm_harness_ancestry_pid() { - local pid=$$ comm args best='' bc extending=0 hit=0 is_claude=0 - for _ in 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16; do - comm=$(ps -o comm= -p "$pid" 2>/dev/null) || break + local pid=$$ comm args + for _ in 1 2 3 4 5 6 7 8; do + comm=$(ps -o comm= -p "$pid" 2>/dev/null) || return 1 args=$(ps -o args= -p "$pid" 2>/dev/null) - bc=$(basename -- "$comm") - hit=0; is_claude=0 - if printf '%s' "$bc" | grep -qE "$FM_HARNESS_RE"; then - hit=1 - case "$bc" in *claude*) is_claude=1 ;; esac - else - # Bare interpreter (e.g. node): match the harness name in its script path. - case "$comm" in - *node*|*python*) - if printf '%s' "$args" | grep -qE "$FM_HARNESS_RE"; then - hit=1 - case "$args" in *claude*) is_claude=1 ;; esac - fi - ;; - esac - fi - if [ "$hit" -eq 1 ]; then - best="$pid" - if [ "$is_claude" -eq 1 ]; then - extending=1 - else - break - fi - elif [ "$extending" -eq 1 ]; then - break + if printf '%s' "$(basename "$comm")" | grep -qE "$FM_HARNESS_RE"; then + echo "$pid"; return 0 fi + # Bare interpreter (e.g. node): match the harness name in its script path. + case "$comm" in + *node*|*python*) printf '%s' "$args" | grep -qE "$FM_HARNESS_RE" && { echo "$pid"; return 0; } ;; + esac pid=$(ps -o ppid= -p "$pid" 2>/dev/null | tr -d ' ') - [ -n "$pid" ] && [ "$pid" -gt 1 ] || break + [ -n "$pid" ] && [ "$pid" -gt 1 ] || return 1 done - [ -n "$best" ] && { echo "$best"; return 0; } return 1 } # True if $1 is a live process that looks like a verified harness. fm_harness_pid_alive() { - local pid=$1 comm args + local pid=$1 comm kill -0 "$pid" 2>/dev/null || return 1 comm=$(ps -o comm= -p "$pid" 2>/dev/null) || return 1 - if printf '%s' "$(basename -- "$comm")" | grep -qE "$FM_HARNESS_RE"; then - return 0 - fi - case "$comm" in - *node*|*python*) - args=$(ps -o args= -p "$pid" 2>/dev/null) - printf '%s' "$args" | grep -qE "$FM_HARNESS_RE" - ;; - *) return 1 ;; - esac + printf '%s' "$(basename "$comm") $(ps -o args= -p "$pid" 2>/dev/null)" | grep -qE "$FM_HARNESS_RE" } # True when state dir $1 holds a session lock whose pid is the harness ancestor diff --git a/bin/fm-turnend-guard.sh b/bin/fm-turnend-guard.sh index 2e96fb33e48..515a859cd2b 100755 --- a/bin/fm-turnend-guard.sh +++ b/bin/fm-turnend-guard.sh @@ -11,10 +11,8 @@ # This script is push-based: verified harness turn-end hooks invoke it every time # the primary is about to end a turn. # Claude and codex can block directly by preserving exit status 2 and stderr. -# OpenCode and pi adapters use the same predicate and force one bounded -# follow-up because their turn-end events are passive. Grok delegates native -# blocking when its running Stop payload advertises that capability, with one -# bounded resume fallback for payloads from pre-native processes. +# OpenCode, pi, and grok adapters use the same predicate and force one bounded +# follow-up because their turn-end events are passive. # See docs/turnend-guard.md for the per-harness mechanics, validation evidence, # and fail-open tradeoffs. # @@ -28,10 +26,10 @@ # primary checkout - the main home or a genuinely marked secondmate home - and # stay a silent, fast no-op inside child task worktrees. # -# Loop-guard, codex/Grok (default) mode: never block twice in the same turn. -# Codex uses stop_hook_active and Grok uses stopHookActive; typed camel-case -# takes precedence when both spellings are present. A true value means the -# current stop attempt already follows a block, so this guard always allows it. +# Loop-guard, codex (default) mode: never block twice in the same turn. Codex +# Stop payloads carry stop_hook_active=true when the CURRENT stop attempt was +# itself already forced by an earlier block this turn; on that signal we always +# allow the stop, whether or not watcher supervision actually got resumed. # Passive harness adapters provide their own one-follow-up guard before calling # this script. # That bounds those harnesses to at most one forced continuation per turn - @@ -96,15 +94,7 @@ PAYLOAD=$(cat 2>/dev/null || true) # loop-guard field, so we must never block - fail open, not noisy. command -v jq >/dev/null 2>&1 || exit 0 -STOP_HOOK_ACTIVE=$(printf '%s' "$PAYLOAD" | jq -r ' - if type != "object" then error("payload") - elif has("stopHookActive") then - if ((.stopHookActive | type) == "boolean") then .stopHookActive else error("stopHookActive") end - elif has("stop_hook_active") then - if ((.stop_hook_active | type) == "boolean") then .stop_hook_active else error("stop_hook_active") end - else false - end -' 2>/dev/null) || exit 0 +STOP_HOOK_ACTIVE=$(printf '%s' "$PAYLOAD" | jq -r '.stop_hook_active // false' 2>/dev/null) || exit 0 if [ "$CLAUDE_MODE" -eq 0 ] && [ "$STOP_HOOK_ACTIVE" = "true" ]; then exit 0 fi diff --git a/docs/architecture.md b/docs/architecture.md index eed3f2c9358..8671b44045e 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -54,13 +54,13 @@ The default path remains local-only; live GitHub enrichment exists only behind t Optional X mode integrates with the watcher only after explicit opt-in; [configuration.md](configuration.md#x-mode-env) owns its generated-artifact and dispatch mechanics. At session start, `bin/fm-session-start.sh` emits exactly one primary-harness supervision block rendered by `bin/fm-supervision-instructions.sh` from `docs/supervision-protocols/`. -That block owns the live wait shape for the running primary harness: Claude and Grok use background-notify cycles, Codex uses bounded foreground checkpoints, Pi uses its two tracked primary extensions, and OpenCode uses its TUI plugin. +That block owns the live wait shape for the running primary harness: Claude's Stop `asyncRewake` hook owns tokenless re-arm cycles, Grok uses background-notify cycles, Codex uses bounded foreground checkpoints, Pi uses its two tracked primary extensions, and OpenCode uses its TUI plugin. `bin/fm-watch-arm.sh` remains the verified arm wrapper for protocols that call it; it forks the watcher as a tracked child, verifies it is genuinely alive with a fresh liveness beacon, and prints an honest `started`, `attached`, or nonzero `FAILED` status. On `attached` it stays live across identity-matched successors, and an unexplained clean child close either attaches to a verified healthy successor or becomes the typed nonzero `watcher: FAILED - cycle ended without an actionable reason` result. The arm layer records one bounded lifecycle row per observed cycle in `state/.watch-cycle-exits.log`; `state/.watch-triage.log` remains exclusively the absorbed-wake debug log. Pi and OpenCode verify session-lock ownership and launch one singleton successor from their child-close handlers before delivering an actionable wake prompt, with bounded exponential retry for failed restoration. -Claude keeps its tracked background-task protocol and adds a narrow PreToolUse continuity gate that allows drain, arm recovery, and fail-closed teardown while refusing only other fleet commands when tasks are in flight and no identity-matched live watcher holds the home lock. -The existing turn-end guard is unchanged and remains the final backstop for all five harness protocols. +Claude's `bin/fm-claude-stop-autoarm.sh` hook fires on every Stop and, when the home is eligible and still needs supervision, claims one home-scoped cycle, foregrounds the arm wrapper, and translates an actionable close or typed failure into one exit-2 rewake; its narrow PreToolUse continuity gate allows drain, arm recovery, and fail-closed teardown while refusing only other fleet commands when tasks are in flight and no identity-matched live watcher holds the home lock. +The existing turn-end guard remains the final backstop for all five harness protocols, cooperating with the auto-arm claim in its `--claude` mode. Its `--restart` mode signals only the watcher recorded in the current home's `state/.watch.lock`, so restarting one home cannot kill sibling secondmate watchers. A pull-based guard (`bin/fm-guard.sh`) warns through supervision tool output if the primary checkout is tangled, or if tasks are in flight and that watcher stops running or queued wakes are waiting to be drained. The drain script calls that guard after emptying the queue, which avoids repeating the queued-wakes warning for records it just consumed while still warning on stale watcher liveness. diff --git a/docs/arm-pretool-check.md b/docs/arm-pretool-check.md index f4747e0abdc..a5749ac786b 100644 --- a/docs/arm-pretool-check.md +++ b/docs/arm-pretool-check.md @@ -15,6 +15,17 @@ The seatbelt rejects those command shapes before execution. This policy is not a post-arm liveness guarantee. `bin/fm-guard.sh`, `bin/fm-turnend-guard.sh`, the watcher lock, and the watcher beacon still prove whether supervision is healthy after an allowed call. +## Claude continuity gate + +Claude also registers `bin/fm-continuity-pretool-check.sh` for Bash PreToolUse events. +This is a separate, tightly bounded recovery gate rather than another watcher-shape policy. +It runs only in a primary home, and it denies only an executed `bin/fm-*.sh` command other than `bin/fm-wake-drain.sh`, `bin/fm-watch-arm.sh`, or the ordinary literal `bin/fm-teardown.sh` when task metadata is in flight and no identity-matched live watcher holds that home's lock. +Ordinary shell commands, fleet-script names used as data, all commands in an idle fleet, child worktrees, wake drain, watcher arm, and ordinary literal teardown remain allowed. +The denial gives Claude reason-specific recovery guidance - drain, use fail-closed `bin/fm-teardown.sh` for completed tasks, then end the turn so the Stop-owned auto-arm re-establishes the watcher, with a manual tracked Claude background-task arm only when the Stop auto-arm itself failed - per the contract in [`watcher-continuity.md`](watcher-continuity.md). +`bin/fm-continuity-command-policy.mjs` reuses this document's shell lexer and command-position analysis but owns the recovery-versus-other-fleet classification. +Malformed transport or opaque dynamic syntax fails open so this narrow gate cannot become a blanket Bash block. +The existing `bin/fm-turnend-guard.sh` Stop integration remains the final backstop, cooperating with the Stop-owned auto-arm in its `--claude` mode ([`turnend-guard.md`](turnend-guard.md)). + The classifier never executes, sources, evaluates, or expands any part of the submitted command. It tokenizes the bytes and classifies lexical execution positions only. @@ -24,7 +35,7 @@ It tokenizes the bytes and classifies lexical execution positions only. - Stdin JSON at `.tool_input.command` for Claude and Codex. - Stdin JSON at `.toolInput.command` for Grok. -- `--command <exact string>` for OpenCode, Pi, and pi-signed. +- `--command <exact string>` for OpenCode and Pi. - `--background` as a compatibility-only field that never changes the decision. - `--claude` to preserve Claude's stderr-only deny requirement. @@ -151,7 +162,7 @@ Prose may improve without changing adapter behavior. - `--claude` suppresses stdout completely because Claude ignores a PreToolUse deny when stdout is nonempty. - Codex blocks on exit 2 and displays stderr. - OpenCode throws only when the checker exits 2. -- Pi and pi-signed return `{block: true}` only when the checker exits 2. +- Pi returns `{block: true}` only when the checker exits 2. ## Harness wiring @@ -161,7 +172,7 @@ Prose may improve without changing adapter behavior. | Claude | `.tool_input.command` | `.claude/settings.json` forwards stdin with `--claude`, leaving stdout empty and returning the stderr deny object. | | Grok | `.toolInput.command` | `.grok/hooks/fm-primary-pretool-check.json` forwards stdin and Grok consumes the stdout `decision=deny` object. | | OpenCode | `output.args.command` | `.opencode/plugins/fm-primary-pretool-check.js` passes one `--command` argument and throws only for exit 2. | -| Pi / pi-signed | `event.input.command` | `.pi/extensions/fm-primary-turnend-guard.ts` passes one `--command` argument and returns `{block: true}` only for exit 2. | +| Pi | `event.input.command` | `.pi/extensions/fm-primary-turnend-guard.ts` passes one `--command` argument and returns `{block: true}` only for exit 2. | Grok project hooks require folder trust. Every shell variable reference in a Grok hook command must carry an inline default such as `${GROK_WORKSPACE_ROOT:-}` because Grok expands the raw hook command before `bash -lc` runs it. diff --git a/docs/configuration.md b/docs/configuration.md index abb8bfe9d4b..e392840769d 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -179,7 +179,7 @@ The verified adapter knowledge - busy signatures, interrupt and exit commands, s Launch mechanics, including the verified command templates, live in [`bin/fm-spawn.sh`](../bin/fm-spawn.sh). Primary-session turn-end guard integrations for verified harnesses are tracked as repo-level hook files and documented in [`docs/turnend-guard.md`](turnend-guard.md). Primary-session watcher wake protocols are rendered at session start by [`bin/fm-supervision-instructions.sh`](../bin/fm-supervision-instructions.sh) from [`docs/supervision-protocols/`](supervision-protocols/). -Claude and Grok use background-notify cycles, Codex uses bounded foreground checkpoints, Pi uses its two tracked primary extensions, and OpenCode uses its TUI plugin. +Claude's Stop `asyncRewake` hook owns tokenless re-arm cycles, Grok uses background-notify cycles, Codex uses bounded foreground checkpoints, Pi uses its two tracked primary extensions, and OpenCode uses its TUI plugin. `config/crew-harness` is a local, gitignored file containing one adapter name for crewmate and scout launches. When it is absent or contains `default`, crewmates mirror the firstmate's own harness. `config/secondmate-harness` is a separate local, gitignored file containing the adapter the primary uses to launch secondmate agents, optionally followed by model and effort tokens on the same line. @@ -405,6 +405,9 @@ FMX_FOLLOWUP_MAX_AGE_SECS=604800 # local window for posting X-mode completion FMX_FOLLOWUP_MAX_COUNT=3 # local cap on X-mode completion follow-ups per linked mention FM_LOCK_STALE_AFTER=2 # seconds before dead-pid lock records can be reclaimed; mid-acquire locks keep at least 2s grace FM_GUARD_GRACE=300 # seconds before guard warnings, arm health checks, and the primary turn-end guard treat a watcher beacon as stale +FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=800 # milliseconds the --claude turn-end guard waits for the Stop auto-arm's claim, health, or fresh rewake epoch before re-blocking +FM_CLAUDE_AUTOARM_EPOCH_FRESH=15 # seconds a recorded auto-arm rewake outcome counts as this event epoch's owned recovery +FM_CLAUDE_TURNEND_BLOCK_BUDGET=3 # consecutive --claude guard re-blocks before a degraded allow; safely below Claude Code's 8-block override FM_ARM_CONFIRM_TIMEOUT=10 # seconds fm-watch-arm waits to confirm a fresh watcher before reporting FAILED FM_ARM_ATTACH_POLL=0.5 # seconds between checks while fm-watch-arm is attached to an existing healthy watcher cycle FM_OPENCODE_ARM_READY_TIMEOUT_MS=12000 # milliseconds the OpenCode primary watcher plugin waits for an arm attempt to report started, healthy, wake, or failure diff --git a/docs/subagent-guard.md b/docs/subagent-guard.md index 47aaf10e0f3..3bb39deb899 100644 --- a/docs/subagent-guard.md +++ b/docs/subagent-guard.md @@ -18,7 +18,7 @@ Three consequences were observed, not hypothesized. The deeper defect is that the bypass did not merely skip dispatch, it made the in-flight-work branch of the guard stack structurally inert. Only `bin/fm-spawn.sh` writes `state/<id>.meta`, so untracked project work contributes nothing to the in-flight count used by `bin/fm-supervision-lib.sh` and `bin/fm-turnend-guard.sh`. -Work started through the harness's own delegation tool writes no metadata, so the in-flight count stayed at zero and the turn-end guard never blocked a blind turn end. +Work started through the harness's own delegation tool writes no metadata, so the in-flight count stayed at zero, the turn-end guard never blocked a blind turn end, and the continuity gate was inert. That is the reason the fence has to sit on the harness tool surface, before the primary can create untracked work. No additional guard keyed on task metadata can catch this class of failure, because the failure is precisely the absence of that metadata. @@ -47,22 +47,14 @@ agent subagent task workflow cron schedul worktree delegate spawn dispatch handoff remote sendmessage monitor ``` -Three exclusions keep the shape test from producing false positives. +Two exclusions keep the shape test from producing false positives. - A name beginning `mcp__` is never classified. An MCP server chooses its own tool names, a task or agent noun there is common, and it has no bearing on fleet dispatch. -- `OBSERVE_ONLY_TOOLS`: the exact names `taskoutput`, `taskstop`, `taskget`, `tasklist`, `cronlist`, `bashoutput`, and `killshell` are allowed. +- The exact names `taskoutput`, `taskstop`, `taskget`, `tasklist`, `cronlist`, `bashoutput`, and `killshell` are allowed. These observe or stop work that already exists rather than creating it, and denying them at this layer could strand already-running work with no way to inspect or end it. A Claude primary's optional local deny list may still remove them from the schema. The shipped guard stays narrower on purpose so it can never be the reason a runaway task cannot be stopped. -- `PLAN_ONLY_TOOLS`: the exact names `taskcreate` and `taskupdate` are allowed. - These write, which is why they are a separate list rather than more entries in the observe-or-stop one, but what they write is the harness's session-local todo list. - That list has no executor: it spawns no agent, allocates no worktree, registers no schedule, and starts nothing that could outlive the session or escape a firstmate guard. - So it is not the "work, agent, schedule, or isolated workspace that firstmate would not know about" the guard exists to stop, and the stem match on `task` is a false positive rather than a policy. - The cost of the false positive was concrete: the primary could not track its own plan, and the deny text told it to run `bin/fm-brief.sh` and `bin/fm-spawn.sh` to create a todo entry. - -Both exclusion lists match the whole normalized name, never a substring, so neither can widen by accident: `TaskCreateAgent` and `RemoteTaskCreate` stay denied. -Folding the two lists together would be the drift risk, because the observe-or-stop rationale is not true of a tool that writes. The shipped guard fires on every delegation-shaped name that reaches it, including future names that no deny list knows about yet. That future-name behavior is the reason the tracked matcher must match all tools and let the script filter. @@ -87,8 +79,10 @@ Claude primaries should add this deny list in untracked per-home local settings, "CronCreate", "CronDelete", "CronList", + "TaskCreate", "TaskGet", "TaskList", + "TaskUpdate", "TaskStop", "TaskOutput" ] @@ -109,11 +103,8 @@ It is not tracked for two reasons. The width of the list remains a captain-owned decision, because denying some of these changes how the captain works with the primary session. Keep it as one flat local array that is reviewable at a glance and narrowable in one line. -In particular `TaskOutput`, `TaskStop`, `TaskGet`, `TaskList`, and `CronList` only observe or stop work that already exists, yet the recommended local deny list still removes all five by default. -The hook deliberately allows those five, so the shipped guard can never strand a runaway task with no way to inspect or end it, and it allows `TaskCreate` and `TaskUpdate` too, so it can never be the reason the primary cannot track its own plan. -The two session-local todo tools are no longer recommended for local denial at all, because they write only the harness's session-local todo list, which has no executor and spawns nothing, so removing them from the schema removes no delegation power. -Denying them there would instead reproduce at a stronger layer the exact false positive the shipped guard now avoids, leaving anyone who adopts this list verbatim unable to let a primary track its own plan. -Narrowing the list further, including the five observe-or-stop names, is the captain's call, and this local list is the only layer that can remove a todo tool from the primary's schema. +In particular `TaskOutput`, `TaskStop`, `TaskGet`, `TaskList`, and `CronList` only observe or stop work that already exists, but the recommended local deny list still removes them by default. +The hook deliberately allows those names, so the shipped guard can never strand a runaway task with no way to inspect or end it. `permissions.allow` is a pre-approval list, not an availability list, so there is no fail-closed positive allowlist available. That is why any fixed deny list is fail-open against future tools and why the shape-based guard still exists. @@ -180,7 +171,7 @@ Applicability turns on one question: does the harness expose built-in delegation | Harness | Delegation surface | Status | | --- | --- | --- | -| Claude | 16 known tools, listed above | Scoped guard wired and live-verified; untracked local deny list verified and recommended. | +| Claude | 18 known tools, listed above | Scoped guard wired and live-verified; untracked local deny list verified and recommended. | | Codex | none | Not applicable, verified empirically below. Codex 0.144.1 exposes no subagent, sub-task, or delegated-agent tool, so there is nothing to remove or intercept. `.codex/hooks.json` is unchanged. | | Grok | present, exact tokens unconfirmed | Not wired pending live verification. See below. | | OpenCode | present, exact tokens unconfirmed | Not wired pending live verification. See below. | @@ -294,8 +285,8 @@ This distinction matters when reading the next result: a tool absent from a plai ### Local deny-list hardening -Run in a scratch firstmate-shaped project containing `AGENTS.md`, `state/`, a full copy of `bin/`, and a Claude settings file containing the local deny list exactly as recommended on that date, which was the 18-name form that still included `TaskCreate` and `TaskUpdate`. -The result validates that local deny list rather than tracked repo state, and the recommendation above has since dropped those two session-local todo tools. +Run in a scratch firstmate-shaped project containing `AGENTS.md`, `state/`, a full copy of `bin/`, and a Claude settings file containing the recommended local deny-list JSON above. +The result validates the recommended local deny-list JSON above, not tracked repo state. Asking for deferred entries explicitly returned: ```text @@ -353,7 +344,7 @@ The live consequence is confirmed by the shipped-guard result above: Claude hono ## Automated validation `tests/fm-subagent-pretool-check.test.sh` owns the acceptance matrix and is registered in the `pure-contract-unit` family in `bin/fm-test-run.sh`. -It covers the tracked Claude settings boundary that forbids a `permissions` key; the match-all Claude hook registration; denial of every work-creating delegation tool by shape; denial of twelve hypothetical future tool names that appear on no list; the observe-or-stop, plan-only, and MCP exclusions; the exactness of the plan-only exclusion against six near-miss names a substring or shorter-stem widening would release; the scout-present and scout-absent message variants; the escape hatch including its fail-closed values; inertness in a linked task worktree and in a non-firstmate repo; in-scope enforcement for a marked secondmate home; both stdin transports; the empty-stdout requirement; fail-open transport behavior; and the preserved `Bash` seatbelts and `Stop` guard. +It covers the tracked Claude settings boundary that forbids a `permissions` key; the match-all Claude hook registration; denial of every work-creating delegation tool by shape; denial of twelve hypothetical future tool names that appear on no list; the observe-or-stop and MCP exclusions; the scout-present and scout-absent message variants; the escape hatch including its fail-closed values; inertness in a linked task worktree and in a non-firstmate repo; in-scope enforcement for a marked secondmate home; both stdin transports; the empty-stdout requirement; fail-open transport behavior; and the preserved `Bash` seatbelts and `Stop` guard. Run: diff --git a/docs/supervision-protocols/claude.md b/docs/supervision-protocols/claude.md index c9913553102..4e6ccb5fcaa 100644 --- a/docs/supervision-protocols/claude.md +++ b/docs/supervision-protocols/claude.md @@ -15,9 +15,8 @@ When this session owns supervision and away mode is not active: A shell `&`, a truncating pipe, or bundling is denied automatically by the PreToolUse seatbelt (`bin/fm-arm-pretool-check.sh`) registered in `.claude/settings.json`. 6. Treat `watcher: started ...` and `watcher: attached ...` inside arm output as proof that one live cycle exists. On attach, the arm follows verified identity-matched successors instead of exiting when the first cycle ends. -7. The durable wake queue preserves actionable events between a rewake and the next Stop-launched arm, while the bounded turn-end guard prevents a blind Stop when recovery did not start. - No PreToolUse hook denies fleet commands based on watcher status. - [`watcher-continuity.md`](../watcher-continuity.md) owns the exact session-lock recovery boundary. +7. The continuity PreToolUse gate allows wake drain, watcher arm recovery, and fail-closed teardown, and refuses only other `bin/fm-*.sh` fleet commands while tasks are in flight and no identity-matched live watcher holds the home lock. + It covers the bounded gap between a rewake and the next Stop-launched arm. 8. The turn-end guard (`bin/fm-turnend-guard.sh --claude`) remains the final backstop. It allows the stop when a watcher is healthy, when the auto-arm already owns recovery for this event epoch, or when a fresh rewake is recorded; it re-blocks only when none of those materialize, within a bounded budget. 9. Waiting on the hook-owned cycle is silent: do not send idle progress while the watcher is parked. diff --git a/docs/turnend-guard.md b/docs/turnend-guard.md index a015cfe02cc..08e1dc177f9 100644 --- a/docs/turnend-guard.md +++ b/docs/turnend-guard.md @@ -26,7 +26,8 @@ That check keeps crewmate and scout linked worktrees inert because their git dir It also requires `AGENTS.md`, `bin/`, and the effective state directory. For an in-scope primary, the guard counts in-flight work from `state/*.meta`. -It exits silently with no work in flight. +The default cross-harness mode exits silently with no work in flight. +Claude's `--claude` mode also treats `state/x-watch.check.sh` as supervision need, so X-mode relay polling remains guarded without an in-flight task. Otherwise it calls `fm_watcher_healthy <state-dir> <watch-path> [grace-seconds] [home]` from `bin/fm-wake-lib.sh`, the same identity-matched lock and fresh-beacon check used by `bin/fm-watch-arm.sh`. A stale beacon blocks even when a watcher pid is live. A fresh leftover beacon blocks when the lock is missing, dead, or identity-mismatched. @@ -37,7 +38,7 @@ If `jq` is missing or hook stdin is empty, the guard exits 0 because it cannot s ## Harness integrations -- Claude registers a `Stop` hook in `.claude/settings.json`, anchored through `CLAUDE_PROJECT_DIR`. +- Claude registers two `Stop` hooks in `.claude/settings.json`, both anchored through `CLAUDE_PROJECT_DIR`: `bin/fm-turnend-guard.sh --claude`, and `bin/fm-claude-stop-autoarm.sh` with `asyncRewake: true` and `timeout: 28800`. - Codex registers a `Stop` hook in `.codex/hooks.json`, anchors the executable to the hook process working directory, verifies a Firstmate-shaped hook-bearing root, and passes the original payload to the shared guard. - OpenCode listens for `session.idle` in `.opencode/plugins/fm-primary-turnend-guard.js`, lets the watcher coordinator act first, and calls `client.session.promptAsync` once when the guard returns 2. - Pi listens for `agent_settled` in `.pi/extensions/fm-primary-turnend-guard.ts`, runs once per logical agent run, and calls `pi.sendUserMessage(..., { deliverAs: "followUp" })` once when the guard returns 2. @@ -45,7 +46,14 @@ If `jq` is missing or hook stdin is empty, the guard exits 0 because it cannot s The adapter intentionally omits `--permission-mode`, so a passive hook cannot grant stronger permissions than the resumed session default. Claude and Codex can block a Stop directly with exit status 2 and stderr. -Both payloads carry `stop_hook_active`; a true value lets the second stop finish after one forced continuation. +Both payloads carry `stop_hook_active`. +In the default Codex mode, a true value lets the second stop finish after one forced continuation. + +Claude runs the guard with `--claude`, which ignores `stop_hook_active` and cooperates with the Stop-owned auto-arm. +Claude Code sets `stop_hook_active=true` on every stop after any stop-hook continuation, including `asyncRewake` rewakes, which re-opened the 2026-07-21 blind window under the default one-shot behavior. +The Claude mode waits up to `FM_CLAUDE_AUTOARM_SYNC_WAIT_MS` (default 800 milliseconds) and allows the stop when the watcher is healthy, `state/.claude-autoarm.lock` has a live owner, or `state/.claude-autoarm-epoch` contains a fresh rewake outcome. +When none of those proofs appears, it re-blocks up to `FM_CLAUDE_TURNEND_BLOCK_BUDGET` times (default 3, below Claude's 8-block override), then allows degraded with a visible `systemMessage`. +Any allow resets the budget. OpenCode, Pi, and Grok expose passive callbacks for this purpose. Their adapters fail open at the hook boundary to protect the user session but schedule one bounded follow-up when the predicate blocks. @@ -61,7 +69,7 @@ That warning uses `bin/fm-supervision-instructions.sh --repair-line`, so it alwa ## Compatibility limits - Child crewmate and scout worktrees are outside scope. -- A valid secondmate home is in scope; an idle secondmate endpoint remains healthy because no work is in flight there. +- A valid secondmate home is in scope; an idle secondmate endpoint with no X-mode relay poll remains healthy because it has no supervision need. - Claude and Codex block directly, while OpenCode, Pi, and Grok use bounded passive follow-ups. - OpenCode headless mode and untrusted Grok project hooks remain fail-open at the host boundary. - Missing `jq` or unreadable hook input remains fail-open. @@ -69,7 +77,7 @@ That warning uses `bin/fm-supervision-instructions.sh --repair-line`, so it alwa ## Regression coverage -`tests/fm-turnend-guard.test.sh` covers the predicate, main and secondmate primary scope, child-worktree exclusion, `FM_HOME` and `FM_STATE_OVERRIDE` precedence, Pi logical-run latching, missing-`jq` behavior, all five registrations, and Grok resume permission and recursion safety. +`tests/fm-turnend-guard.test.sh` covers the predicate, main and secondmate primary scope, child-worktree exclusion, `FM_HOME` and `FM_STATE_OVERRIDE` precedence, the cooperative `--claude` claim wait, epoch allow, re-block budget, Pi logical-run latching, missing-`jq` behavior, all five registrations, and Grok resume permission and recursion safety. `tests/fm-supervision-instructions.test.sh` covers recovery-line ownership. `FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh` is the opt-in isolated Pi path. -[`verification/supervision.md`](verification/supervision.md#turn-end-guard) records the active cross-harness empirical evidence. +[`verification/supervision.md`](verification/supervision.md#turn-end-guard) records the active cross-harness empirical evidence, including the 2026-07-24 Claude `asyncRewake` revalidation. diff --git a/docs/verification/supervision.md b/docs/verification/supervision.md index 920d172cd64..2896cf7bf8f 100644 --- a/docs/verification/supervision.md +++ b/docs/verification/supervision.md @@ -61,18 +61,34 @@ The detailed reconciliation and task chronology stay in the private audit report ## Turn-end guard -The direct and passive mechanisms were validated across all five harnesses on 2026-07-08 through 2026-07-12. +The direct and passive mechanisms were validated across all five harnesses on 2026-07-08 through 2026-07-12, with Claude's replacement Stop-owned path revalidated on 2026-07-24. | Harness | Version verified | Mechanism | Observed result | | --- | --- | --- | --- | -| Claude | 2.1.204 | Blocking `Stop` hook | First stop blocked, one continuation ran, `stop_hook_active=true` allowed the second stop. | +| Claude | 2.1.219 | Cooperative blocking `Stop` guard plus `asyncRewake` auto-arm | Two tokenless auto-arm rewake cycles completed with no model arm command or guard continuation; deterministic coverage re-blocked genuine auto-arm failure despite `stop_hook_active=true`. | | Codex | 0.142.1 | Blocking `Stop` hook | Hook process root stayed anchored to the trusted checkout and one continuation ran. | | OpenCode | 1.17.6 | Passive `session.idle` callback | Throwing could not block, while `promptAsync` scheduled one TUI follow-up; headless remained fail-open. | | Pi | 0.80.5 | Passive `agent_settled` callback | Exactly one guard follow-up ran for an unhealthy cycle, with no recursion across tool turns. | | Grok | 0.2.93 | Passive `Stop` plus bounded resume | Project hook ran under trust, resumed once without inherited bypass permissions, and the environment latch prevented recursion. | -The secondmate-home scope was measured with Claude Code 2.1.207 on 2026-07-12. -A native background completion re-invoked the idle model with no human input, while deterministic tests covered main/secondmate inclusion and child-worktree exclusion. +The secondmate-home scope and manual-repair wake path were measured with Claude Code 2.1.207 on 2026-07-12, when a native background completion re-invoked the idle model with no human input. +The current Stop-owned main/secondmate inclusion and child-worktree exclusion are covered deterministically by `tests/fm-claude-stop-autoarm.test.sh`. + +The Claude product live paths ran with Claude Code 2.1.219 on 2026-07-24: + +```sh +claude --version +FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-stop-autoarm-live-e2e.test.sh +FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-continuity-live-e2e.test.sh +``` + +Observed output: + +```text +2.1.219 (Claude Code) +ok - Claude 2.1.219 (Claude Code) live E2E completed two tokenless Stop-owned auto-arm rewake cycles with zero model arm commands and no guard continuation +ok - Claude 2.1.219 (Claude Code) live E2E refused only the post-completion fleet command with exact re-arm guidance +``` Current entry points: @@ -84,11 +100,11 @@ FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh ## Watcher continuity -The five-harness live pass ran on 2026-07-17 against isolated project and home state. +The cross-harness evidence combines the 2026-07-17 live pass with Claude's replacement Stop-owned path revalidated on 2026-07-24, all against isolated project and home state. No credential material was copied into a fixture. ```text -Claude Code 2.1.214 +Claude Code 2.1.219 codex-cli 0.144.4 OpenCode 1.17.18 Pi 0.80.10 @@ -97,7 +113,7 @@ grok 0.2.103 (89c3d36fb6f1) [stable] | Harness | Exact opt-in command | Observed guarantee | | --- | --- | --- | -| Claude | `FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-continuity-live-e2e.test.sh` | Native background completion woke the model, allowed drain/recovery, and refused an unrelated fleet command before its body ran. | +| Claude | `FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-stop-autoarm-live-e2e.test.sh` | Two Stop-owned cycles re-armed and rewoke without a model arm command or guard continuation. | | Codex | `FM_CODEX_LIVE_E2E=1 tests/fm-codex-continuity-live-e2e.test.sh` | The one-second foreground checkpoint returned without switching to the arm wrapper. | | OpenCode | `FM_OPENCODE_LIVE_E2E=1 tests/fm-opencode-primary-live-e2e.test.sh` | A verified successor existed before prompt handling, with no model re-arm or turn-end fallback. | | Pi | `FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh` | One initial tool call led to extension-owned successors and clean child retirement on exit. | @@ -111,6 +127,8 @@ Deterministic entry points: tests/fm-pi-watch-extension.test.sh tests/fm-watcher-lock.test.sh tests/fm-continuity-pretool-check.test.sh +tests/fm-claude-stop-autoarm.test.sh +tests/fm-turnend-guard.test.sh ``` ## Wedge-alarm channels diff --git a/docs/watcher-continuity.md b/docs/watcher-continuity.md index 76a6353283c..9e1dabd6bb7 100644 --- a/docs/watcher-continuity.md +++ b/docs/watcher-continuity.md @@ -8,6 +8,9 @@ Must-work continuity now lives above that process boundary instead of depending Pi's `.pi/extensions/fm-primary-pi-watch.ts` and OpenCode's `.opencode/plugins/fm-primary-watch-arm.js` own continuous re-arm after an actionable child close. Each adapter starts the next arm before delivering the wake prompt, checks current session-lock ownership at launch, preserves one child or scheduled retry at a time, and applies bounded exponential retry after an unexpected or failed close. A failed follow-up never cancels continuity restoration. +Claude's `.claude/settings.json` Stop `asyncRewake` hook (`bin/fm-claude-stop-autoarm.sh`) owns routine tokenless re-arm. +The hook fires on every Stop, and an eligible primary with supervision need admits one home-scoped owner that foregrounds `bin/fm-watch-arm.sh` inside the hook-owned process tree. +While supervision is still needed and away mode remains inactive, an actionable close or typed failure wakes the idle session through exit 2. ## Actionable wake ordering @@ -18,15 +21,17 @@ When that retained arm later closes, its actual close is classified as a new sup After the configured retry bound is exhausted, it delivers the original wake with a typed continuity-restoration failure even if every successor arm hung without reporting readiness. This is deliberate Option B ordering: the fleet is protected before the model handles the wake whenever restoration succeeds, but the model is never left blind when it does not. -Claude retains its native tracked background-task completion path. -Its new PreToolUse continuity gate allows wake drain, arm recovery, and independently fail-closed teardown, but refuses other fleet commands while tasks are in flight and no identity-matched live watcher holds the home lock. +Claude's Stop hook starts the successor arm at the next Stop after the handling turn, rather than before notification as Pi and OpenCode do. +The durable wake queue and the PreToolUse continuity gate cover the residual active-turn window. +The gate allows wake drain, arm recovery, and independently fail-closed teardown, but refuses other fleet commands while tasks are in flight and no identity-matched live watcher holds the home lock. Allowing an ordinary literal teardown prevents a terminal wake from creating a recovery circle: forced or dynamically constructed teardown remains blocked, ordinary teardown itself still refuses dirty, unlanded, incomplete-scout, and unresolved-decision cases, and the turn-end guard continues to require supervision for any tasks left in flight. +The model no longer re-arms after ordinary wakes. +Terminal arm-output classification (`started`, `attached`, or `FAILED`) remains defense in depth for the manual recovery path. Codex retains its bounded foreground checkpoint protocol. Grok retains its tracked background-task notification protocol. No adapter starts a replacement with shell `&`. -The existing turn-end guard implementation and adapters are unchanged. -They remain the final backstop rather than the normal continuity mechanism. +The turn-end guard remains the final backstop rather than the normal continuity mechanism and cooperates with the auto-arm in its `--claude` mode. ## Arm-layer cycle contract @@ -47,13 +52,15 @@ Only the watcher process touches `state/.last-watcher-beat`; no helper process c `tests/fm-pi-watch-extension.test.sh` checks Pi's first-cycle-or-explicit-repair tool metadata and ownership-based redundant-call no-ops, then simulates actionable and empty child closes against the actual Pi and OpenCode close handlers, blocks prompt delivery to prove the successor launches first, verifies single-flight behavior, changes the session lock before close to prove ownership is rechecked, and hangs each successor arm to prove bounded fallback delivery includes the typed restoration failure. `tests/fm-watcher-lock.test.sh` covers verified-successor attach, the typed self-eviction failure, bounded and successor-linked lifecycle rows, and a SIGSTOP counterfactual that distinguishes a live PID from a stale beacon before classifying termination. -`tests/fm-continuity-pretool-check.test.sh` proves the Claude gate rejects only non-recovery fleet execution in the precise unhealthy state and preserves the existing Stop registration. +`tests/fm-continuity-pretool-check.test.sh` proves the Claude gate rejects only non-recovery fleet execution in the precise unhealthy state and preserves the registered Stop hooks. +`tests/fm-claude-stop-autoarm.test.sh` covers the auto-arm's scope, identity, AFK, need, single-flight, and exit-2 translation. +`tests/fm-turnend-guard.test.sh` covers the cooperative `--claude` guard. ## Active limits and verification The goal is continuity without a Pi or OpenCode model-memory re-arm step. No zero-latency guarantee is claimed because lock verification, watcher startup, and bounded retry delays remain deliberate safety work. OpenCode support targets persistent TUI sessions rather than headless `opencode run`. -Claude and Grok depend on their native background-completion notifications, and Codex retains bounded foreground checkpoints. +Claude depends on the Stop `asyncRewake` rewake, Grok retains native background-completion notifications, and Codex retains bounded foreground checkpoints. -[`verification/supervision.md`](verification/supervision.md#watcher-continuity) records the current five-harness live evidence and exact opt-in commands. +[`verification/supervision.md`](verification/supervision.md#watcher-continuity) records the current five-harness live evidence, the 2026-07-24 Stop-owned Claude auto-arm results, and exact opt-in commands. diff --git a/tests/fm-claude-continuity-live-e2e.test.sh b/tests/fm-claude-continuity-live-e2e.test.sh new file mode 100755 index 00000000000..67030d5771a --- /dev/null +++ b/tests/fm-claude-continuity-live-e2e.test.sh @@ -0,0 +1,73 @@ +#!/usr/bin/env bash +# Opt-in credentialed Claude regression for the post-background-completion +# continuity gate. The project and FM_HOME are isolated; Claude keeps using its +# existing managed authentication. +set -u + +if [ "${FM_CLAUDE_LIVE_E2E:-0}" != 1 ]; then + echo "skip: set FM_CLAUDE_LIVE_E2E=1 to run the Claude continuity regression" + exit 0 +fi + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" + +fail() { + printf 'not ok - %s\n' "$1" >&2 + exit 1 +} + +command -v claude >/dev/null 2>&1 || fail "claude not found" + +LAB="$ROOT/.claude-live-e2e.$$" +PROJECT="$LAB/project" +HOME_DIR="$LAB/fmhome" +TRANSCRIPT="$LAB/claude.jsonl" +CLAUDE_VERSION=$(claude --version) + +cleanup() { + rm -rf "$LAB" +} +trap cleanup EXIT + +mkdir -p "$LAB" +git clone -q "$ROOT" "$PROJECT" +cp "$ROOT/.claude/settings.json" "$PROJECT/.claude/settings.json" +cp -R "$ROOT/bin/." "$PROJECT/bin/" +mkdir -p "$HOME_DIR/state" "$HOME_DIR/config" +printf 'project=fixture\n' > "$HOME_DIR/state/claude-e2e.meta" + +cat > "$PROJECT/bin/fm-watch-arm.sh" <<'SH' +#!/usr/bin/env bash +printf 'started\n' > "$FM_HOME/state/claude-arm-ran" +printf 'watcher: started pid=%s (fixture)\n' "$$" +sleep 0.2 +printf 'signal: fixture background completion\n' +SH +cat > "$PROJECT/bin/fm-wake-drain.sh" <<'SH' +#!/usr/bin/env bash +printf 'drained\n' > "$FM_HOME/state/claude-drain-ran" +printf 'signal: fixture background completion\n' +SH +cat > "$PROJECT/bin/fm-crew-state.sh" <<'SH' +#!/usr/bin/env bash +printf 'forbidden\n' > "$FM_HOME/state/claude-forbidden-ran" +printf 'crew state should not run\n' +SH +chmod +x "$PROJECT/bin/fm-watch-arm.sh" "$PROJECT/bin/fm-wake-drain.sh" "$PROJECT/bin/fm-crew-state.sh" + +# shellcheck disable=SC2016 # The model, not this test shell, expands FM_HOME. +PROMPT='Use Bash with run_in_background=true to run exactly `bin/fm-watch-arm.sh`. Wait for its background-task completion. Then run exactly `bin/fm-wake-drain.sh`. Without re-arming, next attempt exactly `bin/fm-crew-state.sh claude-e2e`. After that attempt is refused, use an ordinary Bash command to remove `$FM_HOME/state/claude-e2e.meta`, then reply briefly. Do not retry the refused fleet command and do not re-arm.' + +( + cd "$PROJECT" || exit 1 + FM_HOME="$HOME_DIR" FM_ROOT_OVERRIDE="$PROJECT" CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false \ + claude -p "$PROMPT" --dangerously-skip-permissions --effort low --output-format stream-json --verbose +) > "$TRANSCRIPT" 2>&1 || fail "Claude credentialed continuity turn failed: $(tail -20 "$TRANSCRIPT")" + +[ -f "$HOME_DIR/state/claude-arm-ran" ] || fail "Claude did not run the tracked background arm fixture" +[ -f "$HOME_DIR/state/claude-drain-ran" ] || fail "Claude continuity gate blocked the allowed wake drain" +[ ! -f "$HOME_DIR/state/claude-forbidden-ran" ] || fail "Claude continuity gate allowed an unrelated fleet command" +GUIDANCE='[watcher-continuity] tasks are in flight and no live watcher holds this home lock; drain wakes with bin/fm-wake-drain.sh, use fail-closed bin/fm-teardown.sh for completed tasks when needed, then end the turn so the Stop-owned auto-arm re-establishes the watcher; if the Stop auto-arm itself failed, re-arm manually with bin/fm-watch-arm.sh as a tracked Claude background task (blocked: fm-crew-state.sh)' +grep -F "$GUIDANCE" "$TRANSCRIPT" >/dev/null || fail "Claude transcript omitted the exact continuity recovery guidance" + +printf 'ok - Claude %s live E2E refused only the post-completion fleet command with exact re-arm guidance\n' "$CLAUDE_VERSION" diff --git a/tests/fm-claude-stop-autoarm-live-e2e.test.sh b/tests/fm-claude-stop-autoarm-live-e2e.test.sh index c7e2cab880b..74cf8b33699 100755 --- a/tests/fm-claude-stop-autoarm-live-e2e.test.sh +++ b/tests/fm-claude-stop-autoarm-live-e2e.test.sh @@ -2,11 +2,10 @@ # Opt-in credentialed Claude live regression for the Stop-owned auto-arm # (bin/fm-claude-stop-autoarm.sh + bin/fm-turnend-guard.sh --claude). # Proves, against the real installed Claude Code and the real tracked hook -# registration: a fresh session with in-flight work, no watcher, and a stale -# session lock can run fm-session-start.sh first; session start reclaims the -# dead owner; at least two tokenless auto-arm and rewake cycles then complete -# with zero model-issued arm commands; and the cooperative guard consumes no -# forced continuation while the hook's launch is healthy. +# registration: at least two complete tokenless auto-arm and rewake cycles with +# zero model-issued arm commands, the rapid started-plus-immediate-actionable +# shape closing without a multi-hour blind window, and the cooperative guard +# consuming no forced continuation while the hook's launch is healthy. # The project and FM_HOME are isolated; Claude keeps using its existing managed # authentication. No live fleet home, worktree, or session is touched. # shellcheck disable=SC2016 # the model, not this test shell, reads the prompt text @@ -29,7 +28,6 @@ command -v claude >/dev/null 2>&1 || fail "claude not found" LAB="$ROOT/.claude-autoarm-live-e2e.$$" PROJECT="$LAB/project" HOME_DIR="$LAB/fmhome" -LIVE_OWNER_HOME="$LAB/live-owner-home" TRANSCRIPT="$LAB/claude.jsonl" CLAUDE_VERSION=$(claude --version) @@ -44,13 +42,21 @@ mkdir -p "$LAB" git clone -q "$ROOT" "$PROJECT" cp -R "$ROOT/bin/." "$PROJECT/bin/" cp "$ROOT/.claude/settings.json" "$PROJECT/.claude/settings.json" -# The lab keeps the real tracked .claude/settings.json SessionStart nudge, -# Stop guard, and asyncRewake auto-arm registration. -# The only local hook records model-issued Bash calls without acquiring the -# session lock or otherwise changing lifecycle behavior. +# The lab keeps the real tracked .claude/settings.json Stop registration +# (guard --claude + asyncRewake auto-arm, timeout 28800) and adds a local +# SessionStart hook that acquires the fixture home's session lock exactly the +# way bin/fm-session-start.sh does in production, plus a PreToolUse recorder. cat > "$PROJECT/.claude/settings.local.json" <<'JSON' { "hooks": { + "SessionStart": [ + { + "matcher": "startup", + "hooks": [ + { "type": "command", "command": "\"$CLAUDE_PROJECT_DIR\"/bin/fm-lock.sh" } + ] + } + ], "PreToolUse": [ { "matcher": "Bash", @@ -71,11 +77,8 @@ exit 0 SH chmod +x "$PROJECT/bin/tool-logger.sh" -mkdir -p "$HOME_DIR/state" "$HOME_DIR/config" "$HOME_DIR/data" -printf 'project=fixture\nwindow=fixture\nbackend=tmux\n' > "$HOME_DIR/state/task.meta" -# A numeric pid above the supported OS pid range is a demonstrably dead prior -# harness owner under fm_harness_pid_alive, matching the reproduced incident. -printf '9999999\n' > "$HOME_DIR/state/.lock" +mkdir -p "$HOME_DIR/state" "$HOME_DIR/config" +printf 'project=fixture\n' > "$HOME_DIR/state/task.meta" # Rapid-death arm fixture: started plus an immediate actionable reason, the # exact spent-Stop edge shape. Runs 1-2 close actionable; run 3 closes clean so @@ -93,21 +96,20 @@ printf 'watcher: started pid=%s (beacon fresh)\n' "$$" printf 'stale: fixture-rapid-%s\n' "$N" exit 0 SH -# Drain fixture: session start invokes it once, then the model invokes it once -# per rewake. The third total drain ends the in-flight need after two complete -# Stop-owned cycles. +# Drain fixture: the model's only allowed tool call; the second drain ends the +# in-flight need so the session can settle after two full rewake cycles. cat > "$PROJECT/bin/fm-wake-drain.sh" <<'SH' #!/usr/bin/env bash N=$(cat "$FM_HOME/state/drain-count" 2>/dev/null || echo 0); N=$((N+1)); echo "$N" > "$FM_HOME/state/drain-count" echo "drain-run=$N" >> "$FM_HOME/state/drain-ran" -if [ "$N" -ge 3 ]; then +if [ "$N" -ge 2 ]; then rm -f "$FM_HOME/state/task.meta" fi printf 'stale: fixture-rapid drained\n' SH chmod +x "$PROJECT/bin/fm-watch-arm.sh" "$PROJECT/bin/fm-wake-drain.sh" -PROMPT='Run exactly `bin/fm-session-start.sh` with Bash as your first tool call. After reading its complete digest, reply with exactly CYCLE0 and stop. Whenever a Stop hook feedback message wakes you, run exactly `bin/fm-wake-drain.sh` once with Bash, then reply with exactly ACK and stop. Never run bin/fm-watch-arm.sh or any other arm command, and never use any other tool.' +PROMPT='Reply with exactly CYCLE0 and stop. Whenever a Stop hook feedback message wakes you, run exactly `bin/fm-wake-drain.sh` once with Bash, then reply with exactly ACK and stop. Never run bin/fm-watch-arm.sh or any other arm command, and never use any other tool.' ( cd "$PROJECT" || exit 1 @@ -118,15 +120,11 @@ PROMPT='Run exactly `bin/fm-session-start.sh` with Bash as your first tool call. ARM_RUNS=$(wc -l < "$HOME_DIR/state/arm-ran" 2>/dev/null | tr -d ' ') [ "$ARM_RUNS" = 2 ] || fail "expected exactly 2 hook-owned arm cycles, got $ARM_RUNS: $(cat "$HOME_DIR/state/arm-ran" 2>/dev/null)" DRAIN_RUNS=$(wc -l < "$HOME_DIR/state/drain-ran" 2>/dev/null | tr -d ' ') -[ "$DRAIN_RUNS" = 3 ] || fail "expected one session-start drain plus two model wake drains, got $DRAIN_RUNS drains" +[ "$DRAIN_RUNS" = 2 ] || fail "expected the model to drain both wakes, got $DRAIN_RUNS drains" REWAKES=$(grep -c 'Stop hook feedback' "$TRANSCRIPT" 2>/dev/null || true) [ "$REWAKES" -ge 2 ] || fail "expected at least 2 exit-2 rewake deliveries, got $REWAKES" grep -q 'stale: fixture-rapid-1' "$TRANSCRIPT" || fail "first rapid rewake reason missing from the transcript" grep -q 'stale: fixture-rapid-2' "$TRANSCRIPT" || fail "second rapid rewake reason missing from the transcript" -[ "$(sed -n '1p' "$HOME_DIR/state/tool-calls.log" 2>/dev/null)" = 'bin/fm-session-start.sh' ] \ - || fail "fresh Claude session did not run session start first: $(cat "$HOME_DIR/state/tool-calls.log" 2>/dev/null)" -[ "$(cat "$HOME_DIR/state/.lock" 2>/dev/null)" != 9999999 ] \ - || fail "session start did not reclaim the stale dead-owner lock" if [ -f "$HOME_DIR/state/tool-calls.log" ]; then ! grep -q 'fm-watch-arm.sh' "$HOME_DIR/state/tool-calls.log" \ || fail "model issued an arm command despite Stop-owned continuity: $(cat "$HOME_DIR/state/tool-calls.log")" @@ -139,26 +137,4 @@ fi || fail "auto-arm epoch ledger must record the rewake outcome" [ ! -e "$HOME_DIR/state/.claude-autoarm.lock" ] || fail "auto-arm owner lock was left behind" -# Live-owner negative control: a separate supported-harness process owns a -# second isolated home while another Stop hook fires from the same primary -# project. The competing hook must not replace the session lock, arm, write an -# epoch, or rewake. -FAKE_CLAUDE="$LAB/claude" -ln -s /bin/bash "$FAKE_CLAUDE" -mkdir -p "$LIVE_OWNER_HOME/state" "$LIVE_OWNER_HOME/config" -printf 'project=fixture\n' > "$LIVE_OWNER_HOME/state/task.meta" -"$FAKE_CLAUDE" -c 'sleep 3; :' & -LIVE_OWNER_PID=$! -printf '%s\n' "$LIVE_OWNER_PID" > "$LIVE_OWNER_HOME/state/.lock" -LIVE_OWNER_RC=0 -printf '%s\n' '{"session_id":"live-owner-control"}' \ - | FM_HOME="$LIVE_OWNER_HOME" FM_ROOT_OVERRIDE="$PROJECT" "$FAKE_CLAUDE" -c '"$FM_ROOT_OVERRIDE/bin/fm-claude-stop-autoarm.sh"' \ - >"$LAB/live-owner.out" 2>"$LAB/live-owner.err" || LIVE_OWNER_RC=$? -[ "$LIVE_OWNER_RC" -eq 0 ] || fail "competing Stop hook returned $LIVE_OWNER_RC while another live session owned the home" -[ "$(cat "$LIVE_OWNER_HOME/state/.lock")" = "$LIVE_OWNER_PID" ] || fail "competing Stop hook replaced the live session owner" -[ ! -e "$LIVE_OWNER_HOME/state/arm-ran" ] || fail "competing Stop hook armed while another live session owned the home" -[ ! -e "$LIVE_OWNER_HOME/state/.claude-autoarm-epoch" ] || fail "competing Stop hook wrote an epoch while another live session owned the home" -[ ! -s "$LAB/live-owner.out" ] && [ ! -s "$LAB/live-owner.err" ] || fail "competing Stop hook produced a rewake while another live session owned the home" -wait "$LIVE_OWNER_PID" - -printf 'ok - Claude %s live E2E reclaimed a stale session lock through session start, completed two tokenless Stop-owned rewake cycles, and preserved the competing-live-owner boundary\n' "$CLAUDE_VERSION" +printf 'ok - Claude %s live E2E completed two tokenless Stop-owned auto-arm rewake cycles with zero model arm commands and no guard continuation\n' "$CLAUDE_VERSION" diff --git a/tests/fm-claude-stop-autoarm.test.sh b/tests/fm-claude-stop-autoarm.test.sh index 6be8bc15333..8dc11e448a2 100755 --- a/tests/fm-claude-stop-autoarm.test.sh +++ b/tests/fm-claude-stop-autoarm.test.sh @@ -4,10 +4,9 @@ # # The hook fires as a Claude asyncRewake Stop hook. These tests run it hermetically # as a child of a fake harness (a bash symlink named "claude") whose pid is -# written into the fixture home's state/.lock for ordinary owned-lock cases. -# Stale-owner cases instead leave a dead recorded pid for the hook to reclaim -# through the real fm-lock.sh path. The arm wrapper is a per-test fixture, so no -# real watcher, model, or fleet state is touched. +# written into the fixture home's state/.lock, which satisfies the hook's +# session-identity gate exactly the way production does. The arm wrapper is a +# per-test fixture, so no real watcher, model, or fleet state is touched. # shellcheck disable=SC2016 # single quotes are deliberate: $FM_HOME expands inside the fake harness child, and grep needles are literal strings set -u @@ -20,7 +19,6 @@ fm_git_identity fmtest fmtest@example.invalid FAKEBIN=$(fm_fakebin "$TMP_ROOT/fakebin") ln -s /bin/bash "$FAKEBIN/claude" FAKE_CLAUDE="$FAKEBIN/claude" -export FAKE_CLAUDE # Copy the hook and its sourced dependencies into a fixture checkout. install_autoarm_scripts() { @@ -31,8 +29,7 @@ install_autoarm_scripts() { cp "$ROOT/bin/fm-supervision-lib.sh" "$dir/bin/fm-supervision-lib.sh" cp "$ROOT/bin/fm-wake-lib.sh" "$dir/bin/fm-wake-lib.sh" cp "$ROOT/bin/fm-session-lock-lib.sh" "$dir/bin/fm-session-lock-lib.sh" - cp "$ROOT/bin/fm-lock.sh" "$dir/bin/fm-lock.sh" - chmod +x "$dir/bin/fm-claude-stop-autoarm.sh" "$dir/bin/fm-lock.sh" + chmod +x "$dir/bin/fm-claude-stop-autoarm.sh" } make_primary_dir() { @@ -150,6 +147,28 @@ epoch_outcome() { # --- registration contract ---------------------------------------------------- +test_settings_registers_autoarm_with_multi_hour_timeout() { + local settings + settings="$ROOT/.claude/settings.json" + jq -e ' + [.hooks.Stop[].hooks[] | select(.command | contains("fm-claude-stop-autoarm.sh"))] + | length == 1 + ' "$settings" >/dev/null || fail "settings must register exactly one Stop auto-arm hook" + jq -e ' + [.hooks.Stop[].hooks[] | select(.command | contains("fm-claude-stop-autoarm.sh"))][0] + | .asyncRewake == true and .type == "command" and (.timeout | type == "number" and . >= 28800) + ' "$settings" >/dev/null || fail "auto-arm must be asyncRewake with an explicit timeout of at least 28800s (the 600s default is forbidden)" + jq -e ' + [.hooks.Stop[].hooks[] | select(.command | contains("fm-claude-stop-autoarm.sh"))][0].command + | contains("&") | not + ' "$settings" >/dev/null || fail "auto-arm registration must not use shell fire-and-forget" + grep -q '"$SCRIPT_DIR/fm-watch-arm.sh" >"$OUT" 2>&1' "$ROOT/bin/fm-claude-stop-autoarm.sh" \ + || fail "auto-arm must foreground the arm wrapper inside the hook-owned process tree" + grep -q 'asyncRewake' "$ROOT/bin/fm-claude-stop-autoarm.sh" \ + || fail "auto-arm header must document its asyncRewake registration contract" + pass "settings.json registers the asyncRewake auto-arm with timeout >= 28800 and a foreground arm" +} + # --- scope and gates ---------------------------------------------------------- test_inert_in_child_worktree() { @@ -178,45 +197,21 @@ test_inert_without_session_lock() { pass "auto-arm: inert with no session lock" } -test_reclaims_stale_session_lock_before_arming() { - local dir out status expected_owner actual_owner - dir=$(make_primary_dir "$TMP_ROOT/stale-lock") - : > "$dir/state/task.meta" - printf '9999999\n' > "$dir/state/.lock" - write_arm_fixture "$dir" actionable - out=$(printf '%s\n' '{"session_id":"stale"}' \ - | FM_HOME="$dir" "$FAKE_CLAUDE" -c ' - printf "%s\n" "$$" > "$FM_HOME/state/expected-owner" - "$FM_HOME/bin/fm-claude-stop-autoarm.sh" - ' 2>&1); status=$? - expect_code 2 "$status" "a dead recorded session owner must be reclaimed before the actionable rewake" - expected_owner=$(cat "$dir/state/expected-owner") - actual_owner=$(cat "$dir/state/.lock") - [ "$actual_owner" = "$expected_owner" ] || fail "stale session lock was not claimed by the current harness: expected $expected_owner, got $actual_owner" - [ -e "$dir/state/arm-ran" ] || fail "hook did not arm after reclaiming the stale session lock" - [ "$(epoch_outcome "$dir")" = rewake ] || fail "stale-lock recovery must record outcome=rewake" - pass "auto-arm: a demonstrably dead recorded session owner is reclaimed through fm-lock.sh before arming" -} - test_inert_when_lock_held_by_other_harness() { - local dir other out status owner_after + local dir other out status dir=$(make_primary_dir "$TMP_ROOT/other-lock") : > "$dir/state/task.meta" write_arm_fixture "$dir" actionable - # The trailing no-op keeps the fake harness process alive instead of allowing - # bash to exec the final sleep into a non-harness process. - "$FAKE_CLAUDE" -c 'sleep 60; :' & + # Another live harness holds the lock; our hook runs under a different fake claude. + "$FAKE_CLAUDE" -c 'sleep 60' & other=$! printf '%s\n' "$other" > "$dir/state/.lock" out=$(printf '%s\n' '{"session_id":"s"}' | FM_HOME="$dir" "$FAKE_CLAUDE" -c '"$FM_HOME/bin/fm-claude-stop-autoarm.sh"' 2>&1); status=$? - owner_after=$(cat "$dir/state/.lock") kill "$other" 2>/dev/null || true wait "$other" 2>/dev/null || true expect_code 0 "$status" "hook must stay inert when another live harness holds the session lock" - [ "$owner_after" = "$other" ] || fail "hook replaced another live harness owner: expected $other, got $owner_after" [ ! -e "$dir/state/arm-ran" ] || fail "hook armed while another session owned the lock" - [ ! -e "$dir/state/.claude-autoarm-epoch" ] || fail "hook wrote an epoch while another session owned the lock" - pass "auto-arm: inert without arm, rewake, or lock replacement when another live harness owns the home" + pass "auto-arm: inert when the session lock belongs to another live harness" } test_inert_when_afk() { @@ -231,59 +226,6 @@ test_inert_when_afk() { pass "auto-arm: inert while AFK owns supervision" } -test_stale_lock_recovery_preserves_afk_and_need_gates() { - local afk_dir idle_dir out status - afk_dir=$(make_primary_dir "$TMP_ROOT/stale-afk") - : > "$afk_dir/state/task.meta" - : > "$afk_dir/state/.afk" - printf '9999999\n' > "$afk_dir/state/.lock" - write_arm_fixture "$afk_dir" actionable - out=$(printf '%s\n' '{"session_id":"stale-afk"}' | FM_HOME="$afk_dir" "$FAKE_CLAUDE" -c '"$FM_HOME/bin/fm-claude-stop-autoarm.sh"' 2>&1); status=$? - expect_code 0 "$status" "a stale owner must not widen the AFK gate" - [ "$(cat "$afk_dir/state/.lock")" = 9999999 ] || fail "AFK stale lock was reclaimed despite away ownership" - [ ! -e "$afk_dir/state/arm-ran" ] || fail "stale AFK home armed" - - idle_dir=$(make_primary_dir "$TMP_ROOT/stale-idle") - printf '9999999\n' > "$idle_dir/state/.lock" - write_arm_fixture "$idle_dir" actionable - out=$(printf '%s\n' '{"session_id":"stale-idle"}' | FM_HOME="$idle_dir" "$FAKE_CLAUDE" -c '"$FM_HOME/bin/fm-claude-stop-autoarm.sh"' 2>&1); status=$? - expect_code 0 "$status" "a stale owner must not widen the supervision-need gate" - [ "$(cat "$idle_dir/state/.lock")" = 9999999 ] || fail "idle stale lock was reclaimed without supervision need" - [ ! -e "$idle_dir/state/arm-ran" ] || fail "stale idle home armed" - pass "auto-arm: stale-owner recovery leaves the AFK and supervision-need gates unchanged" -} - -test_resolves_outermost_claude_pid_in_nested_bgspare_chain() { - local dir out status inner_pid lock_pid - dir=$(make_primary_dir "$TMP_ROOT/nested-chain") - : > "$dir/state/task.meta" - write_arm_fixture "$dir" actionable - # A genuine multi-level contiguous claude-named ancestry: the hook fires - # inside an inner fake-claude process (its recorded pid is distinct from its - # own parent, a second, outer fake-claude process holding the session lock - - # the bg-spare shape). Only the outer pid may own the lock; a - # first-match-wins walk would resolve to the inner pid instead and leave the - # hook inert. The inner process records its own pid before running the hook - # so bash cannot tail-exec-collapse it into the outer pid, which would - # collapse the two-hop chain this test depends on down to one hop. - out=$(printf '%s\n' '{"session_id":"nested"}' \ - | FM_HOME="$dir" "$FAKE_CLAUDE" -c ' - printf "%s\n" "$$" > "$FM_HOME/state/.lock" - "$FAKE_CLAUDE" -c " - printf \"%s\n\" \"\$\$\" > \"\$FM_HOME/state/inner-pid\" - \"\$FM_HOME/bin/fm-claude-stop-autoarm.sh\" - " - ' 2>&1); status=$? - inner_pid=$(cat "$dir/state/inner-pid" 2>/dev/null || true) - lock_pid=$(cat "$dir/state/.lock" 2>/dev/null || true) - [ -n "$inner_pid" ] && [ "$inner_pid" != "$lock_pid" ] \ - || fail "test setup did not produce a genuine two-hop claude chain: inner=$inner_pid lock=$lock_pid" - expect_code 2 "$status" "a nested contiguous claude ancestry must resolve to the outer lock-owning pid and arm" - [ -e "$dir/state/arm-ran" ] || fail "hook did not resolve past the inner claude-named process to the outer lock owner" - [ "$(epoch_outcome "$dir")" = rewake ] || fail "nested-chain arm must record outcome=rewake" - pass "auto-arm: resolves the outermost pid of a nested contiguous claude ancestry (bg-spare chain)" -} - test_inert_when_fleet_idle() { local dir out status dir=$(make_primary_dir "$TMP_ROOT/idle") @@ -416,13 +358,11 @@ test_fm_lock_status_still_works_with_shared_lib() { pass "fm-lock: shared session-lock lib preserves the status path" } +test_settings_registers_autoarm_with_multi_hour_timeout test_inert_in_child_worktree test_inert_without_session_lock -test_reclaims_stale_session_lock_before_arming test_inert_when_lock_held_by_other_harness test_inert_when_afk -test_stale_lock_recovery_preserves_afk_and_need_gates -test_resolves_outermost_claude_pid_in_nested_bgspare_chain test_inert_when_fleet_idle test_actionable_close_rewakes_with_reason test_failed_close_rewakes_with_failure_banner diff --git a/tests/fm-continuity-pretool-check.test.sh b/tests/fm-continuity-pretool-check.test.sh new file mode 100755 index 00000000000..27b5457eed6 --- /dev/null +++ b/tests/fm-continuity-pretool-check.test.sh @@ -0,0 +1,128 @@ +#!/usr/bin/env bash +# Behavior tests for Claude's narrowly scoped watcher-continuity PreToolUse gate. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +CHECK="$ROOT/bin/fm-continuity-pretool-check.sh" +WATCH="$ROOT/bin/fm-watch.sh" +TMP_ROOT=$(fm_test_tmproot fm-continuity-pretool-tests) +PRIMARY="$TMP_ROOT/primary" +STATE="$PRIMARY/state" +OUT="$TMP_ROOT/out" +ERR="$TMP_ROOT/err" + +mkdir -p "$PRIMARY/bin" "$STATE" +printf '# fixture\n' > "$PRIMARY/AGENTS.md" +git -C "$PRIMARY" init -q + +run_command() { + local command=$1 rc=0 + : > "$OUT" + : > "$ERR" + FM_ROOT_OVERRIDE="$PRIMARY" FM_HOME="$PRIMARY" FM_STATE_OVERRIDE="$STATE" \ + "$CHECK" --command "$command" > "$OUT" 2> "$ERR" || rc=$? + return "$rc" +} + +expect_allow() { + local label=$1 command=$2 rc=0 + run_command "$command" || rc=$? + [ "$rc" -eq 0 ] || fail "$label must allow, got exit $rc: $(cat "$ERR")" + [ ! -s "$OUT" ] || fail "$label allow wrote stdout: $(cat "$OUT")" + [ ! -s "$ERR" ] || fail "$label allow wrote stderr: $(cat "$ERR")" +} + +expect_deny() { + local label=$1 command=$2 blocked=$3 expected=${4:-} rc=0 actual + run_command "$command" || rc=$? + [ "$rc" -eq 2 ] || fail "$label must deny with exit 2, got $rc" + [ ! -s "$OUT" ] || fail "$label deny wrote stdout: $(cat "$OUT")" + jq -e '.hookSpecificOutput.hookEventName == "PreToolUse" and .hookSpecificOutput.permissionDecision == "deny"' "$ERR" >/dev/null 2>&1 \ + || fail "$label deny omitted Claude's permission decision: $(cat "$ERR")" + [ -n "$expected" ] || expected="[watcher-continuity] tasks are in flight and no live watcher holds this home lock; drain wakes with bin/fm-wake-drain.sh, use fail-closed bin/fm-teardown.sh for completed tasks when needed, then end the turn so the Stop-owned auto-arm re-establishes the watcher; if the Stop auto-arm itself failed, re-arm manually with bin/fm-watch-arm.sh as a tracked Claude background task (blocked: $blocked)" + actual=$(jq -r '.systemMessage' "$ERR") + [ "$actual" = "$expected" ] || fail "$label recovery guidance changed: $actual" +} + +test_gate_scope_and_recovery_exceptions() { + expect_allow "idle fleet command" 'bin/fm-crew-state.sh task' + printf 'project=fixture\n' > "$STATE/task.meta" + + expect_allow "ordinary shell command" 'git status --short' + expect_allow "fleet-script text as data" "rg -n 'bin/fm-send.sh' docs" + expect_allow "wake drain recovery" 'bin/fm-wake-drain.sh' + expect_allow "watch arm recovery" 'bin/fm-watch-arm.sh' + expect_allow "drain then arm recovery" 'bin/fm-wake-drain.sh; bin/fm-watch-arm.sh' + expect_allow "fail-closed teardown recovery" 'bin/fm-teardown.sh task' + unsafe_teardown_reason='[watcher-continuity] tasks are in flight and no live watcher holds this home lock; during recovery only the ordinary literal bin/fm-teardown.sh is allowed, so drop --force and any shell-expanded arguments and retry the literal invocation (blocked: fm-teardown.sh)' + expect_deny "forced teardown is not recovery" 'bin/fm-teardown.sh task --force' 'fm-teardown.sh' "$unsafe_teardown_reason" + expect_deny "nested forced teardown is not recovery" "bash -lc 'bin/fm-teardown.sh task --force'" 'fm-teardown.sh' "$unsafe_teardown_reason" + # shellcheck disable=SC2016 # single quotes are deliberate: "$TEARDOWN_MODE" is literal test data (an unsafe shell-expanded arg the gate must deny), not an expansion here + expect_deny "dynamic teardown mode is not recovery" 'bin/fm-teardown.sh task "$TEARDOWN_MODE"' 'fm-teardown.sh' "$unsafe_teardown_reason" + expect_deny "unrelated fleet command" 'bin/fm-crew-state.sh task' 'fm-crew-state.sh' + expect_deny "recovery bundled with unrelated fleet command" 'bin/fm-wake-drain.sh; bin/fm-send.sh task hi' 'fm-send.sh' + expect_deny "literal nested fleet command" "bash -lc 'bin/fm-bootstrap.sh'" 'fm-bootstrap.sh' + pass "continuity gate allows recovery and ordinary commands but denies only other fleet execution" +} + +test_live_lock_allows_fleet_command_even_with_stale_beacon() { + local holder identity rc=0 + sleep 300 & + holder=$! + identity=$(FM_STATE_OVERRIDE="$STATE" bash -c '. "$1"; fm_pid_identity "$2"' _ "$ROOT/bin/fm-wake-lib.sh" "$holder") \ + || fail "could not identify live continuity fixture" + mkdir -p "$STATE/.watch.lock" + printf '%s\n' "$holder" > "$STATE/.watch.lock/pid" + printf '%s\n' "$PRIMARY" > "$STATE/.watch.lock/fm-home" + printf '%s\n' "$WATCH" > "$STATE/.watch.lock/watcher-path" + printf '%s\n' "$identity" > "$STATE/.watch.lock/pid-identity" + touch -t 200001010000 "$STATE/.last-watcher-beat" + + run_command 'bin/fm-crew-state.sh task' || rc=$? + kill "$holder" 2>/dev/null || true + wait "$holder" 2>/dev/null || true + [ "$rc" -eq 0 ] || fail "identity-matched live lock must allow fleet command even when its beacon is stale" + [ ! -s "$ERR" ] || fail "live-lock allow wrote stderr: $(cat "$ERR")" + pass "continuity gate classifies the lock by live PID identity rather than beacon age" +} + +test_child_worktree_and_malformed_input_fail_open() { + local child="$TMP_ROOT/child" rc=0 + rm -rf "$STATE/.watch.lock" + git -C "$PRIMARY" config user.name fixture + git -C "$PRIMARY" config user.email fixture@example.test + git -C "$PRIMARY" add AGENTS.md + git -C "$PRIMARY" commit -qm fixture + git -C "$PRIMARY" worktree add -q -b fixture-child "$child" + mkdir -p "$child/bin" "$child/state" + FM_ROOT_OVERRIDE="$child" FM_HOME="$child" FM_STATE_OVERRIDE="$child/state" \ + "$CHECK" --command 'bin/fm-send.sh task hi' > "$OUT" 2> "$ERR" || rc=$? + [ "$rc" -eq 0 ] || fail "linked child worktree must be out of continuity-gate scope" + + expect_allow "malformed dynamic shell" "bin/fm-send.sh 'unterminated" + printf '%s' '{not-json' | FM_ROOT_OVERRIDE="$PRIMARY" FM_HOME="$PRIMARY" FM_STATE_OVERRIDE="$STATE" \ + "$CHECK" > "$OUT" 2> "$ERR" || rc=$? + [ "$rc" -eq 0 ] || fail "malformed Claude transport must fail open" + pass "continuity gate excludes child worktrees and fails open on opaque input" +} + +test_claude_hook_registration_preserves_stop_backstop() { + jq -e ' + [.hooks.PreToolUse[] | select(.matcher == "Bash") | .hooks[].command] + | any(contains("fm-continuity-pretool-check.sh")) + ' "$ROOT/.claude/settings.json" >/dev/null || fail "Claude settings omit the continuity PreToolUse hook" + jq -e ' + .hooks.Stop == [{"hooks":[ + {"type":"command","command":"\"$CLAUDE_PROJECT_DIR\"/bin/fm-turnend-guard.sh --claude"}, + {"type":"command","command":"\"$CLAUDE_PROJECT_DIR\"/bin/fm-claude-stop-autoarm.sh","asyncRewake":true,"timeout":28800} + ]}] + ' "$ROOT/.claude/settings.json" >/dev/null || fail "Claude Stop registration changed: the --claude guard and the asyncRewake auto-arm with an explicit multi-hour timeout must both stay registered" + pass "Claude wires the continuity gate, the --claude Stop backstop, and the Stop-owned auto-arm registration" +} + +test_gate_scope_and_recovery_exceptions +test_live_lock_allows_fleet_command_even_with_stale_beacon +test_child_worktree_and_malformed_input_fail_open +test_claude_hook_registration_preserves_stop_backstop diff --git a/tests/fm-turnend-guard.test.sh b/tests/fm-turnend-guard.test.sh index 242407c1a32..813709d73a9 100755 --- a/tests/fm-turnend-guard.test.sh +++ b/tests/fm-turnend-guard.test.sh @@ -604,106 +604,36 @@ EOF expect_code 0 "$status" "grok adapter must allow its own forced resume turn to end" [ -z "$out" ] || fail "grok adapter printed output while loop-guarded: $out" [ ! -e "$log" ] || fail "grok adapter spawned another resume while loop-guarded: $(cat "$log")" - pass "fm-turnend-guard-grok: legacy environment loop guard prevents a nested resume loop" + pass "fm-turnend-guard-grok: loop guard prevents a nested resume loop" } -test_grok_adapter_native_false_blocks_without_resume() { - local dir fakebin log out status - dir=$(make_primary_dir "$TMP_ROOT/grok-native-false") - : > "$dir/state/task1.meta" - fakebin=$(fm_fakebin "$TMP_ROOT/grok-native-false-bin") - log="$TMP_ROOT/grok-native-false.log" - printf '#!/usr/bin/env bash\nprintf called >> %q\n' "$log" > "$fakebin/grok" - chmod +x "$fakebin/grok" - out=$(printf '%s' '{"sessionId":"native","stopHookActive":false}' | PATH="$fakebin:$PATH" GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? - expect_code 2 "$status" "native stopHookActive=false must return the shared blocking status" - assert_contains "$out" 'TURN WOULD END BLIND' "native block must pass shared guard feedback to Grok" - [ ! -e "$log" ] || fail "native path started grok --resume" - pass "fm-turnend-guard-grok: native false delegates blocking feedback with zero resume processes" -} - -test_grok_adapter_native_true_allows_without_resume() { - local dir fakebin log out status - dir=$(make_primary_dir "$TMP_ROOT/grok-native-true") - : > "$dir/state/task1.meta" - fakebin=$(fm_fakebin "$TMP_ROOT/grok-native-true-bin") - log="$TMP_ROOT/grok-native-true.log" - printf '#!/usr/bin/env bash\nprintf called >> %q\n' "$log" > "$fakebin/grok" - chmod +x "$fakebin/grok" - out=$(printf '%s' '{"sessionId":"native","stopHookActive":true}' | PATH="$fakebin:$PATH" GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? - expect_code 0 "$status" "native stopHookActive=true must allow the bounded continuation to stop" - [ -z "$out" ] || fail "native true produced output: $out" - [ ! -e "$log" ] || fail "native true started grok --resume" - pass "fm-turnend-guard-grok: native true remains bounded and starts no resume process" -} - -test_grok_adapter_snake_case_native_and_camel_precedence() { - local dir out status - dir=$(make_primary_dir "$TMP_ROOT/grok-native-spellings") - : > "$dir/state/task1.meta" - out=$(printf '%s' '{"sessionId":"native","stop_hook_active":false}' | GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? - expect_code 2 "$status" "typed snake_case false must select native blocking" - assert_contains "$out" 'TURN WOULD END BLIND' "snake_case native block lost feedback" - out=$(printf '%s' '{"sessionId":"native","stopHookActive":true,"stop_hook_active":false}' | GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? - expect_code 0 "$status" "camelCase true must win over snake_case false" - out=$(printf '%s' '{"sessionId":"native","stopHookActive":false,"stop_hook_active":true}' | GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? - expect_code 2 "$status" "camelCase false must win over snake_case true" - pass "fm-turnend-guard-grok: both spellings are typed and camelCase has deterministic precedence" -} - -test_grok_adapter_invalid_inputs_start_neither_path() { - local dir fakebin log payload out status - dir=$(make_primary_dir "$TMP_ROOT/grok-invalid-inputs") - : > "$dir/state/task1.meta" - fakebin=$(fm_fakebin "$TMP_ROOT/grok-invalid-bin") - log="$TMP_ROOT/grok-invalid.log" - printf '#!/usr/bin/env bash\nprintf called >> %q\n' "$log" > "$fakebin/grok" - chmod +x "$fakebin/grok" - for payload in \ - ' ' \ - '{' \ - '{"sessionId":"x","stopHookActive":"false"}' \ - '{"sessionId":"x","stop_hook_active":1}' \ - '{"sessionId":"x"}{"sessionId":"y"}' \ - '{"sessionId":"x","stopHookActive":false}{"sessionId":"y","stopHookActive":false}' \ - '{"sessionId":"x","stopHookActive":"bad","stopHookActive":false}' \ - '{"sessionId":"x","stop_hook_active":false,"stop_hook_active":false}' \ - '{"sessionId":"x","sessionId":"y"}' - do - out=$(printf '%s' "$payload" | PATH="$fakebin:$PATH" GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? - expect_code 0 "$status" "invalid Grok payload must conservatively allow without choosing a path" - [ -z "$out" ] || fail "invalid Grok payload produced output: $out" - done - [ ! -e "$log" ] || fail "invalid Grok payload started a resume process" - out=$(printf '%s' '{"sessionId":"x","stopHookActive":false}' | PATH="$fakebin:$PATH" GROK_WORKSPACE_ROOT="$TMP_ROOT/missing-grok-root" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? - expect_code 0 "$status" "missing shared-guard prerequisite must conservatively allow" - [ -z "$out" ] || fail "missing prerequisite produced output: $out" - [ ! -e "$log" ] || fail "missing prerequisite started a resume process" - pass "fm-turnend-guard-grok: malformed, invalidly typed, and missing-prerequisite payloads start neither path" -} - -test_grok_adapter_missing_jq_and_no_supervision_allow() { - local dir fakebin log out status tool tool_path - dir=$(make_primary_dir "$TMP_ROOT/grok-nojq") - : > "$dir/state/task1.meta" - fakebin=$(fm_fakebin "$TMP_ROOT/grok-nojq-bin") - log="$TMP_ROOT/grok-nojq.log" - for tool in bash cat printf; do - tool_path=$(command -v "$tool") || fail "test host must provide $tool" - ln -s "$tool_path" "$fakebin/$tool" - done - printf '#!/usr/bin/env bash\nprintf called >> %q\n' "$log" > "$fakebin/grok" - chmod +x "$fakebin/grok" - out=$(printf '%s' '{"sessionId":"x","stopHookActive":false}' | PATH="$fakebin" GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? - expect_code 0 "$status" "missing jq must conservatively allow" - [ -z "$out" ] || fail "missing jq produced output: $out" - [ ! -e "$log" ] || fail "missing jq started a resume process" - - dir=$(make_primary_dir "$TMP_ROOT/grok-native-no-work") - out=$(printf '%s' '{"sessionId":"x","stopHookActive":false}' | GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? - expect_code 0 "$status" "healthy no-supervision-needed native stop must allow" - [ -z "$out" ] || fail "no-supervision-needed native stop produced output: $out" - pass "fm-turnend-guard-grok: missing jq and no-supervision-needed stops stay silent and bounded" +test_settings_hook_uses_claude_project_dir() { + local settings command + settings="$ROOT/.claude/settings.json" + [ -f "$settings" ] || fail "tracked .claude/settings.json is missing" + command=$(jq -r '.hooks.Stop[0].hooks[0].command // empty' "$settings") + [ -n "$command" ] || fail "Stop hook command is missing from .claude/settings.json" + assert_contains "$command" 'CLAUDE_PROJECT_DIR' "Stop hook must resolve via CLAUDE_PROJECT_DIR, not a cwd-relative path" + assert_contains "$command" 'fm-turnend-guard.sh --claude' "Stop hook must invoke fm-turnend-guard.sh in cooperative --claude mode" + case "$command" in + bin/fm-turnend-guard.sh|./bin/fm-turnend-guard.sh) + fail "Stop hook must not use a bare relative path (cwd-dependent): $command" + ;; + esac + pass ".claude/settings.json: Stop hook uses CLAUDE_PROJECT_DIR-anchored --claude guard command" +} + +test_codex_hook_invokes_shared_guard() { + local settings command + settings="$ROOT/.codex/hooks.json" + [ -f "$settings" ] || fail "tracked .codex/hooks.json is missing" + command=$(jq -r '.hooks.Stop[0].hooks[0].command // empty' "$settings") + [ -n "$command" ] || fail "Stop hook command is missing from .codex/hooks.json" + assert_contains "$command" 'pwd -P' "codex hook must anchor from the hook process working directory" + assert_contains "$command" '.codex/hooks.json' "codex hook must verify the hook-loaded firstmate root" + assert_contains "$command" 'fm-turnend-guard.sh' "codex hook must invoke the shared guard" + assert_not_contains "$command" '.cwd' "codex hook must not use payload cwd to select the guard executable" + pass ".codex/hooks.json: Stop hook invokes the shared primary guard" } test_codex_hook_uses_process_pwd_when_payload_cwd_is_outside_root() { @@ -769,6 +699,23 @@ EOF pass ".codex/hooks.json: Stop hook ignores nested git root guard scripts" } +test_opencode_plugin_forces_followup() { + local plugin content + plugin="$ROOT/.opencode/plugins/fm-primary-turnend-guard.js" + [ -f "$plugin" ] || fail "tracked OpenCode primary plugin is missing" + content=$(cat "$plugin") + assert_contains "$content" 'session.idle' "OpenCode plugin must run on session.idle" + assert_contains "$content" 'fm-turnend-guard.sh' "OpenCode plugin must invoke the shared guard" + assert_contains "$content" 'promptAsync' "OpenCode plugin must force a follow-up turn" + assert_contains "$content" 'encodeFirstmateOperationalInput' "OpenCode plugin must use the typed operational-input constructor" + assert_contains "$content" 'skipNextIdle' "OpenCode plugin must carry a loop guard" + assert_contains "$content" 'worktree' "OpenCode plugin must anchor the guard from the git worktree path" + assert_contains "$content" 'watcher cycle is missing, failed, or unhealthy' "OpenCode plugin must identify a blind turn as watcher recovery" + assert_contains "$content" 'harness recovery instruction below' "OpenCode plugin must delegate recovery action to the shared guard line" + assert_not_contains "$content" 'Resume supervision according to the session-start operating block' "OpenCode plugin must not route a blind turn through ordinary continuity" + pass ".opencode primary plugin: session.idle forces one follow-up through the shared guard" +} + test_opencode_plugin_anchors_guard_to_worktree() { local plugin parent worktree_dir wrong_dir out status plugin="$ROOT/.opencode/plugins/fm-primary-turnend-guard.js" @@ -828,6 +775,30 @@ EOF pass ".opencode primary plugin: guard path is anchored to worktree, not directory" } +test_pi_extension_forces_followup() { + local ext content + ext="$ROOT/.pi/extensions/fm-primary-turnend-guard.ts" + [ -f "$ext" ] || fail "tracked pi primary extension is missing" + content=$(cat "$ext") + assert_contains "$content" 'agent_settled' "pi extension must run after one logical agent run settles" + assert_contains "$content" 'fm-turnend-guard.sh' "pi extension must invoke the shared guard" + assert_contains "$content" 'sendUserMessage' "pi extension must force a follow-up turn" + assert_contains "$content" 'encodeFirstmateOperationalInput' "pi extension must use the typed operational-input constructor" + assert_contains "$content" 'deliverAs: "followUp"' "pi extension must queue the follow-up safely" + assert_contains "$content" 'guardFollowupActive' "pi extension must carry a logical-run loop guard" + assert_not_contains "$content" 'skipNextTurnEnd' "pi extension kept the internal-turn loop guard" + assert_contains "$content" 'watcher cycle is missing, failed, or unhealthy' "pi extension must identify a blind turn as watcher recovery" + assert_contains "$content" 'harness recovery instruction below' "pi extension must delegate recovery action to the shared guard line" + assert_not_contains "$content" 'Resume supervision according to the session-start operating block' "pi extension must not route a blind turn through ordinary continuity" + assert_contains "$content" '.pi-turnend-extension-loaded' "pi extension must write its loaded marker for session-start diagnostics" + assert_contains "$content" 'lockOwnership' "pi extension loaded marker must respect the session lock" + assert_contains "$content" 'const command = String((event.input as { command?: unknown })?.command ?? "")' "pi extension changed bash command extraction for the PreToolUse contract" + assert_contains "$content" 'runPretoolCheck(command)' "pi extension changed the PreToolUse checker invocation" + assert_contains "$content" 'return { block: true, reason:' "pi extension changed the checker exit-2 block result" + assert_not_contains "$content" 'Run bin/fm-watch-arm.sh as a background task' "pi extension must not hardcode the old watcher-arm instruction" + pass ".pi primary extension: agent_settled forces one follow-up through the shared guard" +} + test_pi_extension_injects_once_per_logical_agent_run() { local repo home ext log out status repo="$TMP_ROOT/pi-logical-run-root" @@ -1104,6 +1075,17 @@ test_hook_claude_mode_secondmate_reblocks_like_primary() { pass "fm-turnend-guard --claude: secondmate home re-blocks unclaimed and allows auto-arm-claimed stops" } +test_grok_hook_invokes_adapter() { + local settings command + settings="$ROOT/.grok/hooks/fm-primary-turnend-guard.json" + [ -f "$settings" ] || fail "tracked grok primary hook config is missing" + command=$(jq -r '.hooks.Stop[0].hooks[0].command // empty' "$settings") + [ -n "$command" ] || fail "Stop hook command is missing from grok primary hook config" + assert_contains "$command" 'GROK_WORKSPACE_ROOT' "grok hook must anchor from GROK_WORKSPACE_ROOT" + assert_contains "$command" 'fm-turnend-guard-grok.sh' "grok hook must invoke the adapter" + pass ".grok primary hook: Stop hook invokes the grok adapter" +} + test_predicate_healthy_no_inflight test_predicate_unhealthy_no_beacon test_predicate_unhealthy_stale_beacon @@ -1135,16 +1117,16 @@ test_hook_silent_without_stdin test_hook_runs_fast test_grok_adapter_forces_one_resume_when_unhealthy test_grok_adapter_loop_guard_skips_resume -test_grok_adapter_native_false_blocks_without_resume -test_grok_adapter_native_true_allows_without_resume -test_grok_adapter_snake_case_native_and_camel_precedence -test_grok_adapter_invalid_inputs_start_neither_path -test_grok_adapter_missing_jq_and_no_supervision_allow +test_settings_hook_uses_claude_project_dir +test_codex_hook_invokes_shared_guard test_codex_hook_uses_process_pwd_when_payload_cwd_is_outside_root test_codex_hook_ignores_nested_git_root_guard +test_opencode_plugin_forces_followup test_opencode_plugin_anchors_guard_to_worktree +test_pi_extension_forces_followup test_pi_extension_injects_once_per_logical_agent_run test_pi_extension_retries_after_followup_delivery_failure +test_grok_hook_invokes_adapter test_hook_claude_mode_reblocks_stop_hook_active_when_unhealthy test_hook_claude_mode_reblocks_x_mode_without_tasks test_hook_claude_mode_allows_when_autoarm_owner_alive From ae89c5c8a2481e463eb3bf6d6c8c5554573054bf Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Fri, 24 Jul 2026 15:23:00 -0700 Subject: [PATCH 08/52] fix(herdr): clean stale projections at session start (#996) * Clean stale Herdr projections at session start * no-mistakes(document): Document stale Herdr session-start projection cleanup * no-mistakes(review): Enforce locked exact Herdr projection cleanup * no-mistakes(review): Fail closed on unverified session lock ownership * no-mistakes(review): Serialize session lock acquisition atomically * no-mistakes(document): Align session-start and Herdr cleanup documentation * no-mistakes(document): Generalize lock-refusal diagnostics * no-mistakes(lint): Avoid reserved keyword in concurrency test * no-mistakes: apply CI fixes * no-mistakes: apply CI fixes --- AGENTS.md | 6 +-- bin/fm-lock.sh | 60 ++++++++++++++++++++++++--- bin/fm-test-run.sh | 10 ++++- docs/architecture.md | 2 +- docs/configuration.md | 2 +- docs/herdr-backend.md | 24 ++++++++++- docs/verification/runtime-backends.md | 9 ++++ 7 files changed, 98 insertions(+), 15 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index c8b61b8077d..bf13013b337 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -129,16 +129,16 @@ Read the complete digest once and trust it as this turn's startup and recovery i Do not separately re-read the context, backlog, metadata, or bulk status inputs it just printed unless a source was reported absent or corrupt, older history is specifically needed, or a targeted workflow must inspect before writing. An `ABSENT` captain, shared-captain, secondmate, or learnings file means the firstmate repo's built-in defaults, no shared captain preferences, no registered secondmates, or no captured learnings; rebuild an absent or stale project registry from the clones before dispatch. -If the session lock is refused, tell the captain another active session is managing the fleet and remain read-only. +If the session lock cannot be acquired and verified, report its exact diagnostic and remain read-only; another active session is only one possible cause. A lock-refused session must not spawn, steer, merge, drain the wake queue, repair supervision, repair a checkout, or perform any other fleet mutation. 1. **Lock** - acquires the per-home session lock first, before anything mutates shared state. 2. **Bootstrap** - detect-only checks (tool/version problems, GitHub auth, the worktree-tangle check, harness override, dispatch-profile validation, backlog-backend status) always run, but routine confirmations stay silent by default. When the lock could not be acquired, the worktree-tangle check uses read-only advisory wording without a checkout repair command. - The five MUTATING sweeps - non-executing legacy PR-check migration, fleet sync, the local secondmate fast-forward sweep, the secondmate liveness sweep, and X-mode artifact writes - run only when this session actually holds the lock from step 1. + Home-local stale Herdr projection cleanup and the five bootstrap MUTATING sweeps - non-executing legacy PR-check migration, fleet sync, the local secondmate fast-forward sweep, the secondmate liveness sweep, and X-mode artifact writes - run only when this session actually holds the lock from step 1. The secondmate liveness sweep deterministically accounts for every registered secondmate: it relaunches only from the recovery-grade `dead` or `missing` states, preserves ambiguous or unreadable targets, and reports skipped or failed guarantees as `SECONDMATE_LIVENESS:` lines (`bin/fm-bootstrap.sh`; `bin/fm-backend.sh`'s `fm_backend_agent_state`). 3. **Wake queue** - when locked, drains the durable wake queue and prints the raw records prominently as this turn's first work queue; a bounded, clearly labeled historical status-event annotation may follow a valid `signal` record but never replaces it or current-state reconciliation, and a lapsed watcher chain still surfaces here via the same guard alarm. - When the lock could not be acquired, the queue is left untouched because another session owns it, and the guard's tangle/watcher-liveness alarms still print in read-only advisory mode without drain, supervision repair, or checkout repair commands. + When the lock could not be acquired and verified, the queue is left untouched because no session mutation is authorized, and the guard's tangle/watcher-liveness alarms still print in read-only advisory mode without drain, supervision repair, or checkout repair commands. 4. **Context digest** - the full contents of `data/projects.md`, `data/secondmates.md`, `data/captain.md`, `data/captain-shared.md`, and `data/learnings.md`, each clearly delimited. A file that does not exist prints an explicit `ABSENT` marker, never confused with an empty-but-present file: absence is meaningful (`captain.md` absent means use the firstmate repo's built-in defaults, `projects.md` absent means rebuild it from the clones under `projects/`, etc.). 5. **Fleet-state digest** - the compact backlog listing owned by `bin/fm-session-start.sh`; every `state/<id>.meta`; a bounded tail of each task's `state/<id>.status` (labeled as wake-EVENT history, not current state, with the full log path printed for a deeper read); the `state/.afk` flag; and one cheap alive/dead read of each task's recorded backend endpoint. diff --git a/bin/fm-lock.sh b/bin/fm-lock.sh index 2e82574ecaf..083675b2beb 100755 --- a/bin/fm-lock.sh +++ b/bin/fm-lock.sh @@ -3,7 +3,7 @@ # Writes the harness (agent) process PID found by walking the shell's ancestry, # which lives as long as the firstmate session - unlike the transient subshell # PID of any one tool call, which is dead moments after it is written. -# Usage: fm-lock.sh acquire; exit 1 if another live session holds it +# Usage: fm-lock.sh acquire; exit 1 unless ownership is verified # fm-lock.sh status print holder and liveness; always exits 0 set -u @@ -12,7 +12,10 @@ FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" LOCK="$STATE/.lock" -mkdir -p "$STATE" +mkdir -p "$STATE" 2>/dev/null || { + echo "error: cannot create session-lock state directory $STATE; operate read-only until resolved" >&2 + exit 1 +} # Harness identity (FM_HARNESS_RE, ancestry walk, holder liveness) is owned by # the shared session-lock lib so the Claude Stop auto-arm applies the exact @@ -22,18 +25,63 @@ mkdir -p "$STATE" if [ "${1:-}" = "status" ]; then if [ ! -f "$LOCK" ]; then echo "lock: free"; exit 0; fi - old=$(cat "$LOCK") + old=$(cat "$LOCK" 2>/dev/null) || { + echo "lock: unreadable" + exit 0 + } if fm_harness_pid_alive "$old"; then echo "lock: held by live harness pid $old"; else echo "lock: stale (pid $old dead or not a harness)"; fi exit 0 fi me=$(fm_harness_ancestry_pid) || { echo "error: cannot locate harness process in ancestry" >&2; exit 1; } -if [ -f "$LOCK" ]; then - old=$(cat "$LOCK") +probe=$(mktemp "$STATE/.lock-write.XXXXXX" 2>/dev/null) || { + echo "error: cannot write session lock; operate read-only until resolved" >&2 + exit 1 +} +rm -f "$probe" 2>/dev/null || { + echo "error: cannot clean session-lock publication probe; operate read-only until resolved" >&2 + exit 1 +} +# shellcheck source=bin/fm-wake-lib.sh +. "$SCRIPT_DIR/fm-wake-lib.sh" +CLAIM_LOCK="$STATE/.lock.acquire" +CLAIM_LOCK_HELD=0 +release_claim_lock() { + if [ "$CLAIM_LOCK_HELD" -eq 1 ]; then + fm_lock_release "$CLAIM_LOCK" + CLAIM_LOCK_HELD=0 + fi +} +trap release_claim_lock EXIT +trap 'exit 1' HUP INT TERM +fm_lock_acquire_wait "$CLAIM_LOCK" +CLAIM_LOCK_HELD=1 + +if [ -e "$LOCK" ] || [ -L "$LOCK" ]; then + if [ ! -f "$LOCK" ] || [ -L "$LOCK" ]; then + echo "error: session lock is not a regular file; operate read-only until resolved" >&2 + exit 1 + fi + old=$(cat "$LOCK" 2>/dev/null) || { + echo "error: session lock is unreadable; operate read-only until resolved" >&2 + exit 1 + } if [ "$old" != "$me" ] && fm_harness_pid_alive "$old"; then echo "error: another live firstmate session holds the lock (pid $old); operate read-only until resolved" >&2 exit 1 fi fi -echo "$me" > "$LOCK" +if ! { printf '%s\n' "$me" > "$LOCK"; } 2>/dev/null; then + echo "error: cannot write session lock; operate read-only until resolved" >&2 + exit 1 +fi +written=$(cat "$LOCK" 2>/dev/null) || { + echo "error: cannot verify session lock ownership; operate read-only until resolved" >&2 + exit 1 +} +if [ ! -f "$LOCK" ] || [ -L "$LOCK" ] || [ "$written" != "$me" ]; then + echo "error: session lock ownership verification failed; operate read-only until resolved" >&2 + exit 1 +fi +release_claim_lock echo "lock acquired: harness pid $me" diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 8e06d169e91..28be3a25c77 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -140,6 +140,7 @@ family_for_basename() { fm-afk-inject-herdr-e2e.test.sh|fm-afk-launch.test.sh|fm-backend-autodetect-smoke.test.sh|\ fm-backend-herdr-eventwait-smoke.test.sh|fm-backend-herdr-presentation-e2e.test.sh|\ fm-backend-herdr-prune-safety-e2e.test.sh|fm-backend-herdr-respawn-idem-e2e.test.sh|\ + fm-herdr-session-cleanup-e2e.test.sh|\ fm-backend-herdr-smoke.test.sh|fm-backend-herdr-workspace-per-home-e2e.test.sh) printf '%s\n' real-herdr-gated ;; @@ -160,8 +161,8 @@ family_for_basename() { printf '%s\n' live-harness-optin ;; fm-backend-herdr.test.sh|fm-backend-tmux-smoke.test.sh|fm-backend.test.sh|\ - fm-send-strict.test.sh|fm-spawn-batch.test.sh|fm-spawn-dispatch-profile.test.sh|\ - fm-spawn-worktree-settle.test.sh) + fm-herdr-session-cleanup.test.sh|fm-send-strict.test.sh|fm-spawn-batch.test.sh|\ + fm-spawn-dispatch-profile.test.sh|fm-spawn-worktree-settle.test.sh) printf '%s\n' backend-dispatch ;; fm-pr-check-security.test.sh|fm-pr-merge.test.sh|fm-review-diff.test.sh|\ @@ -621,6 +622,11 @@ families_for_changed_path() { printf '%s\n' backend-dispatch printf '%s\n' pure-contract-unit ;; + bin/fm-herdr-session-cleanup.sh) + printf '%s\n' session-bootstrap + printf '%s\n' real-herdr-gated + printf '%s\n' backend-dispatch + ;; bin/backends/zellij*|tests/zellij-test-safety.sh) printf '%s\n' zellij printf '%s\n' backend-dispatch diff --git a/docs/architecture.md b/docs/architecture.md index 8671b44045e..3a330e6cd94 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -99,7 +99,7 @@ For capable Herdr sessions, the same watcher replaces its terminal sleep with a The deeper session-start agent-process liveness probe is separate from that busy-state poll: tmux and Herdr have verified classifiers for secondmate recovery, Zellij remains unverified, and Orca and cmux do not support secondmate spawns. Herdr is experimental and can be selected explicitly or by runtime auto-detection: Treehouse remains its worktree provider, [`herdr-backend.md`](herdr-backend.md) owns current setup and safety limits, and [`verification/runtime-backends.md`](verification/runtime-backends.md#herdr) owns active empirical evidence. Herdr's durable default container shape is workspace-per-home plus tab-per-task: the primary home uses workspace label `firstmate`, secondmate homes use `2ndmate-<secondmate-id>`, and recovery/list-live scopes to the current `FM_HOME`'s workspace. -Its optional default-off presentation projection may place one clean new task in a disposable workspace without changing endpoint authority or lifecycle ownership; [Optional presentation spaces](herdr-backend.md#optional-presentation-spaces) owns that conditional design. +Its optional default-off presentation projection may place one clean new task in a disposable workspace without changing endpoint authority or lifecycle ownership; [Optional presentation spaces](herdr-backend.md#optional-presentation-spaces) owns that conditional design and its narrow home-local restored-shell cleanup at locked session start. Zellij is experimental and selected only explicitly: Treehouse remains its worktree provider, [`zellij-backend.md`](zellij-backend.md) owns current setup and limits, and [`verification/runtime-backends.md`](verification/runtime-backends.md#zellij) owns active empirical evidence. Zellij's container shape is simpler than herdr's: one shared `firstmate` session, one tab per task, with no per-home workspace split; visible tab titles are scoped by the active home label plus a short hash of the resolved `FM_ROOT` path. Orca is experimental and selected only explicitly: Orca owns both worktree and terminal lifecycle, records `orca_worktree_id=` and `terminal=`, and removes worktrees through `orca worktree rm` only after the usual firstmate teardown checks pass. diff --git a/docs/configuration.md b/docs/configuration.md index e392840769d..d8e1c39f9ed 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -80,7 +80,7 @@ These five sentences are the single owner of the task-selector vocabulary; backe `fm-teardown.sh <id>` takes a task id directly and uses the same recorded backend target fields after loading `state/<id>.meta`. By default, Herdr workspaces are derived from `FM_HOME`: the primary home uses `firstmate`, and a secondmate home marked by `.fm-secondmate-home` uses `2ndmate-<secondmate-id>`. The default-container spawn, list-live, and recovery paths read that label from the active home, so a secondmate's own crewmates stay inside that secondmate home's herdr space. -The optional local `config/herdr-presentation-spaces` presence flag instead enables Herdr's default-off disposable single-task visual projection; [Optional presentation spaces](herdr-backend.md#optional-presentation-spaces) owns its behavior, safety limits, and recovery contract. +The optional local `config/herdr-presentation-spaces` presence flag instead enables Herdr's default-off disposable single-task visual projection; [Optional presentation spaces](herdr-backend.md#optional-presentation-spaces) owns its behavior, safety limits, recovery contract, and narrow locked session-start cleanup of exact restored idle-shell children. The flag is default-off and inherited into secondmate homes under the primary-authoritative contract owned by [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md). For normal herdr operations, `HERDR_SESSION` selects the named session, but destructive test cleanup must not rely on `HERDR_SESSION` alone. Use the explicit guarded cleanup path described in [`docs/herdr-backend.md`](herdr-backend.md) instead of `herdr server stop`. diff --git a/docs/herdr-backend.md b/docs/herdr-backend.md index d71a1bfdc95..c9e7c59ff0a 100644 --- a/docs/herdr-backend.md +++ b/docs/herdr-backend.md @@ -96,17 +96,35 @@ A failed replacement rolls back only the exact response-derived new pane when fo Version 1 journals, dead or missing panes, duplicate or absent tokens, renamed or detached spaces, cross-home mismatches, inconsistent endpoint bindings, active target tabs, and ambiguous identity or focus fall back flat without mutating the old projection when duplicate-agent risk is positively absent. A live or unknown recorded or token-matched endpoint refuses duplicate launch. +Locked session start has one narrower cleanup for a restored projected child that is no longer current task state. +It runs only when the current home has at least one ordinary presentation journal and considers only that home; a primary never recursively sweeps a secondmate home. +Discovery starts from the exact current `└ <concise-task> · p:<22-character-token>` grammar, but a title or token alone is never mutation authority. +The title must contain exactly one token occurrence across the named-session snapshot and must equal the title derived from exactly one valid presentation journal in this home's own `state/`; a version 2 journal additionally must bind this exact physical home, named session, workspace, tab, and pane. +The task's ordinary metadata must be absent, and the candidate must have exactly one tab and exactly one pane. +Before cleanup, Firstmate acquires the existing task-id spawn lock and then the shared named-session presentation lock. +Inside both locks it takes one exact snapshot, requires one unambiguous non-target focus and the exact title, token, tab, and pane shape, positively confirms no registered agent, and reads Herdr's process information for the exact named-session pane. +The process proof requires one recognized idle shell as both the shell process and the sole foreground process-group member, an operating-system process-table row for that shell, no child process, and a sleeping or idle shell state. +Any foreground command, child process, active shell job, unknown shell, unreadable process table, missing field, or API error preserves the pane. +Firstmate immediately revalidates the same journal, metadata absence, workspace title and token uniqueness, one-tab and one-pane topology, exact pane relationship, absent agent, process proof, and non-target focus before calling the existing exact-pane focus-preserving close helper. +It closes only that pane, never a workspace. +The matching journal is retired only after the exact pane is positively confirmed gone; an unconfirmed close retains the journal, while a confirmed close may retire it even when focus restoration reported an error after the close. +A second run finds no matching title or journal and is a no-op. +A malformed or missing title or token, duplicate token, zero or multiple journal matches, cross-home version 2 binding, current metadata, registered or unknown agent, extra tab or pane, active target, busy lock, changed revalidation, unreadable check, or any error preserves the candidate and lets session startup continue with at most a concise warning. + Operational compromises: - Grouping is best-effort; only an exact same-identity version 2 binding survives a Herdr restart in place. - Existing layouts are not force-renamed or rearranged. - Missing or ambiguous restart bindings fall back to the ordinary home workspace while the old projection remains untouched. -- Crashes, lost responses, failed exact-pane cleanup, or human renames can leave quarantined spaces. -- Spaces have no cross-home cleanup path. +- Crashes, lost responses, failed exact-pane cleanup, or human renames can leave quarantined spaces; session start removes only the exact home-local, uniquely journal-correlated, childless idle-shell shape above. +- Spaces have no cross-home cleanup path, and a secondmate child can clean up only from its exact home. +- Every stale-looking space outside that narrow startup proof still requires manual cleanup in Herdr's UI after human inspection. - Regaining a dedicated space after degradation requires stopping the flat task, manually checking the stale projection, and clearing its journal before a genuinely fresh launch. - The visible token is only a restart-stable correlator and never substitutes for the exact binding. `tests/fm-backend-herdr-presentation-e2e.test.sh` covers multi-home ordering, concurrency, lock contention, legacy coexistence, focus preservation, exact same-identity restart replacement, ambiguous bindings and tokens, and exact-pane cleanup through the guarded lab path. +`tests/fm-herdr-session-cleanup.test.sh` covers every discovery, ownership, topology, process, locking, revalidation, focus, retirement, and continue-on-error boundary. +`tests/fm-herdr-session-cleanup-e2e.test.sh` covers the restored-shell cleanup in a guarded non-default named lab; [`verification/runtime-backends.md`](verification/runtime-backends.md#per-home-and-presentation-topology) owns the active versioned evidence. ## Default-tab prune safety @@ -258,6 +276,8 @@ tests/fm-backend-herdr-respawn-idem-e2e.test.sh tests/fm-backend-herdr-workspace-per-home-e2e.test.sh tests/fm-backend-herdr-presentation-e2e.test.sh tests/fm-backend-herdr-eventwait-smoke.test.sh +tests/fm-herdr-session-cleanup.test.sh +tests/fm-herdr-session-cleanup-e2e.test.sh tests/fm-afk-inject-herdr-e2e.test.sh tests/fm-afk-pi-herdr-return-e2e.test.sh ``` diff --git a/docs/verification/runtime-backends.md b/docs/verification/runtime-backends.md index e012f0dc5e5..046c12a02d0 100644 --- a/docs/verification/runtime-backends.md +++ b/docs/verification/runtime-backends.md @@ -143,6 +143,15 @@ ok - real Herdr lab: missing, renamed, and duplicate tokens trigger zero destruc ok - real Herdr lab validation completed on Herdr 0.7.5 with the default-session tripwire intact ``` +The restored-shell session-start cleanup ran on 2026-07-24 against Herdr 0.7.5 protocol 17: + +```sh +HERDR_LAB_HELPER=bin/fm-herdr-lab.sh \ + tests/fm-herdr-session-cleanup-e2e.test.sh +``` + +Observed guarantee: one exact home-local, journal-correlated, one-tab and one-pane childless idle shell was closed after restoration while the exact non-target focus and default fleet session remained unchanged, and a repeat run was a no-op. + ### Composer and operational input Real captures verified these active distinctions: From 38a0693459066251e66bce8b51a7f3ebff16e613 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Fri, 24 Jul 2026 16:15:45 -0700 Subject: [PATCH 09/52] fix: recover Claude supervision without watcher-status gate (#1001) * fix: recover Claude supervision at session start * fix: remove Claude watcher-status command gate * no-mistakes(document): docs: remove stale continuity gate references --- .claude/settings.json | 4 - bin/fm-claude-stop-autoarm.sh | 36 +++- bin/fm-continuity-pretool-check.sh | 113 ----------- bin/fm-test-run.sh | 4 +- docs/architecture.md | 3 +- docs/arm-pretool-check.md | 11 -- docs/fm-test-isolation-proof.md | 181 +++++++++++++----- docs/subagent-guard.md | 2 +- docs/supervision-protocols/claude.md | 5 +- docs/verification/supervision.md | 12 +- docs/watcher-continuity.md | 12 +- tests/fm-claude-continuity-live-e2e.test.sh | 73 ------- tests/fm-claude-stop-autoarm-live-e2e.test.sh | 72 ++++--- tests/fm-claude-stop-autoarm.test.sh | 66 ++++++- tests/fm-continuity-pretool-check.test.sh | 128 ------------- tests/fm-subagent-pretool-check.test.sh | 71 ++++--- tests/fm-test-isolation-proof.test.sh | 134 ++++++++++++- 17 files changed, 450 insertions(+), 477 deletions(-) delete mode 100755 bin/fm-continuity-pretool-check.sh delete mode 100755 tests/fm-claude-continuity-live-e2e.test.sh delete mode 100755 tests/fm-continuity-pretool-check.test.sh diff --git a/.claude/settings.json b/.claude/settings.json index 9f145c88913..e77613c98a4 100644 --- a/.claude/settings.json +++ b/.claude/settings.json @@ -22,10 +22,6 @@ { "type": "command", "command": "\"$CLAUDE_PROJECT_DIR\"/bin/fm-cd-pretool-check.sh --claude" - }, - { - "type": "command", - "command": "\"$CLAUDE_PROJECT_DIR\"/bin/fm-continuity-pretool-check.sh" } ] }, diff --git a/bin/fm-claude-stop-autoarm.sh b/bin/fm-claude-stop-autoarm.sh index f730d35c7f7..df9ee1128fc 100755 --- a/bin/fm-claude-stop-autoarm.sh +++ b/bin/fm-claude-stop-autoarm.sh @@ -10,8 +10,11 @@ # - Scope: only a genuine primary checkout (plain checkout or validly marked # secondmate home) with AGENTS.md, bin/, and the effective state dir - the # exact fm-turnend-guard.sh scope. Child crew/scout worktrees stay inert. -# - Identity: only when THIS session's harness ancestor holds state/.lock, so -# a scratch or read-only session in the same checkout never arms or rewakes. +# - Identity: only when THIS session's harness ancestor holds state/.lock. +# When an existing numeric owner fails the shared harness-liveness predicate, +# the hook delegates guarded recovery to bin/fm-lock.sh and then re-verifies +# ownership. A live owner, missing lock, malformed lock, or unresolved +# ancestry remains inert, so a competing session never arms or rewakes. # - AFK: while state/.afk exists the away daemon owns the watcher and triage; # this hook exits 0 and NEVER rewakes the primary (checked again at # translation time so a mid-cycle AFK transition is honored). @@ -37,8 +40,9 @@ # # This hook never blocks the Stop decision itself and never prints to stdout: # exit 0 is always silent, and exit 2 carries the rewake banner on stderr. -# On any uncertainty such as unresolvable ancestry or lock contention, it exits -# 0 and leaves continuity to the synchronous guard and the model. +# On any uncertainty such as unresolvable ancestry, malformed lock state, or +# lock contention, it exits 0 and leaves continuity to the synchronous guard and +# the model. set -u SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -67,7 +71,20 @@ cat >/dev/null 2>&1 || true fm_primary_scope_matches "$FM_ROOT" "$STATE" || exit 0 # --- identity: only the lock-owning session's hooks may arm ------------------ -fm_session_lock_owned_by_self "$STATE" || exit 0 +# A prior session may have died after leaving its numeric harness pid in .lock. +# Use the shared liveness predicate to recognize only that stale-owner case. +# Defer the mutating claim until after the unchanged AFK and need gates, so an +# idle or away home remains byte-for-byte inert. Missing or malformed locks are +# uncertainty rather than stale-owner evidence and remain inert. +RECOVER_SESSION_LOCK=0 +if ! fm_session_lock_owned_by_self "$STATE"; then + LOCK_PID=$(cat "$STATE/.lock" 2>/dev/null || true) + case "$LOCK_PID" in + ''|*[!0-9]*) exit 0 ;; + esac + fm_harness_pid_alive "$LOCK_PID" && exit 0 + RECOVER_SESSION_LOCK=1 +fi # --- AFK: the away daemon owns the watcher and triage; never rewake ---------- [ -e "$STATE/.afk" ] && exit 0 @@ -78,6 +95,15 @@ need_supervision() { } need_supervision || exit 0 +# --- stale session-lock recovery --------------------------------------------- +# Delegate the claim to fm-lock.sh so its live-owner refusal and write semantics +# remain the single acquisition owner, then re-verify current-session identity +# before touching any auto-arm state. +if [ "$RECOVER_SESSION_LOCK" -eq 1 ]; then + "$SCRIPT_DIR/fm-lock.sh" >/dev/null 2>&1 || exit 0 + fm_session_lock_owned_by_self "$STATE" || exit 0 +fi + # --- single-flight owner claim ------------------------------------------------ # Claude runs one background process per firing with no dedupe. Exactly one # owner foregrounds the arm and translates its close; every other firing exits diff --git a/bin/fm-continuity-pretool-check.sh b/bin/fm-continuity-pretool-check.sh deleted file mode 100755 index e8db8431929..00000000000 --- a/bin/fm-continuity-pretool-check.sh +++ /dev/null @@ -1,113 +0,0 @@ -#!/usr/bin/env bash -# Claude primary watcher-continuity PreToolUse gate. -# -# This hook is deliberately narrow. It denies only an executed bin/fm-*.sh fleet -# command other than bin/fm-wake-drain.sh, bin/fm-watch-arm.sh, or the -# independently fail-closed bin/fm-teardown.sh, and only when the active primary -# home has task metadata in flight but no identity-matched live watcher holds the -# home lock. Ordinary shell commands, recovery commands, healthy supervision, -# fleet-idle homes, and child worktrees are always allowed. -# -# The turn-end guard remains the final backstop, cooperating with the -# Stop-owned auto-arm in its --claude mode. This gate closes the long-turn gap -# before another fleet mutation, but does not replace or weaken the Stop hooks. -# -# Input is Claude PreToolUse JSON on stdin. Tests may pass --command directly. -# Malformed transport, missing jq/Node, a missing classifier, or classifier -# failure all fail open. A deny writes Claude's hook decision to stderr only and -# exits 2. -set -u - -COMMAND= -COMMAND_SET=0 - -usage() { - cat <<'EOF' -Usage: fm-continuity-pretool-check.sh [--command <shell-command>] - -Reads Claude PreToolUse JSON from stdin unless --command is supplied. -Exits 0 to allow. Exits 2 with a Claude deny object on stderr only when an -unhealthy primary tries to execute a non-recovery firstmate fleet script. -EOF -} - -while [ "$#" -gt 0 ]; do - case "$1" in - --command) - [ "$#" -gt 1 ] || { echo "error: --command requires a value" >&2; exit 2; } - COMMAND=$2 - COMMAND_SET=1 - shift 2 - ;; - --command=*) - COMMAND=${1#--command=} - COMMAND_SET=1 - shift - ;; - -h|--help) - usage - exit 0 - ;; - *) - echo "error: unknown argument: $1" >&2 - usage >&2 - exit 2 - ;; - esac -done - -if [ "$COMMAND_SET" -eq 0 ]; then - PAYLOAD=$(cat 2>/dev/null || true) - [ -n "$PAYLOAD" ] || exit 0 - command -v jq >/dev/null 2>&1 || exit 0 - COMMAND=$(printf '%s' "$PAYLOAD" | jq -r '.tool_input.command // empty' 2>/dev/null) || exit 0 -fi -[ -n "$COMMAND" ] || exit 0 - -SCRIPT_DIR=$(CDPATH='' cd -- "$(dirname -- "${BASH_SOURCE[0]}")" 2>/dev/null && pwd -P) || exit 0 -FM_ROOT=${FM_ROOT_OVERRIDE:-$(CDPATH='' cd -- "$SCRIPT_DIR/.." 2>/dev/null && pwd -P)} -FM_HOME=${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}} -STATE=${FM_STATE_OVERRIDE:-$FM_HOME/state} -WATCH="$SCRIPT_DIR/fm-watch.sh" -POLICY="$SCRIPT_DIR/fm-continuity-command-policy.mjs" - -# shellcheck source=bin/fm-supervision-lib.sh -. "$SCRIPT_DIR/fm-supervision-lib.sh" -# shellcheck source=bin/fm-primary-scope-lib.sh -. "$SCRIPT_DIR/fm-primary-scope-lib.sh" -# shellcheck source=bin/fm-wake-lib.sh -. "$SCRIPT_DIR/fm-wake-lib.sh" - -fm_primary_scope_matches "$FM_ROOT" "$STATE" || exit 0 -fm_supervision_status "$STATE" "${FM_GUARD_GRACE:-300}" -[ "$FM_SUP_IN_FLIGHT" -gt 0 ] || exit 0 -LOCK_PID=$(cat "$STATE/.watch.lock/pid" 2>/dev/null || true) -if fm_pid_alive "$LOCK_PID" && fm_watcher_lock_matches_pid "$STATE" "$WATCH" "$LOCK_PID" "$FM_HOME"; then - exit 0 -fi - -command -v node >/dev/null 2>&1 || exit 0 -[ -f "$POLICY" ] || exit 0 -CLASSIFICATION=$(node "$POLICY" --command "$COMMAND" --root "$FM_ROOT" 2>/dev/null) || exit 0 -case "$CLASSIFICATION" in - deny*) ;; - *) exit 0 ;; -esac - -TAB=$(printf '\t') -REST=${CLASSIFICATION#*"$TAB"} -[ -n "$REST" ] && [ "$REST" != "$CLASSIFICATION" ] || exit 0 -BLOCKED_SCRIPT=${REST%%"$TAB"*} -REASON_CODE=${REST#*"$TAB"} -[ "$REASON_CODE" != "$REST" ] || REASON_CODE="" -case "$REASON_CODE" in - unsafe-teardown) - REASON="[watcher-continuity] tasks are in flight and no live watcher holds this home lock; during recovery only the ordinary literal bin/fm-teardown.sh is allowed, so drop --force and any shell-expanded arguments and retry the literal invocation (blocked: $BLOCKED_SCRIPT)" - ;; - *) - REASON="[watcher-continuity] tasks are in flight and no live watcher holds this home lock; drain wakes with bin/fm-wake-drain.sh, use fail-closed bin/fm-teardown.sh for completed tasks when needed, then end the turn so the Stop-owned auto-arm re-establishes the watcher; if the Stop auto-arm itself failed, re-arm manually with bin/fm-watch-arm.sh as a tracked Claude background task (blocked: $BLOCKED_SCRIPT)" - ;; -esac -ESCAPED=$(printf '%s' "$REASON" | sed -e 's/\\/\\\\/g' -e 's/"/\\"/g' | tr '\n' ' ') -printf '{"hookSpecificOutput":{"hookEventName":"PreToolUse","permissionDecision":"deny"},"systemMessage":"%s"}\n' "$ESCAPED" >&2 -exit 2 diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 28be3a25c77..a86bb0b661a 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -120,7 +120,7 @@ family_for_basename() { fm-arm-pretool-check.test.sh|fm-ask-user-authority.test.sh|fm-brief.test.sh|\ fm-calm-pi-extension.test.sh|fm-captain-translation-contract.test.sh|fm-cd-pretool-check.test.sh|\ fm-composer-ghost.test.sh|fm-composer-lib.test.sh|\ - fm-continuity-pretool-check.test.sh|fm-crew-state.test.sh|fm-decision-hold-lifecycle.test.sh|\ + fm-crew-state.test.sh|fm-decision-hold-lifecycle.test.sh|\ fm-dispatch-select.test.sh|fm-documentation-audiences.test.sh|fm-ensure-agents-md.test.sh|fm-grok-harness.test.sh|\ fm-herdr-lab.test.sh|fm-instruction-owners.test.sh|fm-lint.test.sh|\ fm-install-herdr.test.sh|fm-nm-test-contract.test.sh|fm-no-mistakes-ownership.test.sh|\ @@ -154,7 +154,7 @@ family_for_basename() { fm-update.test.sh) printf '%s\n' session-bootstrap ;; - fm-afk-pi-herdr-return-e2e.test.sh|fm-claude-continuity-live-e2e.test.sh|\ + fm-afk-pi-herdr-return-e2e.test.sh|\ fm-codex-continuity-live-e2e.test.sh|fm-grok-continuity-live-e2e.test.sh|\ fm-opencode-primary-live-e2e.test.sh|fm-pi-primary-live-e2e.test.sh|\ fm-send-secondmate-marker-herdr-e2e.test.sh) diff --git a/docs/architecture.md b/docs/architecture.md index 3a330e6cd94..cb515ba6350 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -59,7 +59,8 @@ That block owns the live wait shape for the running primary harness: Claude's St On `attached` it stays live across identity-matched successors, and an unexplained clean child close either attaches to a verified healthy successor or becomes the typed nonzero `watcher: FAILED - cycle ended without an actionable reason` result. The arm layer records one bounded lifecycle row per observed cycle in `state/.watch-cycle-exits.log`; `state/.watch-triage.log` remains exclusively the absorbed-wake debug log. Pi and OpenCode verify session-lock ownership and launch one singleton successor from their child-close handlers before delivering an actionable wake prompt, with bounded exponential retry for failed restoration. -Claude's `bin/fm-claude-stop-autoarm.sh` hook fires on every Stop and, when the home is eligible and still needs supervision, claims one home-scoped cycle, foregrounds the arm wrapper, and translates an actionable close or typed failure into one exit-2 rewake; its narrow PreToolUse continuity gate allows drain, arm recovery, and fail-closed teardown while refusing only other fleet commands when tasks are in flight and no identity-matched live watcher holds the home lock. +Claude's `bin/fm-claude-stop-autoarm.sh` hook fires on every Stop and, when the home is eligible and still needs supervision, claims one home-scoped cycle, foregrounds the arm wrapper, and translates an actionable close or typed failure into one exit-2 rewake. +[`watcher-continuity.md`](watcher-continuity.md) owns Claude's residual active-turn coverage and watcher-status command-gating boundary. The existing turn-end guard remains the final backstop for all five harness protocols, cooperating with the auto-arm claim in its `--claude` mode. Its `--restart` mode signals only the watcher recorded in the current home's `state/.watch.lock`, so restarting one home cannot kill sibling secondmate watchers. A pull-based guard (`bin/fm-guard.sh`) warns through supervision tool output if the primary checkout is tangled, or if tasks are in flight and that watcher stops running or queued wakes are waiting to be drained. diff --git a/docs/arm-pretool-check.md b/docs/arm-pretool-check.md index a5749ac786b..c56f555f7e6 100644 --- a/docs/arm-pretool-check.md +++ b/docs/arm-pretool-check.md @@ -15,17 +15,6 @@ The seatbelt rejects those command shapes before execution. This policy is not a post-arm liveness guarantee. `bin/fm-guard.sh`, `bin/fm-turnend-guard.sh`, the watcher lock, and the watcher beacon still prove whether supervision is healthy after an allowed call. -## Claude continuity gate - -Claude also registers `bin/fm-continuity-pretool-check.sh` for Bash PreToolUse events. -This is a separate, tightly bounded recovery gate rather than another watcher-shape policy. -It runs only in a primary home, and it denies only an executed `bin/fm-*.sh` command other than `bin/fm-wake-drain.sh`, `bin/fm-watch-arm.sh`, or the ordinary literal `bin/fm-teardown.sh` when task metadata is in flight and no identity-matched live watcher holds that home's lock. -Ordinary shell commands, fleet-script names used as data, all commands in an idle fleet, child worktrees, wake drain, watcher arm, and ordinary literal teardown remain allowed. -The denial gives Claude reason-specific recovery guidance - drain, use fail-closed `bin/fm-teardown.sh` for completed tasks, then end the turn so the Stop-owned auto-arm re-establishes the watcher, with a manual tracked Claude background-task arm only when the Stop auto-arm itself failed - per the contract in [`watcher-continuity.md`](watcher-continuity.md). -`bin/fm-continuity-command-policy.mjs` reuses this document's shell lexer and command-position analysis but owns the recovery-versus-other-fleet classification. -Malformed transport or opaque dynamic syntax fails open so this narrow gate cannot become a blanket Bash block. -The existing `bin/fm-turnend-guard.sh` Stop integration remains the final backstop, cooperating with the Stop-owned auto-arm in its `--claude` mode ([`turnend-guard.md`](turnend-guard.md)). - The classifier never executes, sources, evaluates, or expands any part of the submitted command. It tokenizes the bytes and classifies lexical execution positions only. diff --git a/docs/fm-test-isolation-proof.md b/docs/fm-test-isolation-proof.md index 716dca73a56..7d5db970ae7 100644 --- a/docs/fm-test-isolation-proof.md +++ b/docs/fm-test-isolation-proof.md @@ -1,39 +1,61 @@ -# Firstmate test isolation proof +# Firstmate test isolation proof (Phase 2) -This record is the concurrent isolation proof for the portable parallel candidate set. -`bin/fm-test-isolation-proof.sh` is the authoritative harness and `docs/fm-test-isolation-proof.json` is the machine-readable result. -`bin/fm-test-run.sh` owns the production lane partition. +This document is the archived concurrent isolation proof for the portable parallel candidate set. +It is the human-readable companion to `bin/fm-test-isolation-proof.sh`. +Phase 4 production portable shards and bounded local `fm-test-run.sh --jobs` for this exact set are owned by `bin/fm-test-run.sh` and documented in [fm-test-portable-shards.md](fm-test-portable-shards.md). +The archived proof JSON below still records the Phase 2 proof-time flags (`production_sharding_enabled` / `fm_test_run_jobs_enabled` false at proof time). -## Verification +## Owner -- Date: 2026-07-29 -- Command: `bin/fm-test-isolation-proof.sh --jobs 4 --json /tmp/fm-source-content-test-cleanup-r1-isolation.json` -- Result: `FM_ISOLATION_SUMMARY total=24 failed=0 concurrency=4 duration_ms=149010` +- Harness: `bin/fm-test-isolation-proof.sh` +- Contract tests: `tests/fm-test-isolation-proof.test.sh` +- Family labels (Phase 1): `bin/fm-test-run.sh` +- Timing evidence used for planning: CI artifact `fm-test-timing` from Phase 1 PR #825 + +## Proof posture | Field | Value | |---|---| -| `run_id` | `fm-isolation-1785367157179-18165` | -| `started_at` | `2026-07-29T23:19:17Z` | -| `finished_at` | `2026-07-29T23:21:46Z` | -| concurrency | 4 | -| candidates | 24 | -| failed | 0 | -| wall duration | 149010 ms | +| `run_id` | `fm-isolation-1784693155237-99474` | +| `started_at` | `2026-07-22T04:05:55Z` | +| `finished_at` | `2026-07-22T04:08:06Z` | +| concurrency | **4** | +| candidates | **30** | +| failed | **0** | +| wall duration_ms | **131001** (~131.0s) | +| `production_sharding_enabled` | `False` | +| `fm_test_run_jobs_enabled` | `False` | +| host proof date | 2026-07-22 (UTC day of archive write) | + +Isolation checks that passed with this run: + +- Distinct mode-`0700` temporary roots per worker under a proof-owned parent +- Per-worker `TMPDIR`/`TMP` so `mktemp` / `fm_test_tmproot` stay private +- Ambient `FM_HOME` / `FM_*_OVERRIDE` cleared for each worker +- `git config --global` snapshot unchanged before/after the matrix +- Aggregate failure reporting (any non-zero candidate fails the harness; no retry-until-green) -## Candidate set +## Exact candidate set + +Sorted paths as selected by `bin/fm-test-isolation-proof.sh --list` at proof time: - `tests/fm-arm-pretool-check.test.sh` - `tests/fm-backend-herdr.test.sh` - `tests/fm-brief.test.sh` +- `tests/fm-captain-translation-contract.test.sh` - `tests/fm-cd-pretool-check.test.sh` - `tests/fm-composer-ghost.test.sh` - `tests/fm-composer-lib.test.sh` - `tests/fm-crew-state.test.sh` - `tests/fm-decision-hold-lifecycle.test.sh` +- `tests/fm-dispatch-select.test.sh` - `tests/fm-ensure-agents-md.test.sh` - `tests/fm-grok-harness.test.sh` - `tests/fm-herdr-lab.test.sh` +- `tests/fm-instruction-owners.test.sh` - `tests/fm-lint.test.sh` +- `tests/fm-nm-test-contract.test.sh` +- `tests/fm-no-mistakes-ownership.test.sh` - `tests/fm-pi-primary-types.test.sh` - `tests/fm-pr-merge.test.sh` - `tests/fm-review-diff.test.sh` @@ -41,51 +63,110 @@ This record is the concurrent isolation proof for the portable parallel candidat - `tests/fm-send-settle.test.sh` - `tests/fm-send-strict.test.sh` - `tests/fm-spawn-batch.test.sh` +- `tests/fm-stow-contract.test.sh` - `tests/fm-supervision-instructions.test.sh` - `tests/fm-test-run.test.sh` - `tests/fm-tmux-submit-busy.test.sh` - `tests/fm-transition-lib.test.sh` - `tests/fm-x-mode.test.sh` -## Durations +## Per-candidate durations (concurrent run) | duration_ms | exit | worker | script | |---:|---:|---:|---| -| 52939 | 0 | 24 | `tests/fm-x-mode.test.sh` | -| 48294 | 0 | 2 | `tests/fm-backend-herdr.test.sh` | -| 46788 | 0 | 1 | `tests/fm-arm-pretool-check.test.sh` | -| 34207 | 0 | 4 | `tests/fm-cd-pretool-check.test.sh` | -| 30771 | 0 | 8 | `tests/fm-decision-hold-lifecycle.test.sh` | -| 25365 | 0 | 7 | `tests/fm-crew-state.test.sh` | -| 15674 | 0 | 21 | `tests/fm-test-run.test.sh` | -| 15422 | 0 | 11 | `tests/fm-herdr-lab.test.sh` | -| 9065 | 0 | 5 | `tests/fm-composer-ghost.test.sh` | -| 8564 | 0 | 14 | `tests/fm-pr-merge.test.sh` | -| 6251 | 0 | 10 | `tests/fm-grok-harness.test.sh` | -| 5644 | 0 | 16 | `tests/fm-send-popup-settle.test.sh` | -| 5237 | 0 | 12 | `tests/fm-lint.test.sh` | -| 4816 | 0 | 22 | `tests/fm-tmux-submit-busy.test.sh` | -| 2945 | 0 | 13 | `tests/fm-pi-primary-types.test.sh` | -| 2911 | 0 | 17 | `tests/fm-send-settle.test.sh` | -| 2875 | 0 | 15 | `tests/fm-review-diff.test.sh` | -| 2747 | 0 | 18 | `tests/fm-send-strict.test.sh` | -| 2224 | 0 | 3 | `tests/fm-brief.test.sh` | -| 855 | 0 | 19 | `tests/fm-spawn-batch.test.sh` | -| 703 | 0 | 20 | `tests/fm-supervision-instructions.test.sh` | -| 581 | 0 | 9 | `tests/fm-ensure-agents-md.test.sh` | -| 248 | 0 | 23 | `tests/fm-transition-lib.test.sh` | -| 64 | 0 | 6 | `tests/fm-composer-lib.test.sh` | - -## Scope - -Each worker used a separate mode-`0700` temporary root and private `TMPDIR` and `TMP`. -The harness cleared ambient `FM_HOME` and `FM_*_OVERRIDE` values for every worker and verified that global Git configuration was unchanged. -A candidate failure fails the aggregate run and requires investigation rather than a retry. - -## Re-run +| 38449 | 0 | 30 | `tests/fm-x-mode.test.sh` | +| 35417 | 0 | 2 | `tests/fm-backend-herdr.test.sh` | +| 29102 | 0 | 1 | `tests/fm-arm-pretool-check.test.sh` | +| 21133 | 0 | 9 | `tests/fm-decision-hold-lifecycle.test.sh` | +| 19896 | 0 | 8 | `tests/fm-crew-state.test.sh` | +| 18610 | 0 | 5 | `tests/fm-cd-pretool-check.test.sh` | +| 12517 | 0 | 13 | `tests/fm-herdr-lab.test.sh` | +| 8939 | 0 | 19 | `tests/fm-pr-merge.test.sh` | +| 6953 | 0 | 21 | `tests/fm-send-popup-settle.test.sh` | +| 5963 | 0 | 12 | `tests/fm-grok-harness.test.sh` | +| 4645 | 0 | 27 | `tests/fm-test-run.test.sh` | +| 3524 | 0 | 22 | `tests/fm-send-settle.test.sh` | +| 2803 | 0 | 6 | `tests/fm-composer-ghost.test.sh` | +| 2552 | 0 | 28 | `tests/fm-tmux-submit-busy.test.sh` | +| 2549 | 0 | 20 | `tests/fm-review-diff.test.sh` | +| 1551 | 0 | 23 | `tests/fm-send-strict.test.sh` | +| 1274 | 0 | 15 | `tests/fm-lint.test.sh` | +| 1056 | 0 | 18 | `tests/fm-pi-primary-types.test.sh` | +| 897 | 0 | 3 | `tests/fm-brief.test.sh` | +| 874 | 0 | 10 | `tests/fm-dispatch-select.test.sh` | +| 684 | 0 | 24 | `tests/fm-spawn-batch.test.sh` | +| 348 | 0 | 11 | `tests/fm-ensure-agents-md.test.sh` | +| 283 | 0 | 26 | `tests/fm-supervision-instructions.test.sh` | +| 232 | 0 | 14 | `tests/fm-instruction-owners.test.sh` | +| 201 | 0 | 16 | `tests/fm-nm-test-contract.test.sh` | +| 104 | 0 | 29 | `tests/fm-transition-lib.test.sh` | +| 90 | 0 | 4 | `tests/fm-captain-translation-contract.test.sh` | +| 68 | 0 | 7 | `tests/fm-composer-lib.test.sh` | +| 57 | 0 | 25 | `tests/fm-stow-contract.test.sh` | +| 36 | 0 | 17 | `tests/fm-no-mistakes-ownership.test.sh` | + +## Audit notes (why this set) + +Source families from the Phase 1 manifest and scout report §3.1: + +1. **pure-contract-unit** candidates audited from the Phase 1 family manifest, minus deliberate serial exclusions +2. **Extra hermetic candidates** after static audit: fake backend, private git fixtures, stubbed network + +The harness pins this exact archived set and does not automatically admit later family additions. +A candidate-set change requires a new audit and concurrent proof archive. + +### Included extras (beyond pure-contract-unit) + +| Script | Why included | +|---|---| +| `tests/fm-backend-herdr.test.sh` | Fake Herdr CLI + private temps; no real Herdr binary | +| `tests/fm-send-strict.test.sh` | Fake tmux PATH shim; private `FM_HOME` | +| `tests/fm-spawn-batch.test.sh` | Argument routing only; no real windows/worktrees | +| `tests/fm-pr-merge.test.sh` | Fake `gh`/`gh-axi`; private state | +| `tests/fm-review-diff.test.sh` | Local git fixtures via `fm_git_*`; no live forge | +| `tests/fm-x-mode.test.sh` | Fake `curl`; inert without token | + +### Deliberately serial (kept out of this pool) + +Run `bin/fm-test-isolation-proof.sh --list-exclusions` for the machine-readable list. +High-signal classes: + +| Class | Examples | Reason | +|---|---|---| +| Watcher / wake / locks | `fm-watcher-lock`, `fm-wake-queue`, ... | Intentional process locks and daemon races | +| AFK | `fm-afk-inject-e2e`, ... | Daemon lifecycle and inject path | +| Real Herdr | `fm-backend-herdr-smoke`, presentation e2e, ... | Named labs, session-global locks; Herdr lane is Phase 3+ | +| Real tmux smoke | `fm-backend-tmux-smoke` | Real multiplexer server (even on private socket) | +| Live harness opt-in | `fm-*-live-e2e` | Real interactive agents | +| GUI backends | cmux smoke | Shared GUI app | +| Gray-zone git/spawn | `fm-backend`, spawn settle/profile, teardown | Heavier worktree or lock-race matrices | +| Watcher-adjacent forge security | `fm-pr-check-security` | `.watch.lock` / poll security surface | +| Self | `fm-test-isolation-proof.test.sh` | Must not re-enter the concurrent matrix | + +### Small isolation fix landed with this phase + +`tests/fm-arm-pretool-check.test.sh` no longer writes Claude deny stderr to a fixed `/tmp/fm-arm-pretool-check-claude-stderr.$$` path. +It uses `mktemp` under `TMPDIR` so concurrent workers cannot collide on a global temp name pattern. + +## Failures + +None. +Every candidate exited 0 under concurrency=4. + +Policy: a script that fails only under concurrency is **removed** from the candidate set and investigated. +It is never retried into green, skipped more broadly, or weakened in assertions. + +## What this phase did not do (Phase 2 scope) + +- Did not land production CI Behavior matrix / shard jobs (Phase 4) +- Did not add general `bin/fm-test-run.sh --jobs` (Phase 4 enables it only for this proven set) +- Did not land the Herdr install lane (Phase 3) +- Did not re-run the complete local suite as part of this proof (focused matrix only) + +## How to re-run ```sh bin/fm-test-isolation-proof.sh --list bin/fm-test-isolation-proof.sh --jobs 4 --json /tmp/fm-isolation-proof.json -bin/fm-test-run.sh --check-coverage +bash tests/fm-test-isolation-proof.test.sh ``` diff --git a/docs/subagent-guard.md b/docs/subagent-guard.md index 3bb39deb899..87f194d9d12 100644 --- a/docs/subagent-guard.md +++ b/docs/subagent-guard.md @@ -18,7 +18,7 @@ Three consequences were observed, not hypothesized. The deeper defect is that the bypass did not merely skip dispatch, it made the in-flight-work branch of the guard stack structurally inert. Only `bin/fm-spawn.sh` writes `state/<id>.meta`, so untracked project work contributes nothing to the in-flight count used by `bin/fm-supervision-lib.sh` and `bin/fm-turnend-guard.sh`. -Work started through the harness's own delegation tool writes no metadata, so the in-flight count stayed at zero, the turn-end guard never blocked a blind turn end, and the continuity gate was inert. +Work started through the harness's own delegation tool writes no metadata, so the in-flight count stayed at zero and the turn-end guard never blocked a blind turn end. That is the reason the fence has to sit on the harness tool surface, before the primary can create untracked work. No additional guard keyed on task metadata can catch this class of failure, because the failure is precisely the absence of that metadata. diff --git a/docs/supervision-protocols/claude.md b/docs/supervision-protocols/claude.md index 4e6ccb5fcaa..c9913553102 100644 --- a/docs/supervision-protocols/claude.md +++ b/docs/supervision-protocols/claude.md @@ -15,8 +15,9 @@ When this session owns supervision and away mode is not active: A shell `&`, a truncating pipe, or bundling is denied automatically by the PreToolUse seatbelt (`bin/fm-arm-pretool-check.sh`) registered in `.claude/settings.json`. 6. Treat `watcher: started ...` and `watcher: attached ...` inside arm output as proof that one live cycle exists. On attach, the arm follows verified identity-matched successors instead of exiting when the first cycle ends. -7. The continuity PreToolUse gate allows wake drain, watcher arm recovery, and fail-closed teardown, and refuses only other `bin/fm-*.sh` fleet commands while tasks are in flight and no identity-matched live watcher holds the home lock. - It covers the bounded gap between a rewake and the next Stop-launched arm. +7. The durable wake queue preserves actionable events between a rewake and the next Stop-launched arm, while the bounded turn-end guard prevents a blind Stop when recovery did not start. + No PreToolUse hook denies fleet commands based on watcher status. + [`watcher-continuity.md`](../watcher-continuity.md) owns the exact session-lock recovery boundary. 8. The turn-end guard (`bin/fm-turnend-guard.sh --claude`) remains the final backstop. It allows the stop when a watcher is healthy, when the auto-arm already owns recovery for this event epoch, or when a fresh rewake is recorded; it re-blocks only when none of those materialize, within a bounded budget. 9. Waiting on the hook-owned cycle is silent: do not send idle progress while the watcher is parked. diff --git a/docs/verification/supervision.md b/docs/verification/supervision.md index 2896cf7bf8f..4063f5566dc 100644 --- a/docs/verification/supervision.md +++ b/docs/verification/supervision.md @@ -65,7 +65,7 @@ The direct and passive mechanisms were validated across all five harnesses on 20 | Harness | Version verified | Mechanism | Observed result | | --- | --- | --- | --- | -| Claude | 2.1.219 | Cooperative blocking `Stop` guard plus `asyncRewake` auto-arm | Two tokenless auto-arm rewake cycles completed with no model arm command or guard continuation; deterministic coverage re-blocked genuine auto-arm failure despite `stop_hook_active=true`. | +| Claude | 2.1.219 | Cooperative blocking `Stop` guard plus `asyncRewake` auto-arm | A fresh unsupervised session ran session start first, reclaimed a stale dead-owner lock, completed two tokenless rewake cycles with no model arm command or guard continuation, and left a competing live owner unchanged. | | Codex | 0.142.1 | Blocking `Stop` hook | Hook process root stayed anchored to the trusted checkout and one continuation ran. | | OpenCode | 1.17.6 | Passive `session.idle` callback | Throwing could not block, while `promptAsync` scheduled one TUI follow-up; headless remained fail-open. | | Pi | 0.80.5 | Passive `agent_settled` callback | Exactly one guard follow-up ran for an unhealthy cycle, with no recursion across tool turns. | @@ -74,20 +74,18 @@ The direct and passive mechanisms were validated across all five harnesses on 20 The secondmate-home scope and manual-repair wake path were measured with Claude Code 2.1.207 on 2026-07-12, when a native background completion re-invoked the idle model with no human input. The current Stop-owned main/secondmate inclusion and child-worktree exclusion are covered deterministically by `tests/fm-claude-stop-autoarm.test.sh`. -The Claude product live paths ran with Claude Code 2.1.219 on 2026-07-24: +The Claude product live path ran with Claude Code 2.1.219 on 2026-07-24: ```sh claude --version FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-stop-autoarm-live-e2e.test.sh -FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-continuity-live-e2e.test.sh ``` Observed output: ```text 2.1.219 (Claude Code) -ok - Claude 2.1.219 (Claude Code) live E2E completed two tokenless Stop-owned auto-arm rewake cycles with zero model arm commands and no guard continuation -ok - Claude 2.1.219 (Claude Code) live E2E refused only the post-completion fleet command with exact re-arm guidance +ok - Claude 2.1.219 (Claude Code) live E2E reclaimed a stale session lock through session start, completed two tokenless Stop-owned rewake cycles, and preserved the competing-live-owner boundary ``` Current entry points: @@ -113,7 +111,7 @@ grok 0.2.103 (89c3d36fb6f1) [stable] | Harness | Exact opt-in command | Observed guarantee | | --- | --- | --- | -| Claude | `FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-stop-autoarm-live-e2e.test.sh` | Two Stop-owned cycles re-armed and rewoke without a model arm command or guard continuation. | +| Claude | `FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-stop-autoarm-live-e2e.test.sh` | Session start reclaimed a stale owner before two Stop-owned cycles, and a competing live owner prevented arm, rewake, epoch write, or lock replacement. | | Codex | `FM_CODEX_LIVE_E2E=1 tests/fm-codex-continuity-live-e2e.test.sh` | The one-second foreground checkpoint returned without switching to the arm wrapper. | | OpenCode | `FM_OPENCODE_LIVE_E2E=1 tests/fm-opencode-primary-live-e2e.test.sh` | A verified successor existed before prompt handling, with no model re-arm or turn-end fallback. | | Pi | `FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh` | One initial tool call led to extension-owned successors and clean child retirement on exit. | @@ -126,7 +124,7 @@ Deterministic entry points: ```sh tests/fm-pi-watch-extension.test.sh tests/fm-watcher-lock.test.sh -tests/fm-continuity-pretool-check.test.sh +tests/fm-subagent-pretool-check.test.sh tests/fm-claude-stop-autoarm.test.sh tests/fm-turnend-guard.test.sh ``` diff --git a/docs/watcher-continuity.md b/docs/watcher-continuity.md index 9e1dabd6bb7..457c035268d 100644 --- a/docs/watcher-continuity.md +++ b/docs/watcher-continuity.md @@ -10,6 +10,8 @@ Each adapter starts the next arm before delivering the wake prompt, checks curre A failed follow-up never cancels continuity restoration. Claude's `.claude/settings.json` Stop `asyncRewake` hook (`bin/fm-claude-stop-autoarm.sh`) owns routine tokenless re-arm. The hook fires on every Stop, and an eligible primary with supervision need admits one home-scoped owner that foregrounds `bin/fm-watch-arm.sh` inside the hook-owned process tree. +A numeric session-lock owner that fails the shared `fm_harness_pid_alive` predicate is reclaimed through `bin/fm-lock.sh` before auto-arm state changes, while a live owner, absent lock, or malformed lock keeps the competing hook inert. +The stale-owner claim occurs only after the existing AFK and supervision-need gates pass. While supervision is still needed and away mode remains inactive, an actionable close or typed failure wakes the idle session through exit 2. ## Actionable wake ordering @@ -22,9 +24,8 @@ After the configured retry bound is exhausted, it delivers the original wake wit This is deliberate Option B ordering: the fleet is protected before the model handles the wake whenever restoration succeeds, but the model is never left blind when it does not. Claude's Stop hook starts the successor arm at the next Stop after the handling turn, rather than before notification as Pi and OpenCode do. -The durable wake queue and the PreToolUse continuity gate cover the residual active-turn window. -The gate allows wake drain, arm recovery, and independently fail-closed teardown, but refuses other fleet commands while tasks are in flight and no identity-matched live watcher holds the home lock. -Allowing an ordinary literal teardown prevents a terminal wake from creating a recovery circle: forced or dynamically constructed teardown remains blocked, ordinary teardown itself still refuses dirty, unlanded, incomplete-scout, and unresolved-decision cases, and the turn-end guard continues to require supervision for any tasks left in flight. +The durable wake queue preserves actionable events during the residual active-turn window, and the unchanged bounded turn-end guard enforces recovery at Stop when no watcher or auto-arm claim is present. +No PreToolUse hook denies fleet commands based on watcher status. The model no longer re-arms after ordinary wakes. Terminal arm-output classification (`started`, `attached`, or `FAILED`) remains defense in depth for the manual recovery path. Codex retains its bounded foreground checkpoint protocol. @@ -52,8 +53,9 @@ Only the watcher process touches `state/.last-watcher-beat`; no helper process c `tests/fm-pi-watch-extension.test.sh` checks Pi's first-cycle-or-explicit-repair tool metadata and ownership-based redundant-call no-ops, then simulates actionable and empty child closes against the actual Pi and OpenCode close handlers, blocks prompt delivery to prove the successor launches first, verifies single-flight behavior, changes the session lock before close to prove ownership is rechecked, and hangs each successor arm to prove bounded fallback delivery includes the typed restoration failure. `tests/fm-watcher-lock.test.sh` covers verified-successor attach, the typed self-eviction failure, bounded and successor-linked lifecycle rows, and a SIGSTOP counterfactual that distinguishes a live PID from a stale beacon before classifying termination. -`tests/fm-continuity-pretool-check.test.sh` proves the Claude gate rejects only non-recovery fleet execution in the precise unhealthy state and preserves the registered Stop hooks. -`tests/fm-claude-stop-autoarm.test.sh` covers the auto-arm's scope, identity, AFK, need, single-flight, and exit-2 translation. +`tests/fm-subagent-pretool-check.test.sh` proves Claude retains only the non-status Bash seatbelts. +`tests/fm-claude-stop-autoarm.test.sh` covers the auto-arm's scope, stale and live session owners, unchanged AFK and need boundaries, single-flight, and exit-2 translation. +`FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-stop-autoarm-live-e2e.test.sh` starts with the reproduced stale-lock state, runs session start first, completes two tokenless cycles, and checks the competing-live-owner negative control. `tests/fm-turnend-guard.test.sh` covers the cooperative `--claude` guard. ## Active limits and verification diff --git a/tests/fm-claude-continuity-live-e2e.test.sh b/tests/fm-claude-continuity-live-e2e.test.sh deleted file mode 100755 index 67030d5771a..00000000000 --- a/tests/fm-claude-continuity-live-e2e.test.sh +++ /dev/null @@ -1,73 +0,0 @@ -#!/usr/bin/env bash -# Opt-in credentialed Claude regression for the post-background-completion -# continuity gate. The project and FM_HOME are isolated; Claude keeps using its -# existing managed authentication. -set -u - -if [ "${FM_CLAUDE_LIVE_E2E:-0}" != 1 ]; then - echo "skip: set FM_CLAUDE_LIVE_E2E=1 to run the Claude continuity regression" - exit 0 -fi - -ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" - -fail() { - printf 'not ok - %s\n' "$1" >&2 - exit 1 -} - -command -v claude >/dev/null 2>&1 || fail "claude not found" - -LAB="$ROOT/.claude-live-e2e.$$" -PROJECT="$LAB/project" -HOME_DIR="$LAB/fmhome" -TRANSCRIPT="$LAB/claude.jsonl" -CLAUDE_VERSION=$(claude --version) - -cleanup() { - rm -rf "$LAB" -} -trap cleanup EXIT - -mkdir -p "$LAB" -git clone -q "$ROOT" "$PROJECT" -cp "$ROOT/.claude/settings.json" "$PROJECT/.claude/settings.json" -cp -R "$ROOT/bin/." "$PROJECT/bin/" -mkdir -p "$HOME_DIR/state" "$HOME_DIR/config" -printf 'project=fixture\n' > "$HOME_DIR/state/claude-e2e.meta" - -cat > "$PROJECT/bin/fm-watch-arm.sh" <<'SH' -#!/usr/bin/env bash -printf 'started\n' > "$FM_HOME/state/claude-arm-ran" -printf 'watcher: started pid=%s (fixture)\n' "$$" -sleep 0.2 -printf 'signal: fixture background completion\n' -SH -cat > "$PROJECT/bin/fm-wake-drain.sh" <<'SH' -#!/usr/bin/env bash -printf 'drained\n' > "$FM_HOME/state/claude-drain-ran" -printf 'signal: fixture background completion\n' -SH -cat > "$PROJECT/bin/fm-crew-state.sh" <<'SH' -#!/usr/bin/env bash -printf 'forbidden\n' > "$FM_HOME/state/claude-forbidden-ran" -printf 'crew state should not run\n' -SH -chmod +x "$PROJECT/bin/fm-watch-arm.sh" "$PROJECT/bin/fm-wake-drain.sh" "$PROJECT/bin/fm-crew-state.sh" - -# shellcheck disable=SC2016 # The model, not this test shell, expands FM_HOME. -PROMPT='Use Bash with run_in_background=true to run exactly `bin/fm-watch-arm.sh`. Wait for its background-task completion. Then run exactly `bin/fm-wake-drain.sh`. Without re-arming, next attempt exactly `bin/fm-crew-state.sh claude-e2e`. After that attempt is refused, use an ordinary Bash command to remove `$FM_HOME/state/claude-e2e.meta`, then reply briefly. Do not retry the refused fleet command and do not re-arm.' - -( - cd "$PROJECT" || exit 1 - FM_HOME="$HOME_DIR" FM_ROOT_OVERRIDE="$PROJECT" CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false \ - claude -p "$PROMPT" --dangerously-skip-permissions --effort low --output-format stream-json --verbose -) > "$TRANSCRIPT" 2>&1 || fail "Claude credentialed continuity turn failed: $(tail -20 "$TRANSCRIPT")" - -[ -f "$HOME_DIR/state/claude-arm-ran" ] || fail "Claude did not run the tracked background arm fixture" -[ -f "$HOME_DIR/state/claude-drain-ran" ] || fail "Claude continuity gate blocked the allowed wake drain" -[ ! -f "$HOME_DIR/state/claude-forbidden-ran" ] || fail "Claude continuity gate allowed an unrelated fleet command" -GUIDANCE='[watcher-continuity] tasks are in flight and no live watcher holds this home lock; drain wakes with bin/fm-wake-drain.sh, use fail-closed bin/fm-teardown.sh for completed tasks when needed, then end the turn so the Stop-owned auto-arm re-establishes the watcher; if the Stop auto-arm itself failed, re-arm manually with bin/fm-watch-arm.sh as a tracked Claude background task (blocked: fm-crew-state.sh)' -grep -F "$GUIDANCE" "$TRANSCRIPT" >/dev/null || fail "Claude transcript omitted the exact continuity recovery guidance" - -printf 'ok - Claude %s live E2E refused only the post-completion fleet command with exact re-arm guidance\n' "$CLAUDE_VERSION" diff --git a/tests/fm-claude-stop-autoarm-live-e2e.test.sh b/tests/fm-claude-stop-autoarm-live-e2e.test.sh index 74cf8b33699..c7e2cab880b 100755 --- a/tests/fm-claude-stop-autoarm-live-e2e.test.sh +++ b/tests/fm-claude-stop-autoarm-live-e2e.test.sh @@ -2,10 +2,11 @@ # Opt-in credentialed Claude live regression for the Stop-owned auto-arm # (bin/fm-claude-stop-autoarm.sh + bin/fm-turnend-guard.sh --claude). # Proves, against the real installed Claude Code and the real tracked hook -# registration: at least two complete tokenless auto-arm and rewake cycles with -# zero model-issued arm commands, the rapid started-plus-immediate-actionable -# shape closing without a multi-hour blind window, and the cooperative guard -# consuming no forced continuation while the hook's launch is healthy. +# registration: a fresh session with in-flight work, no watcher, and a stale +# session lock can run fm-session-start.sh first; session start reclaims the +# dead owner; at least two tokenless auto-arm and rewake cycles then complete +# with zero model-issued arm commands; and the cooperative guard consumes no +# forced continuation while the hook's launch is healthy. # The project and FM_HOME are isolated; Claude keeps using its existing managed # authentication. No live fleet home, worktree, or session is touched. # shellcheck disable=SC2016 # the model, not this test shell, reads the prompt text @@ -28,6 +29,7 @@ command -v claude >/dev/null 2>&1 || fail "claude not found" LAB="$ROOT/.claude-autoarm-live-e2e.$$" PROJECT="$LAB/project" HOME_DIR="$LAB/fmhome" +LIVE_OWNER_HOME="$LAB/live-owner-home" TRANSCRIPT="$LAB/claude.jsonl" CLAUDE_VERSION=$(claude --version) @@ -42,21 +44,13 @@ mkdir -p "$LAB" git clone -q "$ROOT" "$PROJECT" cp -R "$ROOT/bin/." "$PROJECT/bin/" cp "$ROOT/.claude/settings.json" "$PROJECT/.claude/settings.json" -# The lab keeps the real tracked .claude/settings.json Stop registration -# (guard --claude + asyncRewake auto-arm, timeout 28800) and adds a local -# SessionStart hook that acquires the fixture home's session lock exactly the -# way bin/fm-session-start.sh does in production, plus a PreToolUse recorder. +# The lab keeps the real tracked .claude/settings.json SessionStart nudge, +# Stop guard, and asyncRewake auto-arm registration. +# The only local hook records model-issued Bash calls without acquiring the +# session lock or otherwise changing lifecycle behavior. cat > "$PROJECT/.claude/settings.local.json" <<'JSON' { "hooks": { - "SessionStart": [ - { - "matcher": "startup", - "hooks": [ - { "type": "command", "command": "\"$CLAUDE_PROJECT_DIR\"/bin/fm-lock.sh" } - ] - } - ], "PreToolUse": [ { "matcher": "Bash", @@ -77,8 +71,11 @@ exit 0 SH chmod +x "$PROJECT/bin/tool-logger.sh" -mkdir -p "$HOME_DIR/state" "$HOME_DIR/config" -printf 'project=fixture\n' > "$HOME_DIR/state/task.meta" +mkdir -p "$HOME_DIR/state" "$HOME_DIR/config" "$HOME_DIR/data" +printf 'project=fixture\nwindow=fixture\nbackend=tmux\n' > "$HOME_DIR/state/task.meta" +# A numeric pid above the supported OS pid range is a demonstrably dead prior +# harness owner under fm_harness_pid_alive, matching the reproduced incident. +printf '9999999\n' > "$HOME_DIR/state/.lock" # Rapid-death arm fixture: started plus an immediate actionable reason, the # exact spent-Stop edge shape. Runs 1-2 close actionable; run 3 closes clean so @@ -96,20 +93,21 @@ printf 'watcher: started pid=%s (beacon fresh)\n' "$$" printf 'stale: fixture-rapid-%s\n' "$N" exit 0 SH -# Drain fixture: the model's only allowed tool call; the second drain ends the -# in-flight need so the session can settle after two full rewake cycles. +# Drain fixture: session start invokes it once, then the model invokes it once +# per rewake. The third total drain ends the in-flight need after two complete +# Stop-owned cycles. cat > "$PROJECT/bin/fm-wake-drain.sh" <<'SH' #!/usr/bin/env bash N=$(cat "$FM_HOME/state/drain-count" 2>/dev/null || echo 0); N=$((N+1)); echo "$N" > "$FM_HOME/state/drain-count" echo "drain-run=$N" >> "$FM_HOME/state/drain-ran" -if [ "$N" -ge 2 ]; then +if [ "$N" -ge 3 ]; then rm -f "$FM_HOME/state/task.meta" fi printf 'stale: fixture-rapid drained\n' SH chmod +x "$PROJECT/bin/fm-watch-arm.sh" "$PROJECT/bin/fm-wake-drain.sh" -PROMPT='Reply with exactly CYCLE0 and stop. Whenever a Stop hook feedback message wakes you, run exactly `bin/fm-wake-drain.sh` once with Bash, then reply with exactly ACK and stop. Never run bin/fm-watch-arm.sh or any other arm command, and never use any other tool.' +PROMPT='Run exactly `bin/fm-session-start.sh` with Bash as your first tool call. After reading its complete digest, reply with exactly CYCLE0 and stop. Whenever a Stop hook feedback message wakes you, run exactly `bin/fm-wake-drain.sh` once with Bash, then reply with exactly ACK and stop. Never run bin/fm-watch-arm.sh or any other arm command, and never use any other tool.' ( cd "$PROJECT" || exit 1 @@ -120,11 +118,15 @@ PROMPT='Reply with exactly CYCLE0 and stop. Whenever a Stop hook feedback messag ARM_RUNS=$(wc -l < "$HOME_DIR/state/arm-ran" 2>/dev/null | tr -d ' ') [ "$ARM_RUNS" = 2 ] || fail "expected exactly 2 hook-owned arm cycles, got $ARM_RUNS: $(cat "$HOME_DIR/state/arm-ran" 2>/dev/null)" DRAIN_RUNS=$(wc -l < "$HOME_DIR/state/drain-ran" 2>/dev/null | tr -d ' ') -[ "$DRAIN_RUNS" = 2 ] || fail "expected the model to drain both wakes, got $DRAIN_RUNS drains" +[ "$DRAIN_RUNS" = 3 ] || fail "expected one session-start drain plus two model wake drains, got $DRAIN_RUNS drains" REWAKES=$(grep -c 'Stop hook feedback' "$TRANSCRIPT" 2>/dev/null || true) [ "$REWAKES" -ge 2 ] || fail "expected at least 2 exit-2 rewake deliveries, got $REWAKES" grep -q 'stale: fixture-rapid-1' "$TRANSCRIPT" || fail "first rapid rewake reason missing from the transcript" grep -q 'stale: fixture-rapid-2' "$TRANSCRIPT" || fail "second rapid rewake reason missing from the transcript" +[ "$(sed -n '1p' "$HOME_DIR/state/tool-calls.log" 2>/dev/null)" = 'bin/fm-session-start.sh' ] \ + || fail "fresh Claude session did not run session start first: $(cat "$HOME_DIR/state/tool-calls.log" 2>/dev/null)" +[ "$(cat "$HOME_DIR/state/.lock" 2>/dev/null)" != 9999999 ] \ + || fail "session start did not reclaim the stale dead-owner lock" if [ -f "$HOME_DIR/state/tool-calls.log" ]; then ! grep -q 'fm-watch-arm.sh' "$HOME_DIR/state/tool-calls.log" \ || fail "model issued an arm command despite Stop-owned continuity: $(cat "$HOME_DIR/state/tool-calls.log")" @@ -137,4 +139,26 @@ fi || fail "auto-arm epoch ledger must record the rewake outcome" [ ! -e "$HOME_DIR/state/.claude-autoarm.lock" ] || fail "auto-arm owner lock was left behind" -printf 'ok - Claude %s live E2E completed two tokenless Stop-owned auto-arm rewake cycles with zero model arm commands and no guard continuation\n' "$CLAUDE_VERSION" +# Live-owner negative control: a separate supported-harness process owns a +# second isolated home while another Stop hook fires from the same primary +# project. The competing hook must not replace the session lock, arm, write an +# epoch, or rewake. +FAKE_CLAUDE="$LAB/claude" +ln -s /bin/bash "$FAKE_CLAUDE" +mkdir -p "$LIVE_OWNER_HOME/state" "$LIVE_OWNER_HOME/config" +printf 'project=fixture\n' > "$LIVE_OWNER_HOME/state/task.meta" +"$FAKE_CLAUDE" -c 'sleep 3; :' & +LIVE_OWNER_PID=$! +printf '%s\n' "$LIVE_OWNER_PID" > "$LIVE_OWNER_HOME/state/.lock" +LIVE_OWNER_RC=0 +printf '%s\n' '{"session_id":"live-owner-control"}' \ + | FM_HOME="$LIVE_OWNER_HOME" FM_ROOT_OVERRIDE="$PROJECT" "$FAKE_CLAUDE" -c '"$FM_ROOT_OVERRIDE/bin/fm-claude-stop-autoarm.sh"' \ + >"$LAB/live-owner.out" 2>"$LAB/live-owner.err" || LIVE_OWNER_RC=$? +[ "$LIVE_OWNER_RC" -eq 0 ] || fail "competing Stop hook returned $LIVE_OWNER_RC while another live session owned the home" +[ "$(cat "$LIVE_OWNER_HOME/state/.lock")" = "$LIVE_OWNER_PID" ] || fail "competing Stop hook replaced the live session owner" +[ ! -e "$LIVE_OWNER_HOME/state/arm-ran" ] || fail "competing Stop hook armed while another live session owned the home" +[ ! -e "$LIVE_OWNER_HOME/state/.claude-autoarm-epoch" ] || fail "competing Stop hook wrote an epoch while another live session owned the home" +[ ! -s "$LAB/live-owner.out" ] && [ ! -s "$LAB/live-owner.err" ] || fail "competing Stop hook produced a rewake while another live session owned the home" +wait "$LIVE_OWNER_PID" + +printf 'ok - Claude %s live E2E reclaimed a stale session lock through session start, completed two tokenless Stop-owned rewake cycles, and preserved the competing-live-owner boundary\n' "$CLAUDE_VERSION" diff --git a/tests/fm-claude-stop-autoarm.test.sh b/tests/fm-claude-stop-autoarm.test.sh index 8dc11e448a2..63ddb8a7b13 100755 --- a/tests/fm-claude-stop-autoarm.test.sh +++ b/tests/fm-claude-stop-autoarm.test.sh @@ -4,9 +4,10 @@ # # The hook fires as a Claude asyncRewake Stop hook. These tests run it hermetically # as a child of a fake harness (a bash symlink named "claude") whose pid is -# written into the fixture home's state/.lock, which satisfies the hook's -# session-identity gate exactly the way production does. The arm wrapper is a -# per-test fixture, so no real watcher, model, or fleet state is touched. +# written into the fixture home's state/.lock for ordinary owned-lock cases. +# Stale-owner cases instead leave a dead recorded pid for the hook to reclaim +# through the real fm-lock.sh path. The arm wrapper is a per-test fixture, so no +# real watcher, model, or fleet state is touched. # shellcheck disable=SC2016 # single quotes are deliberate: $FM_HOME expands inside the fake harness child, and grep needles are literal strings set -u @@ -29,7 +30,8 @@ install_autoarm_scripts() { cp "$ROOT/bin/fm-supervision-lib.sh" "$dir/bin/fm-supervision-lib.sh" cp "$ROOT/bin/fm-wake-lib.sh" "$dir/bin/fm-wake-lib.sh" cp "$ROOT/bin/fm-session-lock-lib.sh" "$dir/bin/fm-session-lock-lib.sh" - chmod +x "$dir/bin/fm-claude-stop-autoarm.sh" + cp "$ROOT/bin/fm-lock.sh" "$dir/bin/fm-lock.sh" + chmod +x "$dir/bin/fm-claude-stop-autoarm.sh" "$dir/bin/fm-lock.sh" } make_primary_dir() { @@ -197,21 +199,45 @@ test_inert_without_session_lock() { pass "auto-arm: inert with no session lock" } +test_reclaims_stale_session_lock_before_arming() { + local dir out status expected_owner actual_owner + dir=$(make_primary_dir "$TMP_ROOT/stale-lock") + : > "$dir/state/task.meta" + printf '9999999\n' > "$dir/state/.lock" + write_arm_fixture "$dir" actionable + out=$(printf '%s\n' '{"session_id":"stale"}' \ + | FM_HOME="$dir" "$FAKE_CLAUDE" -c ' + printf "%s\n" "$$" > "$FM_HOME/state/expected-owner" + "$FM_HOME/bin/fm-claude-stop-autoarm.sh" + ' 2>&1); status=$? + expect_code 2 "$status" "a dead recorded session owner must be reclaimed before the actionable rewake" + expected_owner=$(cat "$dir/state/expected-owner") + actual_owner=$(cat "$dir/state/.lock") + [ "$actual_owner" = "$expected_owner" ] || fail "stale session lock was not claimed by the current harness: expected $expected_owner, got $actual_owner" + [ -e "$dir/state/arm-ran" ] || fail "hook did not arm after reclaiming the stale session lock" + [ "$(epoch_outcome "$dir")" = rewake ] || fail "stale-lock recovery must record outcome=rewake" + pass "auto-arm: a demonstrably dead recorded session owner is reclaimed through fm-lock.sh before arming" +} + test_inert_when_lock_held_by_other_harness() { - local dir other out status + local dir other out status owner_after dir=$(make_primary_dir "$TMP_ROOT/other-lock") : > "$dir/state/task.meta" write_arm_fixture "$dir" actionable - # Another live harness holds the lock; our hook runs under a different fake claude. - "$FAKE_CLAUDE" -c 'sleep 60' & + # The trailing no-op keeps the fake harness process alive instead of allowing + # bash to exec the final sleep into a non-harness process. + "$FAKE_CLAUDE" -c 'sleep 60; :' & other=$! printf '%s\n' "$other" > "$dir/state/.lock" out=$(printf '%s\n' '{"session_id":"s"}' | FM_HOME="$dir" "$FAKE_CLAUDE" -c '"$FM_HOME/bin/fm-claude-stop-autoarm.sh"' 2>&1); status=$? + owner_after=$(cat "$dir/state/.lock") kill "$other" 2>/dev/null || true wait "$other" 2>/dev/null || true expect_code 0 "$status" "hook must stay inert when another live harness holds the session lock" + [ "$owner_after" = "$other" ] || fail "hook replaced another live harness owner: expected $other, got $owner_after" [ ! -e "$dir/state/arm-ran" ] || fail "hook armed while another session owned the lock" - pass "auto-arm: inert when the session lock belongs to another live harness" + [ ! -e "$dir/state/.claude-autoarm-epoch" ] || fail "hook wrote an epoch while another session owned the lock" + pass "auto-arm: inert without arm, rewake, or lock replacement when another live harness owns the home" } test_inert_when_afk() { @@ -226,6 +252,28 @@ test_inert_when_afk() { pass "auto-arm: inert while AFK owns supervision" } +test_stale_lock_recovery_preserves_afk_and_need_gates() { + local afk_dir idle_dir out status + afk_dir=$(make_primary_dir "$TMP_ROOT/stale-afk") + : > "$afk_dir/state/task.meta" + : > "$afk_dir/state/.afk" + printf '9999999\n' > "$afk_dir/state/.lock" + write_arm_fixture "$afk_dir" actionable + out=$(printf '%s\n' '{"session_id":"stale-afk"}' | FM_HOME="$afk_dir" "$FAKE_CLAUDE" -c '"$FM_HOME/bin/fm-claude-stop-autoarm.sh"' 2>&1); status=$? + expect_code 0 "$status" "a stale owner must not widen the AFK gate" + [ "$(cat "$afk_dir/state/.lock")" = 9999999 ] || fail "AFK stale lock was reclaimed despite away ownership" + [ ! -e "$afk_dir/state/arm-ran" ] || fail "stale AFK home armed" + + idle_dir=$(make_primary_dir "$TMP_ROOT/stale-idle") + printf '9999999\n' > "$idle_dir/state/.lock" + write_arm_fixture "$idle_dir" actionable + out=$(printf '%s\n' '{"session_id":"stale-idle"}' | FM_HOME="$idle_dir" "$FAKE_CLAUDE" -c '"$FM_HOME/bin/fm-claude-stop-autoarm.sh"' 2>&1); status=$? + expect_code 0 "$status" "a stale owner must not widen the supervision-need gate" + [ "$(cat "$idle_dir/state/.lock")" = 9999999 ] || fail "idle stale lock was reclaimed without supervision need" + [ ! -e "$idle_dir/state/arm-ran" ] || fail "stale idle home armed" + pass "auto-arm: stale-owner recovery leaves the AFK and supervision-need gates unchanged" +} + test_inert_when_fleet_idle() { local dir out status dir=$(make_primary_dir "$TMP_ROOT/idle") @@ -361,8 +409,10 @@ test_fm_lock_status_still_works_with_shared_lib() { test_settings_registers_autoarm_with_multi_hour_timeout test_inert_in_child_worktree test_inert_without_session_lock +test_reclaims_stale_session_lock_before_arming test_inert_when_lock_held_by_other_harness test_inert_when_afk +test_stale_lock_recovery_preserves_afk_and_need_gates test_inert_when_fleet_idle test_actionable_close_rewakes_with_reason test_failed_close_rewakes_with_failure_banner diff --git a/tests/fm-continuity-pretool-check.test.sh b/tests/fm-continuity-pretool-check.test.sh deleted file mode 100755 index 27b5457eed6..00000000000 --- a/tests/fm-continuity-pretool-check.test.sh +++ /dev/null @@ -1,128 +0,0 @@ -#!/usr/bin/env bash -# Behavior tests for Claude's narrowly scoped watcher-continuity PreToolUse gate. -set -u - -# shellcheck source=tests/lib.sh -. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" - -CHECK="$ROOT/bin/fm-continuity-pretool-check.sh" -WATCH="$ROOT/bin/fm-watch.sh" -TMP_ROOT=$(fm_test_tmproot fm-continuity-pretool-tests) -PRIMARY="$TMP_ROOT/primary" -STATE="$PRIMARY/state" -OUT="$TMP_ROOT/out" -ERR="$TMP_ROOT/err" - -mkdir -p "$PRIMARY/bin" "$STATE" -printf '# fixture\n' > "$PRIMARY/AGENTS.md" -git -C "$PRIMARY" init -q - -run_command() { - local command=$1 rc=0 - : > "$OUT" - : > "$ERR" - FM_ROOT_OVERRIDE="$PRIMARY" FM_HOME="$PRIMARY" FM_STATE_OVERRIDE="$STATE" \ - "$CHECK" --command "$command" > "$OUT" 2> "$ERR" || rc=$? - return "$rc" -} - -expect_allow() { - local label=$1 command=$2 rc=0 - run_command "$command" || rc=$? - [ "$rc" -eq 0 ] || fail "$label must allow, got exit $rc: $(cat "$ERR")" - [ ! -s "$OUT" ] || fail "$label allow wrote stdout: $(cat "$OUT")" - [ ! -s "$ERR" ] || fail "$label allow wrote stderr: $(cat "$ERR")" -} - -expect_deny() { - local label=$1 command=$2 blocked=$3 expected=${4:-} rc=0 actual - run_command "$command" || rc=$? - [ "$rc" -eq 2 ] || fail "$label must deny with exit 2, got $rc" - [ ! -s "$OUT" ] || fail "$label deny wrote stdout: $(cat "$OUT")" - jq -e '.hookSpecificOutput.hookEventName == "PreToolUse" and .hookSpecificOutput.permissionDecision == "deny"' "$ERR" >/dev/null 2>&1 \ - || fail "$label deny omitted Claude's permission decision: $(cat "$ERR")" - [ -n "$expected" ] || expected="[watcher-continuity] tasks are in flight and no live watcher holds this home lock; drain wakes with bin/fm-wake-drain.sh, use fail-closed bin/fm-teardown.sh for completed tasks when needed, then end the turn so the Stop-owned auto-arm re-establishes the watcher; if the Stop auto-arm itself failed, re-arm manually with bin/fm-watch-arm.sh as a tracked Claude background task (blocked: $blocked)" - actual=$(jq -r '.systemMessage' "$ERR") - [ "$actual" = "$expected" ] || fail "$label recovery guidance changed: $actual" -} - -test_gate_scope_and_recovery_exceptions() { - expect_allow "idle fleet command" 'bin/fm-crew-state.sh task' - printf 'project=fixture\n' > "$STATE/task.meta" - - expect_allow "ordinary shell command" 'git status --short' - expect_allow "fleet-script text as data" "rg -n 'bin/fm-send.sh' docs" - expect_allow "wake drain recovery" 'bin/fm-wake-drain.sh' - expect_allow "watch arm recovery" 'bin/fm-watch-arm.sh' - expect_allow "drain then arm recovery" 'bin/fm-wake-drain.sh; bin/fm-watch-arm.sh' - expect_allow "fail-closed teardown recovery" 'bin/fm-teardown.sh task' - unsafe_teardown_reason='[watcher-continuity] tasks are in flight and no live watcher holds this home lock; during recovery only the ordinary literal bin/fm-teardown.sh is allowed, so drop --force and any shell-expanded arguments and retry the literal invocation (blocked: fm-teardown.sh)' - expect_deny "forced teardown is not recovery" 'bin/fm-teardown.sh task --force' 'fm-teardown.sh' "$unsafe_teardown_reason" - expect_deny "nested forced teardown is not recovery" "bash -lc 'bin/fm-teardown.sh task --force'" 'fm-teardown.sh' "$unsafe_teardown_reason" - # shellcheck disable=SC2016 # single quotes are deliberate: "$TEARDOWN_MODE" is literal test data (an unsafe shell-expanded arg the gate must deny), not an expansion here - expect_deny "dynamic teardown mode is not recovery" 'bin/fm-teardown.sh task "$TEARDOWN_MODE"' 'fm-teardown.sh' "$unsafe_teardown_reason" - expect_deny "unrelated fleet command" 'bin/fm-crew-state.sh task' 'fm-crew-state.sh' - expect_deny "recovery bundled with unrelated fleet command" 'bin/fm-wake-drain.sh; bin/fm-send.sh task hi' 'fm-send.sh' - expect_deny "literal nested fleet command" "bash -lc 'bin/fm-bootstrap.sh'" 'fm-bootstrap.sh' - pass "continuity gate allows recovery and ordinary commands but denies only other fleet execution" -} - -test_live_lock_allows_fleet_command_even_with_stale_beacon() { - local holder identity rc=0 - sleep 300 & - holder=$! - identity=$(FM_STATE_OVERRIDE="$STATE" bash -c '. "$1"; fm_pid_identity "$2"' _ "$ROOT/bin/fm-wake-lib.sh" "$holder") \ - || fail "could not identify live continuity fixture" - mkdir -p "$STATE/.watch.lock" - printf '%s\n' "$holder" > "$STATE/.watch.lock/pid" - printf '%s\n' "$PRIMARY" > "$STATE/.watch.lock/fm-home" - printf '%s\n' "$WATCH" > "$STATE/.watch.lock/watcher-path" - printf '%s\n' "$identity" > "$STATE/.watch.lock/pid-identity" - touch -t 200001010000 "$STATE/.last-watcher-beat" - - run_command 'bin/fm-crew-state.sh task' || rc=$? - kill "$holder" 2>/dev/null || true - wait "$holder" 2>/dev/null || true - [ "$rc" -eq 0 ] || fail "identity-matched live lock must allow fleet command even when its beacon is stale" - [ ! -s "$ERR" ] || fail "live-lock allow wrote stderr: $(cat "$ERR")" - pass "continuity gate classifies the lock by live PID identity rather than beacon age" -} - -test_child_worktree_and_malformed_input_fail_open() { - local child="$TMP_ROOT/child" rc=0 - rm -rf "$STATE/.watch.lock" - git -C "$PRIMARY" config user.name fixture - git -C "$PRIMARY" config user.email fixture@example.test - git -C "$PRIMARY" add AGENTS.md - git -C "$PRIMARY" commit -qm fixture - git -C "$PRIMARY" worktree add -q -b fixture-child "$child" - mkdir -p "$child/bin" "$child/state" - FM_ROOT_OVERRIDE="$child" FM_HOME="$child" FM_STATE_OVERRIDE="$child/state" \ - "$CHECK" --command 'bin/fm-send.sh task hi' > "$OUT" 2> "$ERR" || rc=$? - [ "$rc" -eq 0 ] || fail "linked child worktree must be out of continuity-gate scope" - - expect_allow "malformed dynamic shell" "bin/fm-send.sh 'unterminated" - printf '%s' '{not-json' | FM_ROOT_OVERRIDE="$PRIMARY" FM_HOME="$PRIMARY" FM_STATE_OVERRIDE="$STATE" \ - "$CHECK" > "$OUT" 2> "$ERR" || rc=$? - [ "$rc" -eq 0 ] || fail "malformed Claude transport must fail open" - pass "continuity gate excludes child worktrees and fails open on opaque input" -} - -test_claude_hook_registration_preserves_stop_backstop() { - jq -e ' - [.hooks.PreToolUse[] | select(.matcher == "Bash") | .hooks[].command] - | any(contains("fm-continuity-pretool-check.sh")) - ' "$ROOT/.claude/settings.json" >/dev/null || fail "Claude settings omit the continuity PreToolUse hook" - jq -e ' - .hooks.Stop == [{"hooks":[ - {"type":"command","command":"\"$CLAUDE_PROJECT_DIR\"/bin/fm-turnend-guard.sh --claude"}, - {"type":"command","command":"\"$CLAUDE_PROJECT_DIR\"/bin/fm-claude-stop-autoarm.sh","asyncRewake":true,"timeout":28800} - ]}] - ' "$ROOT/.claude/settings.json" >/dev/null || fail "Claude Stop registration changed: the --claude guard and the asyncRewake auto-arm with an explicit multi-hour timeout must both stay registered" - pass "Claude wires the continuity gate, the --claude Stop backstop, and the Stop-owned auto-arm registration" -} - -test_gate_scope_and_recovery_exceptions -test_live_lock_allows_fleet_command_even_with_stale_beacon -test_child_worktree_and_malformed_input_fail_open -test_claude_hook_registration_preserves_stop_backstop diff --git a/tests/fm-subagent-pretool-check.test.sh b/tests/fm-subagent-pretool-check.test.sh index c1a2115897a..6b4868b1d64 100755 --- a/tests/fm-subagent-pretool-check.test.sh +++ b/tests/fm-subagent-pretool-check.test.sh @@ -7,6 +7,7 @@ set -u . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" CHECK="$ROOT/bin/fm-subagent-pretool-check.sh" +SETTINGS="$ROOT/.claude/settings.json" TMP_ROOT=$(fm_test_tmproot fm-subagent-pretool-tests) PRIMARY="$TMP_ROOT/primary" STATE="$PRIMARY/state" @@ -30,17 +31,6 @@ DELEGATION_TOOLS='Task Agent Workflow RemoteTrigger Monitor ScheduleWakeup SendM # Tools that must stay available: denying these would break ordinary work. PRESERVED_TOOLS='Bash Edit Read Write Skill ToolSearch WebFetch WebSearch NotebookEdit ReportFindings DesignSync PushNotification' -# Session-local todo-list tools. They match a delegation stem but create no -# runnable work, so the guard's plan-only exclusion must allow them. -PLAN_ONLY_TOOLS='TaskCreate TaskUpdate' - -# Names the plan-only exclusion must NOT release. Five of them contain a -# plan-only name as a substring and would be let through by a substring rather -# than exact-name match; bare Task is what a shortened entry of "task" would -# release. Together they make the exact-name contract testable instead of -# assumed. -PLAN_ONLY_NEAR_MISSES='TaskCreateAgent TaskCreateWorktree TaskUpdateAgent RemoteTaskCreate Task TaskCreator' - run_tool() { local tool=$1 rc=0 shift @@ -72,15 +62,20 @@ expect_deny() { } # --------------------------------------------------------------------------- -# Delegation-shape PreToolUse guard. +# Tracked settings boundary and delegation-shape PreToolUse guard. # --------------------------------------------------------------------------- +test_tracked_settings_do_not_ship_permissions_deny() { + jq -e 'keys == ["hooks"] and (has("permissions") | not)' "$SETTINGS" >/dev/null \ + || fail "tracked Claude settings must contain only hooks and no permissions key" + pass "tracked Claude settings do not ship permissions.deny" +} + test_guard_denies_every_currently_known_delegation_tool() { local tool for tool in $DELEGATION_TOOLS; do case "$tool" in TaskOutput|TaskStop|TaskGet|TaskList|CronList) continue ;; - TaskCreate|TaskUpdate) continue ;; esac expect_deny "known delegation tool" "$tool" done @@ -112,28 +107,6 @@ test_guard_allows_ordinary_and_observe_only_tools() { pass "the guard leaves ordinary tools and observe-or-stop operations alone" } -test_guard_allows_session_local_todo_tools() { - # These write, so they are not observe-or-stop, but what they write is the - # harness's session-local todo list: no executor, no agent, no worktree, no - # schedule, nothing that outlives the session. Denying them stops the primary - # tracking its own plan and grants no delegation power in exchange. - local tool - for tool in $PLAN_ONLY_TOOLS; do - expect_allow "session-local todo tool" "$tool" - done - pass "the guard leaves the session-local todo list alone" -} - -test_plan_only_exclusion_is_exact_name() { - # The plan-only exclusion must never widen by substring or by a shorter stem. - # Every name here would be released by such a widening and must stay denied. - local tool - for tool in $PLAN_ONLY_NEAR_MISSES; do - expect_deny "plan-only near miss" "$tool" - done - pass "the plan-only exclusion releases exactly two names and nothing that merely contains them" -} - test_guard_never_classifies_mcp_tools() { # An MCP server names its own tools; a task or agent noun there is common and # has nothing to do with fleet dispatch. @@ -276,11 +249,34 @@ test_missing_jq_stdin_transport_fails_open() { pass "missing jq for stdin transport fails open rather than denying every tool call" } +test_claude_hook_registration_preserves_bash_seatbelts() { + jq -e ' + [.hooks.PreToolUse[] | .hooks[].command] + | any(contains("fm-subagent-pretool-check.sh --claude")) + ' "$SETTINGS" >/dev/null || fail "Claude settings omit the delegation-shape PreToolUse guard" + # A stem-enumerating matcher repeats the fail-open-by-enumeration defect the + # script exists to remove. Match all tools and let the script be the single + # owner of classification. + jq -e ' + [.hooks.PreToolUse[] | select(.hooks[].command | contains("fm-subagent-pretool-check.sh")) | .matcher] | .[0] + | . == ".*" + ' "$SETTINGS" >/dev/null || fail "the guard matcher must match all tools" + jq -e ' + [.hooks.PreToolUse[] | select(.matcher == "Bash") | .hooks[].command] + == [ + "\"$CLAUDE_PROJECT_DIR\"/bin/fm-arm-pretool-check.sh --claude", + "\"$CLAUDE_PROJECT_DIR\"/bin/fm-cd-pretool-check.sh --claude" + ] + ' "$SETTINGS" >/dev/null || fail "Claude Bash PreToolUse must retain only the arm-shape and persistent-cd seatbelts" + jq -e '.hooks.Stop[0].hooks[0].command | contains("fm-turnend-guard.sh")' "$SETTINGS" >/dev/null \ + || fail "the Stop turn-end guard changed" + pass "Claude wires the delegation guard, retains only non-status Bash seatbelts, and preserves the Stop guard" +} + +test_tracked_settings_do_not_ship_permissions_deny test_guard_denies_every_currently_known_delegation_tool test_guard_denies_hypothetical_future_tools test_guard_allows_ordinary_and_observe_only_tools -test_guard_allows_session_local_todo_tools -test_plan_only_exclusion_is_exact_name test_guard_never_classifies_mcp_tools test_deny_message_defers_to_intake_classification test_escape_hatch_allows_deliberate_use @@ -289,3 +285,4 @@ test_secondmate_home_is_in_scope test_stdin_transports_and_output_shapes test_malformed_transport_fails_open test_missing_jq_stdin_transport_fails_open +test_claude_hook_registration_preserves_bash_seatbelts diff --git a/tests/fm-test-isolation-proof.test.sh b/tests/fm-test-isolation-proof.test.sh index 1847338e8cd..e34e97e8515 100755 --- a/tests/fm-test-isolation-proof.test.sh +++ b/tests/fm-test-isolation-proof.test.sh @@ -1,12 +1,24 @@ #!/usr/bin/env bash -# Behavioral tests for the isolation-proof and test-run public interfaces. +# Contract tests for bin/fm-test-isolation-proof.sh - the Phase 2 pre-shard +# isolation proof harness. +# +# These tests assert the candidate-set contract, serial exclusions, aggregate +# failure reporting, and that Phase 4 production shards consume this exact set. +# They deliberately do NOT re-run the full concurrent candidate matrix on every +# invocation (that matrix is owned by the harness itself and archived under +# docs/fm-test-isolation-proof.md after a deliberate proof run). set -u +# shellcheck disable=SC1091 # shellcheck source=tests/lib.sh . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" PROOF="$ROOT/bin/fm-test-isolation-proof.sh" RUNNER="$ROOT/bin/fm-test-run.sh" +CI="$ROOT/.github/workflows/ci.yml" +CONTRIB="$ROOT/CONTRIBUTING.md" +PROOF_DOC="$ROOT/docs/fm-test-isolation-proof.md" +PROOF_JSON="$ROOT/docs/fm-test-isolation-proof.json" assert_present "$PROOF" "bin/fm-test-isolation-proof.sh is missing" [ -x "$PROOF" ] || fail "bin/fm-test-isolation-proof.sh must be executable" @@ -19,6 +31,7 @@ test_list_candidates_nonempty_and_stable() { [ "$count" -ge 10 ] || fail "expected a bounded non-trivial candidate set, got $count" sorted=$(printf '%s\n' "$listed" | LC_ALL=C sort) [ "$listed" = "$sorted" ] || fail "--list must be sorted for a stable matrix" + # No duplicates. [ "$(printf '%s\n' "$listed" | uniq | wc -l | tr -d ' ')" = "$count" ] \ || fail "--list must not duplicate candidates" while IFS= read -r line; do @@ -34,8 +47,11 @@ test_list_candidates_nonempty_and_stable() { test_candidates_exclude_serial_classes() { local listed listed=$("$PROOF" --list) + # Self must never re-enter the concurrent matrix. + printf '%s\n' "$listed" | grep -Fq 'tests/fm-test-isolation-proof.test.sh' \ + && fail "isolation-proof test must not be a parallel candidate" + # Real tmux smoke, watcher lock, real herdr, AFK, live harnesses stay serial. for banned in \ - tests/fm-test-isolation-proof.test.sh \ tests/fm-backend-tmux-smoke.test.sh \ tests/fm-watcher-lock.test.sh \ tests/fm-wake-queue.test.sh \ @@ -50,6 +66,16 @@ test_candidates_exclude_serial_classes() { pass "serial classes remain excluded from the parallel candidate set" } +test_candidates_match_archived_proof() { + local listed archived + assert_present "$PROOF_JSON" "docs/fm-test-isolation-proof.json missing" + listed=$("$PROOF" --list) + archived=$(jq -r '.scripts[].path' "$PROOF_JSON" | LC_ALL=C sort) + [ "$listed" = "$archived" ] \ + || fail "candidate set must exactly match the archived isolation proof" + pass "candidate set exactly matches the archived isolation proof" +} + test_extra_hermetic_candidates_present() { local listed listed=$("$PROOF" --list) @@ -63,7 +89,7 @@ test_extra_hermetic_candidates_present() { printf '%s\n' "$listed" | grep -Fxq "$want" \ || fail "extra hermetic candidate missing: $want" done - pass "audited fake-backend and stub-network extras are candidates" + pass "audited fake-backend / stub-network extras are candidates" } test_list_exclusions_documents_reasons() { @@ -85,7 +111,82 @@ test_family_map_labels_this_contract() { pass "isolation-proof contract test is family-mapped" } -test_parallel_shards_consume_the_proven_set() { +test_aggregate_failure_under_concurrency() { + local tmp pass_f fail_f harness rc out + tmp=$(mktemp -d "${TMPDIR:-/tmp}/fm-isolation-agg.XXXXXX") + pass_f="$tmp/pass.test.sh" + fail_f="$tmp/fail.test.sh" + cat >"$pass_f" <<'SH' +#!/usr/bin/env bash +echo "ok - pass" +exit 0 +SH + cat >"$fail_f" <<'SH' +#!/usr/bin/env bash +echo "not ok - fail" +exit 1 +SH + chmod +x "$pass_f" "$fail_f" + # Minimal fixture harness mirroring aggregate + concurrent wait semantics. + harness="$tmp/harness.sh" + cat >"$harness" <<'SH' +#!/usr/bin/env bash +set -eu +jobs=$1 +shift +pids=() +rcs=() +paths=() +idx=0 +for s in "$@"; do + idx=$((idx + 1)) + ( + bash "$s" + echo $? >"${TMPDIR:-/tmp}/iso-rc-$idx" + ) & + pids+=("$!") + paths+=("$s") + while [ "${#pids[@]}" -ge "$jobs" ]; do + wait "${pids[0]}" || true + pids=("${pids[@]:1}") + done +done +while [ "${#pids[@]}" -gt 0 ]; do + wait "${pids[0]}" || true + pids=("${pids[@]:1}") +done +failed=0 +for i in $(seq 1 "$idx"); do + rc=$(cat "${TMPDIR:-/tmp}/iso-rc-$i" 2>/dev/null || echo 1) + [ "$rc" -eq 0 ] || failed=$((failed + 1)) + rm -f "${TMPDIR:-/tmp}/iso-rc-$i" +done +echo "FM_ISOLATION_SUMMARY total=$idx failed=$failed" +[ "$failed" -eq 0 ] +SH + chmod +x "$harness" + set +e + out=$(TMPDIR="$tmp" bash "$harness" 2 "$pass_f" "$fail_f" 2>&1) + rc=$? + set -e + [ "$rc" -ne 0 ] || fail "concurrent aggregate must fail when any candidate fails" + printf '%s\n' "$out" | grep -Fq 'FM_ISOLATION_SUMMARY total=2 failed=1' \ + || fail "aggregate summary must report total=2 failed=1: $out" + rm -rf "$tmp" + pass "aggregate failure reporting survives concurrency" +} + +test_phase4_consumes_proven_set_only() { + assert_present "$CI" "ci.yml missing" + assert_present "$RUNNER" "fm-test-run.sh missing" + # Phase 4 portable parallel lanes must exist and use lane selection, not --all. + grep -Fq 'bin/fm-test-run.sh --lane portable-parallel-1' "$CI" \ + || fail "CI portable parallel 1 must use --lane portable-parallel-1" + grep -Fq 'bin/fm-test-run.sh --lane portable-parallel-2' "$CI" \ + || fail "CI portable parallel 2 must use --lane portable-parallel-2" + grep -Fq 'bin/fm-test-run.sh --lane portable-serial' "$CI" \ + || fail "CI portable serial must use --lane portable-serial" + # Shard union must equal this harness's proven list. local proven shards proven=$("$PROOF" --list | LC_ALL=C sort -u) shards=$( @@ -96,12 +197,33 @@ test_parallel_shards_consume_the_proven_set() { ) [ "$proven" = "$shards" ] \ || fail "portable parallel shards must equal isolation-proof --list exactly" - pass "parallel shards consume the proven-isolated set only" + # Local --jobs is bounded to this proven set (refuse is contract-tested in + # fm-test-run.test.sh); the option must exist. + grep -E '^[[:space:]]*--jobs\)' "$RUNNER" >/dev/null 2>&1 \ + || fail "fm-test-run.sh must expose bounded --jobs after Phase 4" + pass "Phase 4 portable shards consume the proven-isolated set only" +} + +test_docs_record_proof_owner() { + assert_present "$PROOF_DOC" "docs/fm-test-isolation-proof.md missing" + grep -Fq 'bin/fm-test-isolation-proof.sh' "$PROOF_DOC" \ + || fail "proof doc must name the harness owner" + grep -Fq 'production_sharding_enabled' "$PROOF_DOC" \ + || fail "proof doc must record the archived proof-time sharding flag" + grep -Fq 'concurrency' "$PROOF_DOC" \ + || fail "proof doc must record concurrency" + assert_present "$CONTRIB" "CONTRIBUTING.md missing" + grep -Fq 'fm-test-isolation-proof' "$CONTRIB" \ + || fail "CONTRIBUTING must document the isolation-proof entry point" + pass "docs archive the isolation-proof owner and posture" } test_list_candidates_nonempty_and_stable test_candidates_exclude_serial_classes +test_candidates_match_archived_proof test_extra_hermetic_candidates_present test_list_exclusions_documents_reasons test_family_map_labels_this_contract -test_parallel_shards_consume_the_proven_set +test_aggregate_failure_under_concurrency +test_phase4_consumes_proven_set_only +test_docs_record_proof_owner From bd33b0e15c16a55066179d284d10182331530722 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Sat, 25 Jul 2026 01:36:18 -0700 Subject: [PATCH 10/52] fix: make quota-aware profile selection agent-owned (#1018) * Replace quota dispatch selector instructions * no-mistakes(review): Align bootstrap docs with agent-owned dispatch selection --- .agents/skills/bootstrap-diagnostics/SKILL.md | 5 +- .agents/skills/harness-adapters/SKILL.md | 16 + AGENTS.md | 11 +- bin/fm-bootstrap.sh | 45 +-- docs/architecture.md | 5 +- docs/configuration.md | 15 +- tests/fm-instruction-owners.test.sh | 307 ++++++++++++++++++ 7 files changed, 352 insertions(+), 52 deletions(-) create mode 100755 tests/fm-instruction-owners.test.sh diff --git a/.agents/skills/bootstrap-diagnostics/SKILL.md b/.agents/skills/bootstrap-diagnostics/SKILL.md index 477980b8df1..641ece4ca99 100644 --- a/.agents/skills/bootstrap-diagnostics/SKILL.md +++ b/.agents/skills/bootstrap-diagnostics/SKILL.md @@ -2,7 +2,7 @@ name: bootstrap-diagnostics description: >- Agent-only handling playbook for session-start bootstrap diagnostics. - Use whenever the session-start digest's bootstrap section prints an actionable diagnostic line - MISSING, MISSING_MANUAL, BACKEND_INVALID, NEEDS_GH_AUTH, TANGLE, STARTUP_MEMORY_BUDGET, CREW_DISPATCH invalid, FLEET_SYNC, PR_CHECK_MIGRATION, SECONDMATE_SYNC, SECONDMATE_LIVENESS, NUDGE_SECONDMATES, or FMX - or when a standalone bin/fm-bootstrap.sh run prints one of those lines. + Use whenever the session-start digest's bootstrap section prints an actionable diagnostic line - MISSING, MISSING_MANUAL, BACKEND_INVALID, NEEDS_GH_AUTH, TANGLE, CREW_DISPATCH invalid, FLEET_SYNC, PR_CHECK_MIGRATION, SECONDMATE_SYNC, SECONDMATE_LIVENESS, NUDGE_SECONDMATES, or FMX - or when a standalone bin/fm-bootstrap.sh run prints one of those lines. A silent bootstrap section, or a BOOTSTRAP_INFO fact, means no skill load. user-invocable: false metadata: @@ -27,8 +27,7 @@ When any diagnostic needs captain attention, report the plain consequence and re - `TANGLE: <remediation>` - the primary checkout is stranded on a feature branch instead of its default branch; `AGENTS.md` section 8 explains why this guard exists and what it protects. The work is safe on that branch ref; restore the primary to its default branch with the printed `git -C <root> checkout <default>`, then re-validate that branch in a proper worktree. This is the only sanctioned firstmate-initiated git write to the primary, and it is a non-destructive branch switch that strands nothing. -- `STARTUP_MEMORY_BUDGET: invalid config/startup-memory-budget - <reason>` - the visible startup-memory budget is not a safe one-line positive decimal file; do not infer the default or propagate it. Correct the local primary file, then rerun session start so the normal convergence path can deliver the validated value to secondmate homes. -- `CREW_DISPATCH: invalid config/crew-dispatch.json - <reason>` - the optional dispatch profile file exists but failed low-cost bootstrap validation; stop profile-based dispatch, report the actionable error, and require correction of the malformed schema, unverified harness name, or invalid harness/effort pair rather than falling back around it or selecting a bad profile. +- `CREW_DISPATCH: invalid config/crew-dispatch.json - <reason>` - the optional dispatch profile file exists but failed low-cost bootstrap validation; stop profile-based dispatch, report the actionable error, and require correction of the malformed schema, unverified harness name, unknown selector, or invalid harness/effort pair rather than falling back around it or selecting a bad profile. - `FLEET_SYNC: <repo>: skipped: <reason>` - a benign one-off skip (offline, no origin, local-only); bootstrap continued, investigate only if it blocks work. A skip can also report the bounded fleet-refresh timeout (`FM_FLEET_SYNC_BOOTSTRAP_TIMEOUT`, or a fleet-size-aware default with a 20 second floor); a timeout never blocks startup. - `FLEET_SYNC: <repo>: recovered: <detail>` - the clone had drifted onto a clean detached HEAD holding no unique commits and the sync self-healed it (re-attached the default branch and fast-forwarded); no action needed, it is reported only so the self-heal is visible. diff --git a/.agents/skills/harness-adapters/SKILL.md b/.agents/skills/harness-adapters/SKILL.md index c284ce30ed1..036eacdc985 100644 --- a/.agents/skills/harness-adapters/SKILL.md +++ b/.agents/skills/harness-adapters/SKILL.md @@ -122,6 +122,22 @@ The supported launch-profile flags below are verified locally; each row records | pi | `--model <model>` | `--thinking <low\|medium\|high\|xhigh\|max>` | Verified 2026-07-13 on Pi 0.80.6. `pi --help` advertises `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, and `max`; `pi --print --model openai-codex/gpt-5.6-sol --thinking max 'Reply with exactly OK.'` completed successfully. | | opencode | `--model <provider/model>` | none for firstmate's interactive launch | Verified on opencode 1.17.6. `opencode run` has `--variant`, but firstmate launches the interactive `opencode --prompt` path, which has no verified effort flag. | +### Model support discovery + +Treat model and provider knowledge as current source-of-truth discovery, not as a permanent namespace or provider mapping. +Use the discovery surface in the current authenticated environment because supported and available models can change by version, account, and configuration. + +| Harness | Authoritative discovery surface | +|---|---| +| claude | Open the current interactive session's `/model` picker; `claude --help` documents the accepted alias or full-model-name input shape. | +| codex | Open the current interactive session's `/model` picker. | +| opencode | Run `opencode models [provider]`, which lists available provider/model identifiers. | +| pi | Run `pi --list-models [search]`; Pi's installed `docs/models.md` owns how built-in, extension-registered, and custom provider/model entries reach that list. | +| grok | Run `grok models`, which lists the models available to the current Grok installation and account. | + +For an unfamiliar harness or model namespace, establish support and provider identity from that harness's authoritative CLI help, model listing, or current documentation rather than guessing from a name or prefix. +If those sources do not establish the relationship needed for dispatch, fail loudly and report the unresolved candidate. + When a requested effort value is outside the harness-specific accepted set, `fm-spawn` records the requested `effort=` in meta but emits no effort flag for that harness. This preserves launch success instead of passing a known-bad value. diff --git a/AGENTS.md b/AGENTS.md index bf13013b337..d972c591709 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -158,11 +158,18 @@ A silent bootstrap section needs no action; for any printed actionable diagnosti Load `harness-adapters` before every spawn or recovery and before trust handling, skill invocation, interrupt, exit, resume, or adapter verification. The verified harnesses are `claude`, `codex`, `opencode`, `pi`, and `grok`; never dispatch on an unverified adapter. -If configured harness data names an unverified adapter, report it and fall back only to a verified adapter rather than launching it. +If static `config/crew-harness` or `config/secondmate-harness` names an unverified adapter, report it and fall back only to a verified adapter rather than launching it. -`docs/configuration.md` owns dispatch-profile and runtime-backend schemas, `bin/fm-dispatch-select.sh` owns selector mechanics, `bin/fm-harness.sh` owns static resolution, and `bin/fm-spawn.sh` owns launch flags and fail-closed validation. +`docs/configuration.md` owns dispatch-profile and runtime-backend schemas, `bin/fm-harness.sh` owns static resolution, and `bin/fm-spawn.sh` owns launch flags and fail-closed validation. When dispatch profiles exist, consult them at every crewmate or scout intake and pass the resolved concrete profile required by `fm-spawn`. Routing precedence is an explicit per-task captain override, then the best-fit configured rule, then the configured default, then the static crewmate harness. +Firstmate alone resolves a matched profile array: run `quota-axi --json` at that intake, evaluate every configured candidate against that current output, and choose the candidate with the most real headroom. +Account for every candidate; if any harness/model/provider relationship, applicable quota data, or interpretation cannot be established, stop and report that candidate instead of omitting it, guessing, falling back, or calling the result quota-informed. +Preserve malformed profile configuration as an actionable error rather than selecting around it. +When every candidate is tight, preserve the captain's strongest-reasoning class rather than silently downgrading it solely to conserve quota; stop and report the tight choice if that class cannot proceed. +Break genuine headroom ties without array-order or harness bias. +`quota-axi` owns how model or product windows relate to bounding account windows; as an explicitly interim rule until successor `quota-axi-interpretation-hints-h3` lands, use the weakest applicable remaining headroom, then remove this interim rule. +`bin/fm-dispatch-select.sh` is vestigial during this transition and must not be called. The generic effort fallback and its precedence are owned by `harness-adapters`: explicit captain and standing configured effort win; otherwise use low for well-understood explicit work, xhigh for ambiguous investigation or design, intermediate levels proportionally, and never max without explicit captain preference. Do not add model-specific versions of that policy. diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index 16102adfa45..903532c480c 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -8,7 +8,6 @@ # Lines: "MISSING: <tool> (install: <command>)", # "MISSING_MANUAL: <tool> (instructions: <url>)", "NEEDS_GH_AUTH", # "BACKEND_INVALID: <name> (known: <names>)", -# "STARTUP_MEMORY_BUDGET: invalid config/startup-memory-budget - <reason>", # "CREW_DISPATCH: invalid config/crew-dispatch.json - <reason>", # "FLEET_SYNC: <repo>: skipped|recovered|STUCK: <detail>", # "PR_CHECK_MIGRATION: <private remediation>", @@ -54,13 +53,7 @@ # with update --archive-body and mv [<id>...]); an installed but # incompatible build reports MISSING like no-mistakes. A compatible # tasks-axi default backend is silent. quota-axi is required for the -# agent-owned dispatch-profile array procedure in AGENTS.md section 4 -# and .agents/skills/quota-array-dispatch/SKILL.md. -# On a primary home, the locked mutable path materializes the visible -# default config/startup-memory-budget=7500 when absent. It never -# guesses at malformed or unsafe existing files, and secondmate homes -# await the primary-authoritative inherited value instead of creating -# their own. +# agent-owned dispatch-profile array procedure in AGENTS.md section 4. # X mode is OPTIONAL and inert unless FM_HOME/.env has a non-empty # FMX_PAIRING_TOKEN. When opted in, bootstrap requires curl+jq, writes # the relay poll shim and 30s cadence config, and prints an FMX line. @@ -103,8 +96,6 @@ DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" . "$SCRIPT_DIR/fm-ff-lib.sh" # shellcheck source=bin/fm-config-inherit-lib.sh disable=SC1091 . "$SCRIPT_DIR/fm-config-inherit-lib.sh" -# shellcheck source=bin/fm-startup-memory-budget-lib.sh disable=SC1091 -. "$SCRIPT_DIR/fm-startup-memory-budget-lib.sh" # shellcheck source=bin/fm-x-lib.sh disable=SC1091 . "$SCRIPT_DIR/fm-x-lib.sh" # shellcheck source=bin/fm-backend.sh disable=SC1091 @@ -445,7 +436,7 @@ secondmate_liveness_sweep() { [ -n "$target" ] || target="$window" agent_state=$(fm_backend_agent_state "$backend" "$target" 2>/dev/null) || agent_state=unreadable case "$harness" in - claude|codex|opencode|pi|pi-signed|grok|kimi) ;; + claude|codex|opencode|pi|grok) ;; *) case "$agent_state" in dead|missing) agent_state=unverified-harness ;; esac ;; @@ -628,7 +619,7 @@ x_mode_remove_artifact() { # applying a cadence transition to a running watcher is the caller's job via # the emitted harness-aware supervision repair instruction. x_mode_setup() { - local env_file token shim cadence shim_body cadence_body tool missing shim_home + local env_file token shim cadence shim_body cadence_body tool missing env_file="$FM_HOME/.env" shim="$STATE/x-watch.check.sh" cadence="$CONFIG/x-mode.env" @@ -691,16 +682,9 @@ x_mode_setup() { mkdir -p "$STATE" "$CONFIG" 2>/dev/null || { fmx_arm_failed; return 0; } - case "$FM_HOME" in - /*) shim_home=$FM_HOME ;; - *) - shim_home=$(CDPATH='' cd -- "$FM_HOME" 2>/dev/null && pwd -P) \ - || { fmx_arm_failed; return 0; } - ;; - esac - shim_body=$(fmx_poll_shim_content "$shim_home" "$FM_ROOT") + shim_body=$(fmx_poll_shim_content "$FM_HOME" "$FM_ROOT") x_mode_write_if_changed "$shim" "$shim_body" 700 || { fmx_arm_failed; return 0; } - fmx_poll_shim_valid "$shim" "$shim_home" "$FM_ROOT" \ + fmx_poll_shim_valid "$shim" "$FM_HOME" "$FM_ROOT" \ || { fmx_arm_failed; return 0; } cadence_body=$(cat <<'EOF' @@ -729,15 +713,15 @@ crew_dispatch_validate() { return 0 fi err=$(jq -r ' - def verified($h): ["claude","codex","opencode","pi","pi-signed","grok","kimi"] | index($h); + def verified($h): ["claude","codex","opencode","pi","grok"] | index($h); def effort_ok($h; $e): if $e == null then true elif ($e | type) != "string" then false elif $h == "claude" then (["low","medium","high","xhigh","max"] | index($e)) elif $h == "codex" then (["low","medium","high","xhigh"] | index($e)) elif $h == "grok" then (["low","medium","high"] | index($e)) - elif $h == "pi" or $h == "pi-signed" then (["low","medium","high","xhigh","max"] | index($e)) - elif $h == "opencode" or $h == "kimi" then false + elif $h == "pi" then (["low","medium","high","xhigh","max"] | index($e)) + elif $h == "opencode" then false else true end; def profiles($value): @@ -813,18 +797,6 @@ crew_dispatch_validate() { fi } -startup_memory_budget_setup() { - # Primary bootstrap owns default publication. A secondmate is deliberately - # passive here because its setting must converge from the primary through the - # inherited-local-material contract rather than becoming a local authority. - if [ -e "$FM_HOME/.fm-secondmate-home" ] || [ -L "$FM_HOME/.fm-secondmate-home" ]; then - return 0 - fi - if ! fm_startup_memory_budget_materialize "$CONFIG"; then - echo "STARTUP_MEMORY_BUDGET: invalid config/$FM_STARTUP_MEMORY_BUDGET_FILE - $FM_STARTUP_MEMORY_BUDGET_ERROR" - fi -} - if [ "${1:-}" = "install" ]; then shift [ $# -gt 0 ] || { echo "usage: fm-bootstrap.sh install <tool>..." >&2; exit 1; } @@ -847,7 +819,6 @@ fi # runnable. Detect-only sessions never touch state. if [ "${FM_BOOTSTRAP_DETECT_ONLY:-0}" != 1 ]; then "$SCRIPT_DIR/fm-pr-check-migrate.sh" || true - startup_memory_budget_setup fi if [ "$BACKEND_VALID" -eq 0 ]; then diff --git a/docs/architecture.md b/docs/architecture.md index cb515ba6350..b8c6620115d 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -140,8 +140,9 @@ The intake and authority contract in `AGENTS.md` owns when separate scout resear ## Dispatch profiles Crewmate and scout dispatch can stay on the static crewmate harness resolved by `config/crew-harness`, or it can use local dispatch profiles in `config/crew-dispatch.json`. -The dispatch file is intentionally judgment-based: firstmate reads the natural-language rules at intake, chooses the best matching rule, resolves that rule directly or through a supported selector, and passes only concrete `--harness`, `--model`, and `--effort` axes to `fm-spawn.sh`. -The shell scripts validate the JSON shape and verified harness/effort combinations, and `fm-dispatch-select.sh` owns quota-aware array selection plus OS-backed random fallback, but they do not parse task intent or match the natural-language rules. +The dispatch file is intentionally judgment-based: firstmate reads the natural-language rules at intake, chooses the best matching rule, resolves profile arrays itself from current quota output under `AGENTS.md` section 4, and passes only concrete `--harness`, `--model`, and `--effort` axes to `fm-spawn.sh`. +The shell scripts validate the JSON shape and verified harness/effort combinations, but they do not parse task intent, match natural-language rules, or own array selection. +`fm-dispatch-select.sh` is vestigial and not called during this transition; its separately based removal follows after the replacement instruction lands. The session-start bootstrap step keeps valid dispatch configuration silent unless verbose facts are enabled and surfaces a concise invalid-config line when validation fails. When the file exists, `fm-spawn.sh` refuses crewmate and scout launches without an explicit harness, so `config/crew-harness` is only automatic when no dispatch profile file is active. Secondmate launches are exempt because they resolve the secondmate harness and any optional secondmate model or effort tokens instead. diff --git a/docs/configuration.md b/docs/configuration.md index d8e1c39f9ed..e42100a2060 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -199,11 +199,11 @@ For Pi secondmate launches, `fm-spawn.sh` starts Pi with `-e` pointed at the sec ## Crew dispatch profiles (config/crew-dispatch.json) `config/crew-dispatch.json` is an optional local, gitignored file containing natural-language rules that firstmate reads before dispatching a crewmate or scout. -The shell scripts do not match those rules; firstmate chooses the best matching rule with judgment, resolves that rule directly or through a supported selector, and passes only concrete `--harness`, `--model`, and `--effort` flags to `fm-spawn.sh`. +The shell scripts do not match those rules; firstmate chooses the best matching rule with judgment, resolves its profile object or array under the operating contract in `AGENTS.md` section 4, and passes only concrete `--harness`, `--model`, and `--effort` flags to `fm-spawn.sh`. When the file exists, `fm-spawn.sh` enforces that contract by refusing crewmate and scout spawns that lack an explicit harness (`--harness`, a positional adapter, or a raw launch command). Batch spawns satisfy the same requirement with a shared `--harness`. Secondmate spawns are exempt and still resolve through `config/secondmate-harness` and its optional model and effort tokens. -This section is the single owner of the canonical schema and its per-field semantics; `AGENTS.md` section 4 keeps only the dispatch procedure and points here. +This section is the single owner of the canonical schema and its per-field semantics; `AGENTS.md` section 4 owns the dispatch and array-selection procedure. ```json { @@ -229,16 +229,15 @@ The single-object form stays fully backward-compatible, and every profile needs Profile `model` and `effort` fields and rule `why` are optional. An omitted model or effort means the selected harness uses its own default for that axis. Every profile array is an implicit quota-aware choice and does not need a selector property. -`select: "quota-balanced"` remains accepted on rules for compatibility and has the same behavior as an implicit array choice. -If no dispatch rule fits, firstmate resolves `default` through the same object-or-array selection path before falling back to `config/crew-harness`. +`select: "quota-balanced"` remains accepted on rules for configuration compatibility and routes to the same agent-judged array procedure. +If no dispatch rule fits, firstmate resolves `default` through the same object-or-array path before falling back to `config/crew-harness`. If a selected profile carries an effort value the chosen harness does not accept, `fm-spawn.sh` records the requested `effort=` in task meta for traceability but omits the launch flag, and bootstrap reports the invalid harness/effort pair as a `CREW_DISPATCH` diagnostic when it is visible in the file. -Quota-aware selection is implemented by `bin/fm-dispatch-select.sh`, whose header owns provider and product mapping, relevant-window scoring, the stale-clear freshness margin, random tie-breaking, OS-backed random operational fallback, and safe selection-basis diagnostics. -Quota-data trouble never blocks dispatch, but malformed profile configuration remains an actionable validation error. +`bin/fm-dispatch-select.sh` is vestigial during the instruction transition and must not be called; a separately based follow-up removes it after the replacement operating contract lands. See [`docs/examples/crew-dispatch.json`](examples/crew-dispatch.json) for a starting point to copy into local `config/crew-dispatch.json`. When the file exists, bootstrap validates it with `jq`. Valid files stay silent by default; with `FM_BOOTSTRAP_VERBOSE_FACTS=1`, bootstrap emits `BOOTSTRAP_INFO: crew dispatch active config/crew-dispatch.json`, one `BOOTSTRAP_INFO:` fact per rule, and one fact for the optional default profile set. Malformed JSON, an empty or malformed rule/default array, an unverified harness, an unknown `select`, or an effort value unsupported by that harness is reported as `CREW_DISPATCH: invalid config/crew-dispatch.json - ...`; missing `jq` is reported through the normal `MISSING: jq` install-consent flow. -Because the spawn backstop is gated by file presence, any fallback path after a missing match, validation error, or missing `jq` still passes a resolved harness explicitly until the file is fixed or removed. +While the file remains present, no crewmate or scout spawn may proceed without an explicit resolved harness; malformed configuration must be reported and corrected rather than selected around. Secondmate homes inherit this file from the primary, so a secondmate's own crewmates apply the same dispatch profile behavior. ## Toolchain @@ -259,7 +258,7 @@ When `config/crew-dispatch.json` exists, bootstrap also requires `jq` for dispat When X mode is opted in, bootstrap also requires `curl` and `jq` before arming the relay poll shim. `tasks-axi` and `quota-axi` are required bootstrap tools in every profile, the same class as `lavish-axi`. An absent or incompatible `tasks-axi` reports `MISSING: tasks-axi (install: npm install -g tasks-axi)`; when `config/backlog-backend` is not `manual` and compatible `tasks-axi` is on `PATH`, bootstrap stays silent and firstmate uses its verbs for routine backlog mutations, otherwise it hand-edits `data/backlog.md` until installation is approved and completed. -An absent `quota-axi` reports `MISSING: quota-axi (install: npm install -g quota-axi)`; `bin/fm-dispatch-select.sh` still selects uniformly from the valid candidate array with an OS-backed random source when quota data is unavailable. +An absent `quota-axi` reports `MISSING: quota-axi (install: npm install -g quota-axi)`; firstmate cannot resolve a profile array until current quota output is available for every candidate. Bootstrap also reports a `TANGLE:` line when `FM_ROOT` is on a named non-default branch; follow the printed checkout remediation rather than treating it as an installable tool problem. In a read-only session that did not get the fleet lock, the same line is advisory and omits the checkout command. The locked session-start bootstrap step also runs a best-effort project clone refresh through `fm-fleet-sync.sh`. diff --git a/tests/fm-instruction-owners.test.sh b/tests/fm-instruction-owners.test.sh new file mode 100755 index 00000000000..bcbf671c38e --- /dev/null +++ b/tests/fm-instruction-owners.test.sh @@ -0,0 +1,307 @@ +#!/usr/bin/env bash +# Static contract tests for conditional instruction owners introduced before the +# AGENTS.md reduction pass. +# shellcheck disable=SC2016 +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +DIAG="$ROOT/.agents/skills/diagnostic-reasoning/SKILL.md" +PROJECT="$ROOT/.agents/skills/project-management/SKILL.md" +HARNESS="$ROOT/.agents/skills/harness-adapters/SKILL.md" +CODING="$ROOT/.agents/skills/firstmate-coding-guidelines/SKILL.md" +RECOVERY="$ROOT/.agents/skills/stuck-crewmate-recovery/SKILL.md" +SECONDMATE="$ROOT/.agents/skills/secondmate-provisioning/SKILL.md" +CONFIG="$ROOT/docs/configuration.md" +AGENTS="$ROOT/AGENTS.md" +BRIEF="$ROOT/bin/fm-brief.sh" +BOOTSTRAP="$ROOT/bin/fm-bootstrap.sh" + +test_new_skill_metadata_and_triggers() { + local skill name count + for pair in "diagnostic-reasoning:$DIAG" "project-management:$PROJECT"; do + name=${pair%%:*} + skill=${pair#*:} + assert_present "$skill" "$name skill is missing" + assert_grep "name: $name" "$skill" "$name skill metadata has the wrong name" + assert_grep "user-invocable: false" "$skill" "$name skill must not be user-invocable" + assert_grep " internal: true" "$skill" "$name skill must be internal" + count=$(grep -Fc -- "- \`$name\` -" "$ROOT/AGENTS.md") + [ "$count" -eq 1 ] || fail "$name must have exactly one AGENTS.md trigger entry, found $count" + done + assert_grep 'Use before scoping a reported bug and before acting on a diagnostic report.' "$DIAG" \ + "diagnostic skill metadata lost its precise load trigger" + assert_grep '`diagnostic-reasoning` - load before scoping a reported bug and before acting on a diagnostic report.' "$ROOT/AGENTS.md" \ + "AGENTS.md lost the diagnostic-reasoning trigger" + assert_grep 'Use before adding, creating, removing, or initializing a project.' "$PROJECT" \ + "project-management skill metadata lost its precise load trigger" + assert_grep '`project-management` - load before adding, creating, removing, or initializing a project.' "$ROOT/AGENTS.md" \ + "AGENTS.md lost the project-management trigger" + pass "new internal skills have one precise AGENTS.md trigger each" +} + +test_diagnostic_owner_covers_causal_procedure() { + assert_grep "single owner of Firstmate's bug-diagnosis reasoning procedure" "$DIAG" \ + "diagnostic skill does not declare ownership" + for phrase in \ + "end-to-end reproduction aligned with the real user path" \ + "initiating trigger" \ + "masking condition" \ + "visible symptom" \ + "proven path" \ + "relevant history" \ + "smallest counterfactual" \ + "disconfirming evidence"; do + assert_grep "$phrase" "$DIAG" "diagnostic owner is missing '$phrase'" + done + assert_grep "evidence, not authorization to change code" "$DIAG" \ + "diagnostic owner lost the diagnosis-only authority boundary" + pass "diagnostic-reasoning owns the approved evidence procedure" +} + +test_project_management_owner_covers_guarded_operations() { + assert_grep "single owner of Firstmate's project-management procedure" "$PROJECT" \ + "project-management skill does not declare ownership" + for phrase in \ + 'bin/fm-project-mode.sh' \ + '`no-mistakes`' \ + '`direct-PR`' \ + '`local-only`' \ + 'Default it off' \ + 'Creating a GitHub repository is outward-facing.' \ + "captain's explicit consent" \ + 'Never issue a raw removal command from Firstmate.' \ + 'no-mistakes init && no-mistakes doctor'; do + assert_grep "$phrase" "$PROJECT" "project-management owner is missing '$phrase'" + done + pass "project-management owns registry, delivery posture, consent, initialization, and removal safety" +} + +test_generic_effort_fallback_respects_precedence() { + local section + section=$(awk ' + /^Effort precedence is / { found = 1 } + found && /^The supported launch-profile flags / { exit } + found { print } + ' "$HARNESS") + assert_contains "$section" "explicit per-task captain instruction first" \ + "effort rubric lost per-task captain precedence" + assert_contains "$section" "standing dispatch profile or secondmate pin" \ + "effort rubric lost standing configuration precedence" + assert_contains "$section" 'Use `low` for well-understood work' \ + "effort rubric lost its low fallback" + assert_contains "$section" '`xhigh` for ambiguous investigation or design' \ + "effort rubric lost its xhigh fallback" + assert_contains "$section" "Choose intermediate levels proportionally" \ + "effort rubric lost proportional intermediate levels" + assert_contains "$section" 'Never select `max` from this fallback' \ + "effort rubric permits max without an explicit captain preference" + if printf '%s\n' "$section" | grep -qi sol; then + fail "generic effort fallback must not contain Sol-specific policy" + fi + pass "generic effort fallback applies only below captain and standing configuration" +} + +test_agent_owned_quota_array_dispatch_contract() { + local phrase + for phrase in \ + 'Firstmate alone resolves a matched profile array' \ + 'run `quota-axi --json` at that intake' \ + 'evaluate every configured candidate against that current output' \ + 'choose the candidate with the most real headroom' \ + 'if any harness/model/provider relationship, applicable quota data, or interpretation cannot be established, stop and report that candidate' \ + 'instead of omitting it, guessing, falling back, or calling the result quota-informed' \ + 'Preserve malformed profile configuration as an actionable error' \ + "preserve the captain's strongest-reasoning class rather than silently downgrading it" \ + 'Break genuine headroom ties without array-order or harness bias' \ + '`quota-axi` owns how model or product windows relate to bounding account windows' \ + 'explicitly interim rule until successor `quota-axi-interpretation-hints-h3` lands' \ + '`bin/fm-dispatch-select.sh` is vestigial during this transition and must not be called'; do + assert_grep "$phrase" "$AGENTS" "array-dispatch contract lost '$phrase'" + done + + for phrase in \ + '| claude | Open the current interactive session' \ + '| codex | Open the current interactive session' \ + '| opencode | Run `opencode models [provider]`' \ + '| pi | Run `pi --list-models [search]`' \ + '| grok | Run `grok models`' \ + "For an unfamiliar harness or model namespace, establish support and provider identity from that harness's authoritative CLI help, model listing, or current documentation rather than guessing" \ + 'If those sources do not establish the relationship needed for dispatch, fail loudly and report the unresolved candidate.'; do + assert_grep "$phrase" "$HARNESS" "model discovery guidance lost '$phrase'" + done + assert_grep 'not as a permanent namespace or provider mapping' "$HARNESS" \ + "model discovery guidance permits a fixed provider table" + assert_grep '`AGENTS.md` section 4 owns the dispatch and array-selection procedure.' "$CONFIG" \ + "configuration docs do not point to the agent-owned array procedure" + assert_grep '`bin/fm-dispatch-select.sh` is vestigial during the instruction transition and must not be called' "$CONFIG" \ + "configuration docs still permit the vestigial selector" + assert_grep 'quota-axi is required for the' "$BOOTSTRAP" \ + "bootstrap docs lost the quota-axi dependency pointer" + assert_grep 'agent-owned dispatch-profile array procedure in AGENTS.md section 4.' "$BOOTSTRAP" \ + "bootstrap docs do not point to the agent-owned array procedure" + assert_no_grep 'every crew-dispatch profile array calls it automatically' "$BOOTSTRAP" \ + "bootstrap docs still claim automatic selector invocation" + assert_no_grep 'OS-backed random selection across' "$BOOTSTRAP" \ + "bootstrap docs still promise quota-unavailable random fallback" + pass "firstmate directly compares every quota candidate with authoritative model discovery" +} + +test_shared_authoring_requirements_are_owned() { + assert_grep "review every affected supported primary harness and runtime backend" "$CODING" \ + "coding guidance lost the supported compatibility matrix review" + assert_grep "prefer deterministic and idempotent enforcement over relying on agent memory alone" "$CODING" \ + "coding guidance lost deterministic idempotent enforcement" + assert_grep "critical safety, routing, startup, and supervision infrastructure" "$CODING" \ + "coding guidance lost the critical infrastructure scope" + pass "firstmate-coding-guidelines owns compatibility review and deterministic enforcement" +} + +test_secondmate_registry_contract_stays_concise() { + local guidance routing_section schema_line + routing_section=$(awk ' + /^## Routing table$/ { found = 1 } + found && /^## Charter and seed$/ { exit } + found { print } + ' "$SECONDMATE") + guidance=$(awk ' + /^## Routing table$/ { found = 1 } + found && /^## Backlog handoff$/ { exit } + found { print } + ' "$SECONDMATE") + schema_line="- <id> - <one-sentence charter summary> (home: <absolute-home-path>; scope: <natural-language responsibility>; projects: <project-a>, <project-b>; added <date>)" + assert_contains "$routing_section" "$schema_line" \ + "secondmate routing table lost the parser-compatible single-line schema" + assert_contains "$routing_section" "Each registry entry stays concise and single-line" \ + "secondmate routing table no longer requires concise single-line entries" + assert_contains "$routing_section" "genuinely domain-specific hard rules" \ + "secondmate routing table no longer limits extra prose to domain-specific hard rules" + assert_contains "$routing_section" "The home-seeded \`data/charter.md\` is the sole owner of boilerplate idle-by-default behavior, the normal delegation lifecycle, and standard escalation contracts" \ + "secondmate routing table lost the explicit charter ownership pointer" + assert_contains "$routing_section" "no extra registry pointer field is needed" \ + "secondmate routing table no longer explains why the existing home field is the charter pointer" + for phrase in \ + "go idle and wait silently" \ + "Act only on tasks" \ + "never spawn a survey" \ + "run normal firstmate bootstrap" \ + "escalation back to the main firstmate status file" \ + "requests-from-main-firstmate contract" \ + "waits for routed tasks, never self-initiating a survey or audit" \ + "marked supervisor requests return through status" \ + "unmarked captain messages stay conversational"; do + if printf '%s\n' "$guidance" | grep -F "$phrase" >/dev/null; then + fail "secondmate provisioning guidance restated charter boilerplate: $phrase" + fi + done + pass "secondmate registry guidance keeps concise routes and points to the charter" +} + +test_state_startup_and_ordinary_recovery_placement() { + assert_grep "single owner of the top-level operational-home layout" "$CONFIG" \ + "configuration docs do not own the operational state layout" + assert_grep "header is the single owner of session-start ordering" "$CONFIG" \ + "session-start mechanism is not assigned to the script header" + assert_grep "Ordinary dead-direct-report recovery is owned by \`stuck-crewmate-recovery\`" "$CONFIG" \ + "D05 ordinary recovery placement is missing" + assert_grep "## Session-start reconciliation for a dead ordinary direct report" "$RECOVERY" \ + "stuck-crewmate-recovery lacks the dead ordinary direct-report procedure" + assert_grep "treehouse status" "$RECOVERY" \ + "ordinary recovery lost treehouse inventory inspection" + assert_grep "recorded \`orca_worktree_id=\` and \`terminal=\`" "$RECOVERY" \ + "ordinary recovery lost Orca inventory inspection" + assert_grep "session-start digest reports an ordinary direct report's endpoint dead or its metadata has no window" "$AGENTS" \ + "AGENTS.md does not trigger ordinary dead-report recovery" + pass "state, startup, and ordinary recovery have focused owners and triggers" +} + +test_compressed_agents_owner_map() { + assert_grep '`docs/configuration.md` is the single owner of the top-level operational-home layout' "$AGENTS" \ + "AGENTS.md lost the state-layout owner pointer" + assert_grep 'header is the single owner of composed commands, ordering, and digest contents' "$AGENTS" \ + "AGENTS.md lost the session-start owner pointer" + assert_grep '`docs/configuration.md` owns dispatch-profile and runtime-backend schemas' "$AGENTS" \ + "AGENTS.md lost the dispatch-schema owner pointer" + assert_grep 'That skill owns registry syntax, delivery-mode selection' "$AGENTS" \ + "AGENTS.md lost the project-management owner pointer" + assert_grep 'The delivery lifecycle is an always-loaded operational contract' "$AGENTS" \ + "AGENTS.md no longer owns the delivery lifecycle" + assert_grep 'Fleet supervision is an always-loaded operational contract' "$AGENTS" \ + "AGENTS.md no longer owns fleet supervision" + assert_grep '`.tasks.toml`, `docs/configuration.md`, and current `tasks-axi --help` own the backlog schema' "$AGENTS" \ + "AGENTS.md lost the backlog-mechanics owner pointer" + assert_grep '`bin/fm-brief.sh` and its help own scaffold syntax' "$AGENTS" \ + "AGENTS.md lost the brief-mechanics owner pointer" + assert_grep '`docs/configuration.md` owns activation, generated state, cadence, wire protocol' "$AGENTS" \ + "AGENTS.md lost the X-mode mechanics owner pointer" + pass "compressed AGENTS.md records the approved one-owner map" +} + +test_intake_reuses_evidence_and_parallelizes_safe_work() { + for phrase in \ + 'consult existing reports and established evidence' \ + 'remaining bounded research inside it' \ + 'unresolved uncertainty could materially change whether or what to build' \ + 'relay it without a design-only scout' \ + 'ask one concise implementation question when useful' \ + 'Never both present a likely-enough solution' \ + 'overlap as a risk signal rather than an automatic reason to wait' \ + 'independently implemented and validated' \ + 'selected delivery path can reconcile ordinary rebases or conflicts' \ + 'Serialize only for a true semantic dependency' \ + 'shared mutable external state' \ + 'incompatible concurrent migration' \ + 'same-file editing alone is insufficient' \ + 'genuine blockers remain durable'; do + assert_grep "$phrase" "$AGENTS" "intake contract lost '$phrase'" + done + assert_grep 'dispatch isolated work immediately with no concurrency cap' "$AGENTS" \ + "intake contract lost unbounded safe parallel dispatch" + assert_grep 'captain explicitly requests a separate knowledge or design deliverable' "$AGENTS" \ + "intake contract lost captain-requested separate scouts" + assert_grep 'When implementation is separately authorized, promote the existing scout' "$AGENTS" \ + "intake contract lost genuine scout promotion" + pass "intake reuses evidence, reserves scouts for uncertainty, and parallelizes safe work" +} + +test_compressed_agents_retains_authority_and_supervision_safety() { + for phrase in \ + 'A lock-refused session must not spawn, steer, merge, drain the wake queue' \ + 'A diagnostic request, report, recommendation, or implementation-ready finding is evidence, not authorization to change code.' \ + 'The selected delivery path owns its own rigor.' \ + 'When no-mistakes is selected, no-mistakes alone owns review, fixes, tests, documentation, push, PR, and CI; otherwise follow the faster path without adding an independent reviewer.' \ + 'Never hold work outside no-mistakes for a manual clean verdict, stack serial manual reviews, or infer authority for one from security, architecture, or risk alone.' \ + 'A separate review or audit is allowed only when the captain explicitly requests that deliverable or the authorized task is a knowledge-only review; one named question remains scoped to that question.' \ + 'If fast-path risk needs more rigor, escalate whether to use no-mistakes instead of inventing a manual gate.' \ + '**local-only** has the worker stop with a clean ready branch, then waits for the configured merge authority' \ + 'A status line is a wake event, not current state' \ + 'keep exactly one live supervision cycle' \ + 'Never broadly kill watchers' \ + 'While `state/.afk` exists, the daemon owns supervision' \ + 'post the final completion follow-up before teardown'; do + assert_grep "$phrase" "$AGENTS" "compressed AGENTS.md lost safety phrase '$phrase'" + done + assert_no_grep 'Firstmate does not personally review code or deliverables' "$AGENTS" \ + "AGENTS.md retained the weaker duplicate review prohibition" + assert_no_grep 'firstmate reviews your branch' "$AGENTS" \ + "AGENTS.md retained a personal branch-review requirement" + assert_no_grep 'firstmate reviews, captain approves' "$BRIEF" \ + "generated brief retained a stacked personal-review requirement" + if grep -q "$(printf '\342\200\224')" "$AGENTS"; then + fail "AGENTS.md contains an em dash" + fi + pass "compressed AGENTS.md retains authority, supervision, AFK, and X safety" +} + +test_new_skill_metadata_and_triggers +test_diagnostic_owner_covers_causal_procedure +test_project_management_owner_covers_guarded_operations +test_generic_effort_fallback_respects_precedence +test_agent_owned_quota_array_dispatch_contract +test_shared_authoring_requirements_are_owned +test_secondmate_registry_contract_stays_concise +test_state_startup_and_ordinary_recovery_placement +test_compressed_agents_owner_map +test_intake_reuses_evidence_and_parallelizes_safe_work +test_compressed_agents_retains_authority_and_supervision_safety From dc16525cd90a38f5ea467c4b5a75787c9b92e717 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Sat, 25 Jul 2026 03:45:10 -0700 Subject: [PATCH 11/52] fix(bin): remove vestigial dispatch selector (#1026) * remove vestigial dispatch selector * no-mistakes(review): Synchronize isolation proof and portable shard evidence * no-mistakes(review): Correct shard history and proof archive date * no-mistakes(review): Remove reintroduced selector documentation reference * no-mistakes(document): Remove stale dispatch strategy documentation --- .agents/skills/bootstrap-diagnostics/SKILL.md | 2 +- AGENTS.md | 1 - bin/fm-test-run.sh | 6 +- docs/architecture.md | 1 - docs/configuration.md | 7 +- docs/examples/crew-dispatch.json | 2 +- docs/fm-test-isolation-proof.json | 210 +++++++++++-- docs/fm-test-isolation-proof.md | 72 +++-- docs/fm-test-portable-shards.md | 133 +++++---- tests/fm-instruction-owners.test.sh | 9 +- tests/fm-spawn-dispatch-profile.test.sh | 281 +----------------- tests/fm-test-isolation-proof.test.sh | 43 +++ tests/fm-test-run.test.sh | 31 ++ 13 files changed, 381 insertions(+), 417 deletions(-) diff --git a/.agents/skills/bootstrap-diagnostics/SKILL.md b/.agents/skills/bootstrap-diagnostics/SKILL.md index 641ece4ca99..2b708799415 100644 --- a/.agents/skills/bootstrap-diagnostics/SKILL.md +++ b/.agents/skills/bootstrap-diagnostics/SKILL.md @@ -27,7 +27,7 @@ When any diagnostic needs captain attention, report the plain consequence and re - `TANGLE: <remediation>` - the primary checkout is stranded on a feature branch instead of its default branch; `AGENTS.md` section 8 explains why this guard exists and what it protects. The work is safe on that branch ref; restore the primary to its default branch with the printed `git -C <root> checkout <default>`, then re-validate that branch in a proper worktree. This is the only sanctioned firstmate-initiated git write to the primary, and it is a non-destructive branch switch that strands nothing. -- `CREW_DISPATCH: invalid config/crew-dispatch.json - <reason>` - the optional dispatch profile file exists but failed low-cost bootstrap validation; stop profile-based dispatch, report the actionable error, and require correction of the malformed schema, unverified harness name, unknown selector, or invalid harness/effort pair rather than falling back around it or selecting a bad profile. +- `CREW_DISPATCH: invalid config/crew-dispatch.json - <reason>` - the optional dispatch profile file exists but failed low-cost bootstrap validation; stop profile-based dispatch, report the actionable error, and require correction of the malformed schema, unverified harness name, or invalid harness/effort pair rather than falling back around it or selecting a bad profile. - `FLEET_SYNC: <repo>: skipped: <reason>` - a benign one-off skip (offline, no origin, local-only); bootstrap continued, investigate only if it blocks work. A skip can also report the bounded fleet-refresh timeout (`FM_FLEET_SYNC_BOOTSTRAP_TIMEOUT`, or a fleet-size-aware default with a 20 second floor); a timeout never blocks startup. - `FLEET_SYNC: <repo>: recovered: <detail>` - the clone had drifted onto a clean detached HEAD holding no unique commits and the sync self-healed it (re-attached the default branch and fast-forwarded); no action needed, it is reported only so the self-heal is visible. diff --git a/AGENTS.md b/AGENTS.md index d972c591709..87dfe18f00d 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -169,7 +169,6 @@ Preserve malformed profile configuration as an actionable error rather than sele When every candidate is tight, preserve the captain's strongest-reasoning class rather than silently downgrading it solely to conserve quota; stop and report the tight choice if that class cannot proceed. Break genuine headroom ties without array-order or harness bias. `quota-axi` owns how model or product windows relate to bounding account windows; as an explicitly interim rule until successor `quota-axi-interpretation-hints-h3` lands, use the weakest applicable remaining headroom, then remove this interim rule. -`bin/fm-dispatch-select.sh` is vestigial during this transition and must not be called. The generic effort fallback and its precedence are owned by `harness-adapters`: explicit captain and standing configured effort win; otherwise use low for well-understood explicit work, xhigh for ambiguous investigation or design, intermediate levels proportionally, and never max without explicit captain preference. Do not add model-specific versions of that policy. diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index a86bb0b661a..c1e8ade8c79 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -121,7 +121,7 @@ family_for_basename() { fm-calm-pi-extension.test.sh|fm-captain-translation-contract.test.sh|fm-cd-pretool-check.test.sh|\ fm-composer-ghost.test.sh|fm-composer-lib.test.sh|\ fm-crew-state.test.sh|fm-decision-hold-lifecycle.test.sh|\ - fm-dispatch-select.test.sh|fm-documentation-audiences.test.sh|fm-ensure-agents-md.test.sh|fm-grok-harness.test.sh|\ + fm-documentation-audiences.test.sh|fm-ensure-agents-md.test.sh|fm-grok-harness.test.sh|\ fm-herdr-lab.test.sh|fm-instruction-owners.test.sh|fm-lint.test.sh|\ fm-install-herdr.test.sh|fm-nm-test-contract.test.sh|fm-no-mistakes-ownership.test.sh|\ fm-operational-input.test.sh|fm-pi-primary-types.test.sh|\ @@ -242,7 +242,6 @@ tests/fm-composer-ghost.test.sh tests/fm-composer-lib.test.sh tests/fm-crew-state.test.sh tests/fm-decision-hold-lifecycle.test.sh -tests/fm-dispatch-select.test.sh tests/fm-ensure-agents-md.test.sh tests/fm-grok-harness.test.sh tests/fm-herdr-lab.test.sh @@ -280,7 +279,6 @@ tests/fm-test-run.test.sh tests/fm-send-popup-settle.test.sh tests/fm-review-diff.test.sh tests/fm-brief.test.sh -tests/fm-dispatch-select.test.sh tests/fm-ensure-agents-md.test.sh tests/fm-instruction-owners.test.sh tests/fm-pi-primary-types.test.sh @@ -670,7 +668,7 @@ families_for_changed_path() { bin/fm-x-*|bin/fm-check*) printf '%s\n' pr-forge ;; - bin/fm-spawn.sh|bin/fm-send.sh|bin/fm-dispatch-select.sh|bin/fm-harness.sh|\ + bin/fm-spawn.sh|bin/fm-send.sh|bin/fm-harness.sh|\ bin/fm-peek.sh|bin/fm-composer*) printf '%s\n' backend-dispatch printf '%s\n' pure-contract-unit diff --git a/docs/architecture.md b/docs/architecture.md index b8c6620115d..e572757dae1 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -142,7 +142,6 @@ The intake and authority contract in `AGENTS.md` owns when separate scout resear Crewmate and scout dispatch can stay on the static crewmate harness resolved by `config/crew-harness`, or it can use local dispatch profiles in `config/crew-dispatch.json`. The dispatch file is intentionally judgment-based: firstmate reads the natural-language rules at intake, chooses the best matching rule, resolves profile arrays itself from current quota output under `AGENTS.md` section 4, and passes only concrete `--harness`, `--model`, and `--effort` axes to `fm-spawn.sh`. The shell scripts validate the JSON shape and verified harness/effort combinations, but they do not parse task intent, match natural-language rules, or own array selection. -`fm-dispatch-select.sh` is vestigial and not called during this transition; its separately based removal follows after the replacement instruction lands. The session-start bootstrap step keeps valid dispatch configuration silent unless verbose facts are enabled and surfaces a concise invalid-config line when validation fails. When the file exists, `fm-spawn.sh` refuses crewmate and scout launches without an explicit harness, so `config/crew-harness` is only automatic when no dispatch profile file is active. Secondmate launches are exempt because they resolve the secondmate harness and any optional secondmate model or effort tokens instead. diff --git a/docs/configuration.md b/docs/configuration.md index e42100a2060..fba8a6ff335 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -213,7 +213,6 @@ This section is the single owner of the canonical schema and its per-field seman "use": [ { "harness": "<adapter>", "model": "<optional model>", "effort": "<low|medium|high|xhigh|max, optional>" } ], - "select": "<optional strategy>", "why": "<optional rationale that helps firstmate choose>" } ], @@ -228,15 +227,13 @@ Both `use` and the optional top-level `default` accept either one profile object The single-object form stays fully backward-compatible, and every profile needs `harness`. Profile `model` and `effort` fields and rule `why` are optional. An omitted model or effort means the selected harness uses its own default for that axis. -Every profile array is an implicit quota-aware choice and does not need a selector property. -`select: "quota-balanced"` remains accepted on rules for configuration compatibility and routes to the same agent-judged array procedure. +Every profile array is an implicit quota-aware choice. If no dispatch rule fits, firstmate resolves `default` through the same object-or-array path before falling back to `config/crew-harness`. If a selected profile carries an effort value the chosen harness does not accept, `fm-spawn.sh` records the requested `effort=` in task meta for traceability but omits the launch flag, and bootstrap reports the invalid harness/effort pair as a `CREW_DISPATCH` diagnostic when it is visible in the file. -`bin/fm-dispatch-select.sh` is vestigial during the instruction transition and must not be called; a separately based follow-up removes it after the replacement operating contract lands. See [`docs/examples/crew-dispatch.json`](examples/crew-dispatch.json) for a starting point to copy into local `config/crew-dispatch.json`. When the file exists, bootstrap validates it with `jq`. Valid files stay silent by default; with `FM_BOOTSTRAP_VERBOSE_FACTS=1`, bootstrap emits `BOOTSTRAP_INFO: crew dispatch active config/crew-dispatch.json`, one `BOOTSTRAP_INFO:` fact per rule, and one fact for the optional default profile set. -Malformed JSON, an empty or malformed rule/default array, an unverified harness, an unknown `select`, or an effort value unsupported by that harness is reported as `CREW_DISPATCH: invalid config/crew-dispatch.json - ...`; missing `jq` is reported through the normal `MISSING: jq` install-consent flow. +Malformed JSON, an empty or malformed rule/default array, an unverified harness, or an effort value unsupported by that harness is reported as `CREW_DISPATCH: invalid config/crew-dispatch.json - ...`; missing `jq` is reported through the normal `MISSING: jq` install-consent flow. While the file remains present, no crewmate or scout spawn may proceed without an explicit resolved harness; malformed configuration must be reported and corrected rather than selected around. Secondmate homes inherit this file from the primary, so a secondmate's own crewmates apply the same dispatch profile behavior. diff --git a/docs/examples/crew-dispatch.json b/docs/examples/crew-dispatch.json index 23a5391d20a..4c8fc36993c 100644 --- a/docs/examples/crew-dispatch.json +++ b/docs/examples/crew-dispatch.json @@ -16,7 +16,7 @@ { "harness": "claude", "model": "claude-sonnet-5", "effort": "high" }, { "harness": "codex", "model": "gpt-5.5", "effort": "high" } ], - "why": "Firstmate compares every candidate with current relevant quota and pace before dispatch, so use a strong coding profile." + "why": "Firstmate compares every candidate with current relevant quota before dispatch, so use a strong coding profile." } ], "default": [ diff --git a/docs/fm-test-isolation-proof.json b/docs/fm-test-isolation-proof.json index ec605bf10f2..92e227c075f 100644 --- a/docs/fm-test-isolation-proof.json +++ b/docs/fm-test-isolation-proof.json @@ -1,36 +1,190 @@ { "concurrency": 4, - "finished_at": "2026-07-29T23:21:46Z", + "finished_at": "2026-07-25T08:44:54Z", "fm_test_run_jobs_enabled": false, "kind": "isolation-proof", "production_sharding_enabled": false, - "run_id": "fm-isolation-1785367157179-18165", + "run_id": "fm-isolation-1784968984050-13742", "scripts": [ - {"duration_ms": 46788, "exit": 0, "path": "tests/fm-arm-pretool-check.test.sh", "worker": 1}, - {"duration_ms": 48294, "exit": 0, "path": "tests/fm-backend-herdr.test.sh", "worker": 2}, - {"duration_ms": 2224, "exit": 0, "path": "tests/fm-brief.test.sh", "worker": 3}, - {"duration_ms": 34207, "exit": 0, "path": "tests/fm-cd-pretool-check.test.sh", "worker": 4}, - {"duration_ms": 9065, "exit": 0, "path": "tests/fm-composer-ghost.test.sh", "worker": 5}, - {"duration_ms": 64, "exit": 0, "path": "tests/fm-composer-lib.test.sh", "worker": 6}, - {"duration_ms": 25365, "exit": 0, "path": "tests/fm-crew-state.test.sh", "worker": 7}, - {"duration_ms": 30771, "exit": 0, "path": "tests/fm-decision-hold-lifecycle.test.sh", "worker": 8}, - {"duration_ms": 581, "exit": 0, "path": "tests/fm-ensure-agents-md.test.sh", "worker": 9}, - {"duration_ms": 6251, "exit": 0, "path": "tests/fm-grok-harness.test.sh", "worker": 10}, - {"duration_ms": 15422, "exit": 0, "path": "tests/fm-herdr-lab.test.sh", "worker": 11}, - {"duration_ms": 5237, "exit": 0, "path": "tests/fm-lint.test.sh", "worker": 12}, - {"duration_ms": 2945, "exit": 0, "path": "tests/fm-pi-primary-types.test.sh", "worker": 13}, - {"duration_ms": 8564, "exit": 0, "path": "tests/fm-pr-merge.test.sh", "worker": 14}, - {"duration_ms": 2875, "exit": 0, "path": "tests/fm-review-diff.test.sh", "worker": 15}, - {"duration_ms": 5644, "exit": 0, "path": "tests/fm-send-popup-settle.test.sh", "worker": 16}, - {"duration_ms": 2911, "exit": 0, "path": "tests/fm-send-settle.test.sh", "worker": 17}, - {"duration_ms": 2747, "exit": 0, "path": "tests/fm-send-strict.test.sh", "worker": 18}, - {"duration_ms": 855, "exit": 0, "path": "tests/fm-spawn-batch.test.sh", "worker": 19}, - {"duration_ms": 703, "exit": 0, "path": "tests/fm-supervision-instructions.test.sh", "worker": 20}, - {"duration_ms": 15674, "exit": 0, "path": "tests/fm-test-run.test.sh", "worker": 21}, - {"duration_ms": 4816, "exit": 0, "path": "tests/fm-tmux-submit-busy.test.sh", "worker": 22}, - {"duration_ms": 248, "exit": 0, "path": "tests/fm-transition-lib.test.sh", "worker": 23}, - {"duration_ms": 52939, "exit": 0, "path": "tests/fm-x-mode.test.sh", "worker": 24} + { + "duration_ms": 26535, + "exit": 0, + "path": "tests/fm-arm-pretool-check.test.sh", + "worker": 1 + }, + { + "duration_ms": 29446, + "exit": 0, + "path": "tests/fm-backend-herdr.test.sh", + "worker": 2 + }, + { + "duration_ms": 973, + "exit": 0, + "path": "tests/fm-brief.test.sh", + "worker": 3 + }, + { + "duration_ms": 181, + "exit": 0, + "path": "tests/fm-captain-translation-contract.test.sh", + "worker": 4 + }, + { + "duration_ms": 17218, + "exit": 0, + "path": "tests/fm-cd-pretool-check.test.sh", + "worker": 5 + }, + { + "duration_ms": 1810, + "exit": 0, + "path": "tests/fm-composer-ghost.test.sh", + "worker": 6 + }, + { + "duration_ms": 66, + "exit": 0, + "path": "tests/fm-composer-lib.test.sh", + "worker": 7 + }, + { + "duration_ms": 15250, + "exit": 0, + "path": "tests/fm-crew-state.test.sh", + "worker": 8 + }, + { + "duration_ms": 18509, + "exit": 0, + "path": "tests/fm-decision-hold-lifecycle.test.sh", + "worker": 9 + }, + { + "duration_ms": 358, + "exit": 0, + "path": "tests/fm-ensure-agents-md.test.sh", + "worker": 10 + }, + { + "duration_ms": 5276, + "exit": 0, + "path": "tests/fm-grok-harness.test.sh", + "worker": 11 + }, + { + "duration_ms": 11199, + "exit": 0, + "path": "tests/fm-herdr-lab.test.sh", + "worker": 12 + }, + { + "duration_ms": 297, + "exit": 0, + "path": "tests/fm-instruction-owners.test.sh", + "worker": 13 + }, + { + "duration_ms": 4882, + "exit": 0, + "path": "tests/fm-lint.test.sh", + "worker": 14 + }, + { + "duration_ms": 180, + "exit": 0, + "path": "tests/fm-nm-test-contract.test.sh", + "worker": 15 + }, + { + "duration_ms": 35, + "exit": 0, + "path": "tests/fm-no-mistakes-ownership.test.sh", + "worker": 16 + }, + { + "duration_ms": 1842, + "exit": 0, + "path": "tests/fm-pi-primary-types.test.sh", + "worker": 17 + }, + { + "duration_ms": 6630, + "exit": 0, + "path": "tests/fm-pr-merge.test.sh", + "worker": 18 + }, + { + "duration_ms": 2410, + "exit": 0, + "path": "tests/fm-review-diff.test.sh", + "worker": 19 + }, + { + "duration_ms": 4496, + "exit": 0, + "path": "tests/fm-send-popup-settle.test.sh", + "worker": 20 + }, + { + "duration_ms": 2179, + "exit": 0, + "path": "tests/fm-send-settle.test.sh", + "worker": 21 + }, + { + "duration_ms": 1390, + "exit": 0, + "path": "tests/fm-send-strict.test.sh", + "worker": 22 + }, + { + "duration_ms": 626, + "exit": 0, + "path": "tests/fm-spawn-batch.test.sh", + "worker": 23 + }, + { + "duration_ms": 52, + "exit": 0, + "path": "tests/fm-stow-contract.test.sh", + "worker": 24 + }, + { + "duration_ms": 336, + "exit": 0, + "path": "tests/fm-supervision-instructions.test.sh", + "worker": 25 + }, + { + "duration_ms": 8900, + "exit": 0, + "path": "tests/fm-test-run.test.sh", + "worker": 26 + }, + { + "duration_ms": 1845, + "exit": 0, + "path": "tests/fm-tmux-submit-busy.test.sh", + "worker": 27 + }, + { + "duration_ms": 96, + "exit": 0, + "path": "tests/fm-transition-lib.test.sh", + "worker": 28 + }, + { + "duration_ms": 34920, + "exit": 0, + "path": "tests/fm-x-mode.test.sh", + "worker": 29 + } ], - "started_at": "2026-07-29T23:19:17Z", - "summary": {"duration_ms": 149010, "failed": 0, "total": 24} + "started_at": "2026-07-25T08:43:04Z", + "summary": { + "duration_ms": 110623, + "failed": 0, + "total": 29 + } } diff --git a/docs/fm-test-isolation-proof.md b/docs/fm-test-isolation-proof.md index 7d5db970ae7..19e4b6a516c 100644 --- a/docs/fm-test-isolation-proof.md +++ b/docs/fm-test-isolation-proof.md @@ -16,16 +16,16 @@ The archived proof JSON below still records the Phase 2 proof-time flags (`produ | Field | Value | |---|---| -| `run_id` | `fm-isolation-1784693155237-99474` | -| `started_at` | `2026-07-22T04:05:55Z` | -| `finished_at` | `2026-07-22T04:08:06Z` | +| `run_id` | `fm-isolation-1784968984050-13742` | +| `started_at` | `2026-07-25T08:43:04Z` | +| `finished_at` | `2026-07-25T08:44:54Z` | | concurrency | **4** | -| candidates | **30** | +| candidates | **29** | | failed | **0** | -| wall duration_ms | **131001** (~131.0s) | +| wall duration_ms | **110623** (~110.6s) | | `production_sharding_enabled` | `False` | | `fm_test_run_jobs_enabled` | `False` | -| host proof date | 2026-07-22 (UTC day of archive write) | +| host proof date | 2026-07-25 (UTC day of archive write) | Isolation checks that passed with this run: @@ -48,7 +48,6 @@ Sorted paths as selected by `bin/fm-test-isolation-proof.sh --list` at proof tim - `tests/fm-composer-lib.test.sh` - `tests/fm-crew-state.test.sh` - `tests/fm-decision-hold-lifecycle.test.sh` -- `tests/fm-dispatch-select.test.sh` - `tests/fm-ensure-agents-md.test.sh` - `tests/fm-grok-harness.test.sh` - `tests/fm-herdr-lab.test.sh` @@ -74,36 +73,35 @@ Sorted paths as selected by `bin/fm-test-isolation-proof.sh --list` at proof tim | duration_ms | exit | worker | script | |---:|---:|---:|---| -| 38449 | 0 | 30 | `tests/fm-x-mode.test.sh` | -| 35417 | 0 | 2 | `tests/fm-backend-herdr.test.sh` | -| 29102 | 0 | 1 | `tests/fm-arm-pretool-check.test.sh` | -| 21133 | 0 | 9 | `tests/fm-decision-hold-lifecycle.test.sh` | -| 19896 | 0 | 8 | `tests/fm-crew-state.test.sh` | -| 18610 | 0 | 5 | `tests/fm-cd-pretool-check.test.sh` | -| 12517 | 0 | 13 | `tests/fm-herdr-lab.test.sh` | -| 8939 | 0 | 19 | `tests/fm-pr-merge.test.sh` | -| 6953 | 0 | 21 | `tests/fm-send-popup-settle.test.sh` | -| 5963 | 0 | 12 | `tests/fm-grok-harness.test.sh` | -| 4645 | 0 | 27 | `tests/fm-test-run.test.sh` | -| 3524 | 0 | 22 | `tests/fm-send-settle.test.sh` | -| 2803 | 0 | 6 | `tests/fm-composer-ghost.test.sh` | -| 2552 | 0 | 28 | `tests/fm-tmux-submit-busy.test.sh` | -| 2549 | 0 | 20 | `tests/fm-review-diff.test.sh` | -| 1551 | 0 | 23 | `tests/fm-send-strict.test.sh` | -| 1274 | 0 | 15 | `tests/fm-lint.test.sh` | -| 1056 | 0 | 18 | `tests/fm-pi-primary-types.test.sh` | -| 897 | 0 | 3 | `tests/fm-brief.test.sh` | -| 874 | 0 | 10 | `tests/fm-dispatch-select.test.sh` | -| 684 | 0 | 24 | `tests/fm-spawn-batch.test.sh` | -| 348 | 0 | 11 | `tests/fm-ensure-agents-md.test.sh` | -| 283 | 0 | 26 | `tests/fm-supervision-instructions.test.sh` | -| 232 | 0 | 14 | `tests/fm-instruction-owners.test.sh` | -| 201 | 0 | 16 | `tests/fm-nm-test-contract.test.sh` | -| 104 | 0 | 29 | `tests/fm-transition-lib.test.sh` | -| 90 | 0 | 4 | `tests/fm-captain-translation-contract.test.sh` | -| 68 | 0 | 7 | `tests/fm-composer-lib.test.sh` | -| 57 | 0 | 25 | `tests/fm-stow-contract.test.sh` | -| 36 | 0 | 17 | `tests/fm-no-mistakes-ownership.test.sh` | +| 34920 | 0 | 29 | `tests/fm-x-mode.test.sh` | +| 29446 | 0 | 2 | `tests/fm-backend-herdr.test.sh` | +| 26535 | 0 | 1 | `tests/fm-arm-pretool-check.test.sh` | +| 18509 | 0 | 9 | `tests/fm-decision-hold-lifecycle.test.sh` | +| 17218 | 0 | 5 | `tests/fm-cd-pretool-check.test.sh` | +| 15250 | 0 | 8 | `tests/fm-crew-state.test.sh` | +| 11199 | 0 | 12 | `tests/fm-herdr-lab.test.sh` | +| 8900 | 0 | 26 | `tests/fm-test-run.test.sh` | +| 6630 | 0 | 18 | `tests/fm-pr-merge.test.sh` | +| 5276 | 0 | 11 | `tests/fm-grok-harness.test.sh` | +| 4882 | 0 | 14 | `tests/fm-lint.test.sh` | +| 4496 | 0 | 20 | `tests/fm-send-popup-settle.test.sh` | +| 2410 | 0 | 19 | `tests/fm-review-diff.test.sh` | +| 2179 | 0 | 21 | `tests/fm-send-settle.test.sh` | +| 1845 | 0 | 27 | `tests/fm-tmux-submit-busy.test.sh` | +| 1842 | 0 | 17 | `tests/fm-pi-primary-types.test.sh` | +| 1810 | 0 | 6 | `tests/fm-composer-ghost.test.sh` | +| 1390 | 0 | 22 | `tests/fm-send-strict.test.sh` | +| 973 | 0 | 3 | `tests/fm-brief.test.sh` | +| 626 | 0 | 23 | `tests/fm-spawn-batch.test.sh` | +| 358 | 0 | 10 | `tests/fm-ensure-agents-md.test.sh` | +| 336 | 0 | 25 | `tests/fm-supervision-instructions.test.sh` | +| 297 | 0 | 13 | `tests/fm-instruction-owners.test.sh` | +| 181 | 0 | 4 | `tests/fm-captain-translation-contract.test.sh` | +| 180 | 0 | 15 | `tests/fm-nm-test-contract.test.sh` | +| 96 | 0 | 28 | `tests/fm-transition-lib.test.sh` | +| 66 | 0 | 7 | `tests/fm-composer-lib.test.sh` | +| 52 | 0 | 24 | `tests/fm-stow-contract.test.sh` | +| 35 | 0 | 16 | `tests/fm-no-mistakes-ownership.test.sh` | ## Audit notes (why this set) diff --git a/docs/fm-test-portable-shards.md b/docs/fm-test-portable-shards.md index 0bfa5e6bee4..ce153cbd74e 100644 --- a/docs/fm-test-portable-shards.md +++ b/docs/fm-test-portable-shards.md @@ -1,67 +1,89 @@ -# Firstmate portable test shards +# Firstmate portable test shards (Phase 4) -`bin/fm-test-run.sh` owns portable lane composition and execution. -`bin/fm-test-isolation-proof.sh` owns the proven-isolated candidate set. +This document records how the two portable parallel CI shards were balanced from measured evidence. +Composition and execution are owned by `bin/fm-test-run.sh` (`--lane portable-parallel-1` / `portable-parallel-2` / `portable-serial`). +The proven-isolated candidate set remains owned by `bin/fm-test-isolation-proof.sh`. -## Verification inputs +## Inputs -The current candidate timings came from the 2026-07-29 concurrent proof recorded in [fm-test-isolation-proof.md](fm-test-isolation-proof.md). -The proof ran 24 candidates with four workers and no failures. +| Input | Owner / source | +|---|---| +| Proven-isolated set (29 scripts) | `bin/fm-test-isolation-proof.sh --list` and `docs/fm-test-isolation-proof.md` | +| Phase 1 serial durations | CI timing artifacts `fm-test-timing` from main after #825 / #832 / #834 | +| Real-Herdr family | `bin/fm-test-run.sh --family real-herdr-gated` (dedicated required CI lane) | -| duration_ms | script | +Phase 1 averages used for balance (mean of available serial `duration_ms` across those artifacts): + +| duration_ms (avg) | script | |---:|---| -| 52939 | `tests/fm-x-mode.test.sh` | -| 48294 | `tests/fm-backend-herdr.test.sh` | -| 46788 | `tests/fm-arm-pretool-check.test.sh` | -| 34207 | `tests/fm-cd-pretool-check.test.sh` | -| 30771 | `tests/fm-decision-hold-lifecycle.test.sh` | -| 25365 | `tests/fm-crew-state.test.sh` | -| 15674 | `tests/fm-test-run.test.sh` | -| 15422 | `tests/fm-herdr-lab.test.sh` | -| 9065 | `tests/fm-composer-ghost.test.sh` | -| 8564 | `tests/fm-pr-merge.test.sh` | -| 6251 | `tests/fm-grok-harness.test.sh` | -| 5644 | `tests/fm-send-popup-settle.test.sh` | -| 5237 | `tests/fm-lint.test.sh` | -| 4816 | `tests/fm-tmux-submit-busy.test.sh` | -| 2945 | `tests/fm-pi-primary-types.test.sh` | -| 2911 | `tests/fm-send-settle.test.sh` | -| 2875 | `tests/fm-review-diff.test.sh` | -| 2747 | `tests/fm-send-strict.test.sh` | -| 2224 | `tests/fm-brief.test.sh` | -| 855 | `tests/fm-spawn-batch.test.sh` | -| 703 | `tests/fm-supervision-instructions.test.sh` | -| 581 | `tests/fm-ensure-agents-md.test.sh` | -| 248 | `tests/fm-transition-lib.test.sh` | -| 64 | `tests/fm-composer-lib.test.sh` | - -## Parallel lanes - -The two parallel lanes use longest-processing-time assignment from those measured durations. - -| Lane | Script count | Estimated duration | +| 29639 | `tests/fm-arm-pretool-check.test.sh` | +| 25402 | `tests/fm-decision-hold-lifecycle.test.sh` | +| 19428 | `tests/fm-x-mode.test.sh` | +| 14979 | `tests/fm-cd-pretool-check.test.sh` | +| 9339 | `tests/fm-backend-herdr.test.sh` | +| 6885 | `tests/fm-herdr-lab.test.sh` | +| 5127 | `tests/fm-crew-state.test.sh` | +| 4044 | `tests/fm-pr-merge.test.sh` | +| 3922 | `tests/fm-grok-harness.test.sh` | +| 2492 | `tests/fm-test-run.test.sh` | +| 1901 | `tests/fm-send-popup-settle.test.sh` | +| 1234 | `tests/fm-spawn-batch.test.sh` | +| 851 | `tests/fm-send-strict.test.sh` | +| 791 | `tests/fm-review-diff.test.sh` | +| 627 | `tests/fm-tmux-submit-busy.test.sh` | +| 525 | `tests/fm-brief.test.sh` | +| 321 | `tests/fm-composer-ghost.test.sh` | +| 276 | `tests/fm-send-settle.test.sh` | +| 189 | `tests/fm-ensure-agents-md.test.sh` | +| 175 | `tests/fm-supervision-instructions.test.sh` | +| 138 | `tests/fm-instruction-owners.test.sh` | +| 133 | `tests/fm-lint.test.sh` | +| 108 | `tests/fm-pi-primary-types.test.sh` | +| 106 | `tests/fm-nm-test-contract.test.sh` | +| 67 | `tests/fm-transition-lib.test.sh` | +| 64 | `tests/fm-captain-translation-contract.test.sh` | +| 48 | `tests/fm-composer-lib.test.sh` | +| 36 | `tests/fm-stow-contract.test.sh` | +| 28 | `tests/fm-no-mistakes-ownership.test.sh` | + +## Balancing history + +The original 30-script set used longest-processing-time (LPT) assignment onto two workers with the Phase 1 averages above. +The current 29-script lanes retain that assignment after one 283 ms candidate was removed from `portable-parallel-1`. +The current totals are therefore intentionally not a fresh LPT balance of the 29-script set. +Do not rebalance alphabetically or by family intuition. +Shard execution order remains longest-first within each retained lane. + +| Lane | Script count | Sum of Phase 1 averages | |---|---:|---:| -| `portable-parallel-1` | 11 | 162436 ms (~162.4 s) | -| `portable-parallel-2` | 13 | 162754 ms (~162.8 s) | -| imbalance | | 318 ms | +| `portable-parallel-1` | 14 | 64296 ms (~64.3 s) | +| `portable-parallel-2` | 15 | 64579 ms (~64.6 s) | +| imbalance | | 283 ms | -`bin/fm-test-run.sh` contains the exact ordered memberships in `list_portable_parallel_1` and `list_portable_parallel_2`. +Exact ordered membership is the heredoc lists in `bin/fm-test-run.sh` (`list_portable_parallel_1` / `list_portable_parallel_2`). ## Portable serial remainder -`portable-serial` includes every `tests/*.test.sh` that is neither proven-isolated nor `real-herdr-gated`. -It keeps watcher, lock, AFK, real tmux, daemon, secondmate lifecycle, bootstrap, live-harness opt-in, GUI-backend, and other unproven work serial. +`portable-serial` is every `tests/*.test.sh` that is neither proven-isolated nor `real-herdr-gated`. +That keeps watcher, lock, AFK, real tmux, daemon, secondmate lifecycle, bootstrap, live-harness opt-in (default skip), GUI backends, and other stateful or unproven work serial. +Measured serial remainder wall (from the same Phase 1 artifacts, excluding Herdr) is about **13 minutes**. ## Coverage guard -`bin/fm-test-run.sh --check-coverage` verifies that both parallel lanes partition the proven-isolated set. -It also verifies that the parallel lanes, portable serial lane, and real-Herdr family are disjoint and cover every `tests/*.test.sh` script. +`bin/fm-test-run.sh --check-coverage` proves: + +1. The two portable parallel shards are a partition of the proven-isolated set. +2. Proven-isolated embeds match `bin/fm-test-isolation-proof.sh --list`. +3. Union of portable parallel shards + portable serial + real-Herdr family equals the complete `tests/*.test.sh` inventory. +4. Those four partitions are pairwise disjoint (no missing scripts, no duplicates). + +CI runs that guard as a required job (`test-coverage`). ## Timing artifacts -Portable shards, the portable serial lane, and the Herdr lane upload runner-generated timing JSON. -`bin/fm-test-run.sh --aggregate-json` creates the combined summary artifact. -`.github/workflows/ci.yml` owns the exact artifact names and aggregation wiring. +Every portable shard, the portable serial lane, and the Herdr lane upload their runner-generated timing JSON even when the behavior run reports failures. +The dependent aggregate job runs after all four lanes, combines every available lane JSON through `bin/fm-test-run.sh --aggregate-json`, and uploads one summary artifact for critical-path review. +The workflow in `.github/workflows/ci.yml` owns the exact artifact names and aggregation wiring. ## Local entry points @@ -72,8 +94,15 @@ Portable shards, the portable serial lane, and the Herdr lane upload runner-gene | Job | timeout-minutes | Rationale | |---|---:|---| -| portable parallel 1/2 | 10 | The measured shard sums are about three minutes and the timeout is a hang tripwire. | -| portable serial | 20 | The serial remainder needs a larger hang tripwire. | -| Herdr | 40 | The real-Herdr lane keeps its dedicated timeout. | +| portable parallel 1/2 | 10 | Measured shard sum ~1 min; hang tripwire with margin | +| portable serial | 20 | Measured ~13 min remainder; reduced from interim 25m full-portable slack after sharding | +| Herdr | 40 | Unchanged hang tripwire for the real-Herdr lane | + +Timeouts remain hang tripwires, not expected healthy ends of green suites. +Do not raise them as a substitute for green results, retries, or weaker assertions. + +## What this phase does not do -Timeouts are hang tripwires rather than expected healthy durations. +- Does not expand the proven-isolated set without a new concurrent isolation proof. +- Does not parallelize watcher, AFK, real Herdr, real tmux, or other stateful families. +- Does not start rollout verification; that waits until this PR is green and merged. diff --git a/tests/fm-instruction-owners.test.sh b/tests/fm-instruction-owners.test.sh index bcbf671c38e..f4fffb8cbe4 100755 --- a/tests/fm-instruction-owners.test.sh +++ b/tests/fm-instruction-owners.test.sh @@ -116,8 +116,7 @@ test_agent_owned_quota_array_dispatch_contract() { "preserve the captain's strongest-reasoning class rather than silently downgrading it" \ 'Break genuine headroom ties without array-order or harness bias' \ '`quota-axi` owns how model or product windows relate to bounding account windows' \ - 'explicitly interim rule until successor `quota-axi-interpretation-hints-h3` lands' \ - '`bin/fm-dispatch-select.sh` is vestigial during this transition and must not be called'; do + 'explicitly interim rule until successor `quota-axi-interpretation-hints-h3` lands'; do assert_grep "$phrase" "$AGENTS" "array-dispatch contract lost '$phrase'" done @@ -135,16 +134,10 @@ test_agent_owned_quota_array_dispatch_contract() { "model discovery guidance permits a fixed provider table" assert_grep '`AGENTS.md` section 4 owns the dispatch and array-selection procedure.' "$CONFIG" \ "configuration docs do not point to the agent-owned array procedure" - assert_grep '`bin/fm-dispatch-select.sh` is vestigial during the instruction transition and must not be called' "$CONFIG" \ - "configuration docs still permit the vestigial selector" assert_grep 'quota-axi is required for the' "$BOOTSTRAP" \ "bootstrap docs lost the quota-axi dependency pointer" assert_grep 'agent-owned dispatch-profile array procedure in AGENTS.md section 4.' "$BOOTSTRAP" \ "bootstrap docs do not point to the agent-owned array procedure" - assert_no_grep 'every crew-dispatch profile array calls it automatically' "$BOOTSTRAP" \ - "bootstrap docs still claim automatic selector invocation" - assert_no_grep 'OS-backed random selection across' "$BOOTSTRAP" \ - "bootstrap docs still promise quota-unavailable random fallback" pass "firstmate directly compares every quota candidate with authoritative model discovery" } diff --git a/tests/fm-spawn-dispatch-profile.test.sh b/tests/fm-spawn-dispatch-profile.test.sh index e5f017608dc..4f587a9a398 100755 --- a/tests/fm-spawn-dispatch-profile.test.sh +++ b/tests/fm-spawn-dispatch-profile.test.sh @@ -42,7 +42,7 @@ esac exit 0 SH chmod +x "$fakebin/tmux" - fm_fake_exit0 "$fakebin" treehouse pi-signed + fm_fake_exit0 "$fakebin" treehouse printf '%s\n' "$fakebin" } @@ -84,15 +84,10 @@ run_spawn() { local home=$1 wt=$2 fakebin=$3 launchlog=$4 shift 4 : > "$launchlog" - # CLAUDE_CONFIG_DIR is forwarded onto claude launches by fm-spawn, so pin it - # explicitly (empty by default) instead of leaking the invoking shell's value, - # which would make launch assertions depend on the developer's environment. - # A test opts in to the set case via FM_TEST_CLAUDE_CONFIG_DIR. FM_ROOT_OVERRIDE='' FM_HOME="$home" \ FM_STATE_OVERRIDE="$home/state" FM_DATA_OVERRIDE="$home/data" \ FM_PROJECTS_OVERRIDE="$home/projects" FM_CONFIG_OVERRIDE="$home/config" \ FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$wt" TMUX="fake,1,0" \ - CLAUDE_CONFIG_DIR="${FM_TEST_CLAUDE_CONFIG_DIR:-}" \ FM_FAKE_LAUNCH_LOG="$launchlog" GROK_HOME="$home/grok-home" PATH="$fakebin:$PATH" \ "$SPAWN" "$@" 2>&1 } @@ -128,153 +123,6 @@ test_no_profile_keeps_claude_profile_defaults() { pass "no --model/--effort records defaults and types the claude launch instructions" } -test_relative_home_overrides_launch_with_absolute_cross_process_paths() { - local rec id out status launch home_real - id=profile-relative-paths-z1b - rec=$(make_spawn_case profile-relative-paths pi "$id") - read_case_record "$rec" - home_real=$(cd "$HOME_DIR" && pwd -P) - mkdir -p "$CASE_DIR/cdpath/home/state" "$CASE_DIR/cdpath/home/data" - : > "$LAUNCH_LOG" - - out=$( - cd "$CASE_DIR" || exit 1 - CDPATH="$CASE_DIR/cdpath" FM_ROOT_OVERRIDE='' FM_HOME=home \ - FM_STATE_OVERRIDE=home/state FM_DATA_OVERRIDE=home/data \ - FM_PROJECTS_OVERRIDE=home/projects FM_CONFIG_OVERRIDE=home/config \ - FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$WT_DIR" TMUX="fake,1,0" \ - CLAUDE_CONFIG_DIR='' FM_FAKE_LAUNCH_LOG="$LAUNCH_LOG" \ - GROK_HOME=home/grok-home PATH="$FAKEBIN_DIR:$PATH" \ - "$SPAWN" "$id" "$PROJ_DIR" 2>&1 - ) - status=$? - expect_code 0 "$status" "spawn with relative home overrides should succeed" - launch=$(cat "$LAUNCH_LOG") - assert_contains "$launch" "-e '$home_real/state/$id.pi-ext.ts'" \ - "relative FM_STATE_OVERRIDE leaked into Pi's cross-process extension path" - assert_contains "$launch" "< '$home_real/data/$id/brief.md'" \ - "relative FM_DATA_OVERRIDE leaked into the cross-process brief path" - pass "relative home overrides ignore CDPATH and become absolute before spawn launch construction" -} - -test_home_defaults_preserve_absolute_or_resolve_relative_paths() { - local rec relative_id absolute_id out status launch home_real linked_home - relative_id=profile-relative-home-defaults-z1c - absolute_id=profile-absolute-home-defaults-z1d - rec=$(make_spawn_case profile-home-defaults pi "$relative_id" "$absolute_id") - read_case_record "$rec" - home_real=$(cd "$HOME_DIR" && pwd -P) - - : > "$LAUNCH_LOG" - out=$( - cd "$CASE_DIR" || exit 1 - FM_ROOT_OVERRIDE='' FM_HOME=home \ - FM_STATE_OVERRIDE='' FM_DATA_OVERRIDE='' \ - FM_PROJECTS_OVERRIDE=home/projects FM_CONFIG_OVERRIDE=home/config \ - FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$WT_DIR" TMUX="fake,1,0" \ - CLAUDE_CONFIG_DIR='' FM_FAKE_LAUNCH_LOG="$LAUNCH_LOG" \ - GROK_HOME=home/grok-home PATH="$FAKEBIN_DIR:$PATH" \ - "$SPAWN" "$relative_id" "$PROJ_DIR" 2>&1 - ) - status=$? - expect_code 0 "$status" "spawn with relative FM_HOME defaults should succeed" - launch=$(cat "$LAUNCH_LOG") - assert_contains "$launch" "-e '$home_real/state/$relative_id.pi-ext.ts'" \ - "relative FM_HOME leaked into Pi's default cross-process extension path" - assert_contains "$launch" "< '$home_real/data/$relative_id/brief.md'" \ - "relative FM_HOME leaked into the default cross-process brief path" - - linked_home="$CASE_DIR/home-link" - ln -s "$HOME_DIR" "$linked_home" - : > "$LAUNCH_LOG" - out=$( - FM_ROOT_OVERRIDE='' FM_HOME="$linked_home" \ - FM_STATE_OVERRIDE='' FM_DATA_OVERRIDE='' \ - FM_PROJECTS_OVERRIDE="$linked_home/projects" FM_CONFIG_OVERRIDE="$linked_home/config" \ - FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$WT_DIR" TMUX="fake,1,0" \ - CLAUDE_CONFIG_DIR='' FM_FAKE_LAUNCH_LOG="$LAUNCH_LOG" \ - GROK_HOME="$linked_home/grok-home" PATH="$FAKEBIN_DIR:$PATH" \ - "$SPAWN" "$absolute_id" "$PROJ_DIR" 2>&1 - ) - status=$? - expect_code 0 "$status" "spawn with absolute symlink-spelled FM_HOME defaults should succeed" - launch=$(cat "$LAUNCH_LOG") - assert_contains "$launch" "-e '$linked_home/state/$absolute_id.pi-ext.ts'" \ - "absolute FM_HOME spelling changed in Pi's default cross-process extension path" - assert_contains "$launch" "< '$linked_home/data/$absolute_id/brief.md'" \ - "absolute FM_HOME spelling changed in the default cross-process brief path" - pass "FM_HOME defaults resolve relative paths and preserve absolute spellings" -} - -test_absolute_override_spelling_is_preserved_in_launch_paths() { - local rec id out status launch linked_home - id=profile-absolute-paths-z1c - rec=$(make_spawn_case profile-absolute-paths pi "$id") - read_case_record "$rec" - linked_home="$CASE_DIR/home-link" - ln -s "$HOME_DIR" "$linked_home" - : > "$LAUNCH_LOG" - - out=$( - FM_ROOT_OVERRIDE='' FM_HOME="$linked_home" \ - FM_STATE_OVERRIDE="$linked_home/state" FM_DATA_OVERRIDE="$linked_home/data" \ - FM_PROJECTS_OVERRIDE="$linked_home/projects" FM_CONFIG_OVERRIDE="$linked_home/config" \ - FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$WT_DIR" TMUX="fake,1,0" \ - CLAUDE_CONFIG_DIR='' FM_FAKE_LAUNCH_LOG="$LAUNCH_LOG" \ - GROK_HOME="$linked_home/grok-home" PATH="$FAKEBIN_DIR:$PATH" \ - "$SPAWN" "$id" "$PROJ_DIR" 2>&1 - ) - status=$? - expect_code 0 "$status" "spawn with absolute symlink-spelled overrides should succeed" - launch=$(cat "$LAUNCH_LOG") - assert_contains "$launch" "-e '$linked_home/state/$id.pi-ext.ts'" \ - "absolute FM_STATE_OVERRIDE spelling changed in Pi's cross-process extension path" - assert_contains "$launch" "< '$linked_home/data/$id/brief.md'" \ - "absolute FM_DATA_OVERRIDE spelling changed in the cross-process brief path" - pass "absolute override spellings are preserved in spawn launch paths" -} - -test_unresolvable_relative_overrides_fail_loudly() { - local rec id out status - id=profile-unresolvable-paths-z1d - rec=$(make_spawn_case profile-unresolvable-paths pi "$id") - read_case_record "$rec" - - out=$( - cd "$CASE_DIR" || exit 1 - FM_ROOT_OVERRIDE='' FM_HOME=missing-home \ - FM_STATE_OVERRIDE='' FM_DATA_OVERRIDE='' \ - "$SPAWN" "$id" "$PROJ_DIR" 2>&1 - ) - status=$? - expect_code 1 "$status" "spawn with an unresolvable relative home should fail" - assert_contains "$out" "FM_HOME directory cannot be resolved: missing-home" \ - "spawn did not name the unresolvable FM_HOME" - - out=$( - cd "$CASE_DIR" || exit 1 - FM_ROOT_OVERRIDE='' FM_HOME=home \ - FM_STATE_OVERRIDE=missing-state FM_DATA_OVERRIDE=home/data \ - "$SPAWN" "$id" "$PROJ_DIR" 2>&1 - ) - status=$? - expect_code 1 "$status" "spawn with an unresolvable relative state override should fail" - assert_contains "$out" "FM_STATE_OVERRIDE directory cannot be resolved: missing-state" \ - "spawn did not name the unresolvable FM_STATE_OVERRIDE" - - out=$( - cd "$CASE_DIR" || exit 1 - FM_ROOT_OVERRIDE='' FM_HOME=home \ - FM_STATE_OVERRIDE=home/state FM_DATA_OVERRIDE=missing-data \ - "$SPAWN" "$id" "$PROJ_DIR" 2>&1 - ) - status=$? - expect_code 1 "$status" "spawn with an unresolvable relative data override should fail" - assert_contains "$out" "FM_DATA_OVERRIDE directory cannot be resolved: missing-data" \ - "spawn did not name the unresolvable FM_DATA_OVERRIDE" - pass "unresolvable relative spawn overrides fail with named diagnostics" -} - test_active_dispatch_profile_requires_explicit_harness_for_ship() { local rec id out status id=profile-required-ship-z11 @@ -494,7 +342,7 @@ test_pi_threads_model_and_max_effort() { expect_code 0 "$status" "pi spawn with max effort should succeed" assert_meta_profile "$HOME_DIR/state/$id.meta" pi openai-codex/gpt-5.6-sol max launch=$(cat "$LAUNCH_LOG") - assert_contains "$launch" "FM_PI_HARNESS=pi pi --model 'openai-codex/gpt-5.6-sol' --thinking 'max' -e" \ + assert_contains "$launch" "pi --model 'openai-codex/gpt-5.6-sol' --thinking 'max' -e" \ "pi launch did not thread the requested model and max thinking level" assert_not_contains "$launch" "FM_FIRSTMATE_PI_LAUNCH_BRIEF=" \ "pi launch still exports the removed Calm input-reroute binding" @@ -503,72 +351,6 @@ test_pi_threads_model_and_max_effort() { pass "pi receives --model and --thinking max profile flags" } -test_pi_signed_threads_shared_pi_profile_and_preserves_identity() { - local rec id out status launch - id=profile-pi-signed-z8b - rec=$(make_spawn_case profile-pi-signed pi-signed "$id") - read_case_record "$rec" - - out=$(run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR" \ - --model openai-codex/gpt-5.6-sol --effort max) - status=$? - expect_code 0 "$status" "pi-signed spawn with max effort should succeed" - assert_contains "$out" "spawned $id harness=pi-signed" "pi-signed spawn did not preserve its visible identity" - assert_meta_profile "$HOME_DIR/state/$id.meta" pi-signed openai-codex/gpt-5.6-sol max - launch=$(cat "$LAUNCH_LOG") - assert_contains "$launch" "FM_PI_HARNESS=pi-signed pi-signed --model 'openai-codex/gpt-5.6-sol' --thinking 'max' -e" \ - "pi-signed launch did not share Pi's model, thinking, and extension semantics" - assert_contains "$launch" "fm-operational-input.sh' encode launch-brief" \ - "pi-signed launch lost the canonical typed launch-brief envelope" - assert_present "$HOME_DIR/state/$id.pi-ext.ts" "pi-signed launch did not install Pi's turn-end extension" - pass "pi-signed shares Pi launch semantics while preserving its configured and recorded identity" -} - -test_pi_signed_missing_binary_refuses_before_endpoint_or_metadata() { - local rec id out status - id=profile-pi-signed-missing-z8c - rec=$(make_spawn_case profile-pi-signed-missing pi-signed "$id") - read_case_record "$rec" - rm -f "$FAKEBIN_DIR/pi-signed" - : > "$LAUNCH_LOG" - - out=$(FM_ROOT_OVERRIDE='' FM_HOME="$HOME_DIR" \ - FM_STATE_OVERRIDE="$HOME_DIR/state" FM_DATA_OVERRIDE="$HOME_DIR/data" \ - FM_PROJECTS_OVERRIDE="$HOME_DIR/projects" FM_CONFIG_OVERRIDE="$HOME_DIR/config" \ - FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$WT_DIR" TMUX="fake,1,0" \ - FM_FAKE_LAUNCH_LOG="$LAUNCH_LOG" PATH="$FAKEBIN_DIR:/usr/bin:/bin:/usr/sbin:/sbin" \ - "$SPAWN" "$id" "$PROJ_DIR" 2>&1) - status=$? - expect_code 1 "$status" "a missing pi-signed executable should refuse the spawn" - assert_contains "$out" "pi-signed executable not found on PATH" \ - "missing pi-signed refusal did not name the actionable requirement" - assert_absent "$HOME_DIR/state/$id.meta" "missing pi-signed refusal wrote task metadata" - [ ! -s "$LAUNCH_LOG" ] || fail "missing pi-signed refusal typed a launch command" - pass "pi-signed refuses safely and actionably when the selected executable is unavailable" -} - -test_pi_signed_persistent_secondmate_uses_pi_extensions_and_identity() { - local rec id sm out status launch - id=profile-pi-signed-secondmate-z8d - rec=$(make_spawn_case profile-pi-signed-secondmate codex "$id") - read_case_record "$rec" - printf '%s\n' pi-signed > "$HOME_DIR/config/secondmate-harness" - sm="$CASE_DIR/secondmate-home" - make_seeded_secondmate_home "$sm" "$id" - sm=$(cd "$sm" && pwd -P) - - out=$(run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$sm" --secondmate) - status=$? - expect_code 0 "$status" "pi-signed persistent secondmate spawn should succeed" - assert_contains "$out" "spawned $id harness=pi-signed kind=secondmate" \ - "pi-signed secondmate spawn did not preserve its runtime identity" - assert_meta_profile "$HOME_DIR/state/$id.meta" pi-signed default default - launch=$(cat "$LAUNCH_LOG") - assert_contains "$launch" "FM_PI_HARNESS=pi-signed pi-signed -e '$sm/.pi/extensions/fm-primary-turnend-guard.ts' -e '$sm/.pi/extensions/fm-primary-pi-watch.ts'" \ - "pi-signed secondmate did not share Pi's primary extension launch shape" - pass "pi-signed is a distinct persistent secondmate runtime with shared Pi supervision semantics" -} - test_batch_forwards_shared_profile_flags() { local rec id1 id2 out status id1=profile-batch-a-z9 @@ -588,55 +370,6 @@ test_batch_forwards_shared_profile_flags() { pass "batch dispatch forwards shared --harness, --model, and --effort to every pair" } -test_claude_forwards_firstmate_config_dir_when_set() { - local rec id out status launch - id=profile-claude-cfgdir-z17 - rec=$(make_spawn_case profile-claude-cfgdir claude "$id") - read_case_record "$rec" - - out=$(FM_TEST_CLAUDE_CONFIG_DIR="/opt/test/claude-work" \ - run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR") - status=$? - expect_code 0 "$status" "claude spawn with CLAUDE_CONFIG_DIR set should succeed" - launch=$(cat "$LAUNCH_LOG") - assert_contains "$launch" "CLAUDE_CONFIG_DIR='/opt/test/claude-work' CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false claude" \ - "claude launch did not forward firstmate's CLAUDE_CONFIG_DIR to the crewmate pane" - pass "claude forwards firstmate's CLAUDE_CONFIG_DIR so the crewmate uses the same credential store" -} - -test_claude_omits_config_dir_prefix_when_unset() { - local rec id out status launch - id=profile-claude-nocfgdir-z18 - rec=$(make_spawn_case profile-claude-nocfgdir claude "$id") - read_case_record "$rec" - - # run_spawn pins CLAUDE_CONFIG_DIR empty by default, exercising the single-store - # default path where fm-spawn adds no prefix. - out=$(run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR") - status=$? - expect_code 0 "$status" "claude spawn without CLAUDE_CONFIG_DIR should succeed" - launch=$(cat "$LAUNCH_LOG") - assert_not_contains "$launch" "CLAUDE_CONFIG_DIR=" \ - "claude launch must not add a config-dir prefix when firstmate has no CLAUDE_CONFIG_DIR set" - pass "claude omits the config-dir prefix when firstmate runs with the single-store default" -} - -test_non_claude_harness_ignores_config_dir() { - local rec id out status launch - id=profile-codex-nocfgdir-z19 - rec=$(make_spawn_case profile-codex-nocfgdir codex "$id") - read_case_record "$rec" - - out=$(FM_TEST_CLAUDE_CONFIG_DIR="/opt/test/claude-work" \ - run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR") - status=$? - expect_code 0 "$status" "codex spawn with CLAUDE_CONFIG_DIR set should succeed" - launch=$(cat "$LAUNCH_LOG") - assert_not_contains "$launch" "CLAUDE_CONFIG_DIR=" \ - "non-claude harness launch must not receive the claude-specific config-dir prefix" - pass "non-claude harnesses do not receive the claude CLAUDE_CONFIG_DIR prefix" -} - test_active_dispatch_profile_does_not_block_secondmate_launch() { local rec id sm out status id=profile-secondmate-z16 @@ -656,10 +389,6 @@ test_active_dispatch_profile_does_not_block_secondmate_launch() { } test_no_profile_keeps_claude_profile_defaults -test_relative_home_overrides_launch_with_absolute_cross_process_paths -test_home_defaults_preserve_absolute_or_resolve_relative_paths -test_absolute_override_spelling_is_preserved_in_launch_paths -test_unresolvable_relative_overrides_fail_loudly test_active_dispatch_profile_requires_explicit_harness_for_ship test_active_dispatch_profile_requires_explicit_harness_for_scout test_active_dispatch_profile_allows_explicit_harness @@ -673,13 +402,7 @@ test_grok_omits_invalid_max_reasoning_effort test_grok_omits_invalid_xhigh_reasoning_effort test_opencode_threads_model_and_ignores_effort_axis test_pi_threads_model_and_max_effort -test_pi_signed_threads_shared_pi_profile_and_preserves_identity -test_pi_signed_missing_binary_refuses_before_endpoint_or_metadata -test_pi_signed_persistent_secondmate_uses_pi_extensions_and_identity test_batch_forwards_shared_profile_flags -test_claude_forwards_firstmate_config_dir_when_set -test_claude_omits_config_dir_prefix_when_unset -test_non_claude_harness_ignores_config_dir test_active_dispatch_profile_does_not_block_secondmate_launch echo "# all fm-spawn-dispatch-profile tests passed" diff --git a/tests/fm-test-isolation-proof.test.sh b/tests/fm-test-isolation-proof.test.sh index e34e97e8515..6a11def0eaa 100755 --- a/tests/fm-test-isolation-proof.test.sh +++ b/tests/fm-test-isolation-proof.test.sh @@ -218,6 +218,48 @@ test_docs_record_proof_owner() { pass "docs archive the isolation-proof owner and posture" } +test_docs_match_archived_proof() { + python3 - "$PROOF_DOC" "$PROOF_JSON" <<'PY' \ + || fail "proof Markdown must match the archived proof JSON" +import json +import re +import sys + +markdown = open(sys.argv[1], encoding="utf-8").read() +with open(sys.argv[2], encoding="utf-8") as stream: + proof = json.load(stream) + +summary = proof["summary"] +posture = [ + f'| `run_id` | `{proof["run_id"]}` |', + f'| `started_at` | `{proof["started_at"]}` |', + f'| `finished_at` | `{proof["finished_at"]}` |', + f'| concurrency | **{proof["concurrency"]}** |', + f'| candidates | **{summary["total"]}** |', + f'| failed | **{summary["failed"]}** |', + f'| wall duration_ms | **{summary["duration_ms"]}** (~{summary["duration_ms"] / 1000:.1f}s) |', + f'| `production_sharding_enabled` | `{str(proof["production_sharding_enabled"]).capitalize()}` |', + f'| `fm_test_run_jobs_enabled` | `{str(proof["fm_test_run_jobs_enabled"]).capitalize()}` |', + f'| host proof date | {proof["finished_at"][:10]} (UTC day of archive write) |', +] +assert all(line in markdown for line in posture) +section = markdown.split("## Per-candidate durations (concurrent run)", 1)[1] +section = section.split("## Audit notes (why this set)", 1)[0] +actual = [ + (int(duration), int(exit_code), int(worker), path) + for duration, exit_code, worker, path in re.findall( + r"^\| (\d+) \| (\d+) \| (\d+) \| `([^`]+)` \|$", section, re.MULTILINE + ) +] +expected = [ + (row["duration_ms"], row["exit"], row["worker"], row["path"]) + for row in sorted(proof["scripts"], key=lambda row: row["duration_ms"], reverse=True) +] +assert actual == expected +PY + pass "proof Markdown matches archived JSON posture and durations" +} + test_list_candidates_nonempty_and_stable test_candidates_exclude_serial_classes test_candidates_match_archived_proof @@ -227,3 +269,4 @@ test_family_map_labels_this_contract test_aggregate_failure_under_concurrency test_phase4_consumes_proven_set_only test_docs_record_proof_owner +test_docs_match_archived_proof diff --git a/tests/fm-test-run.test.sh b/tests/fm-test-run.test.sh index 7df9459d25b..7c7dbc5d1b3 100755 --- a/tests/fm-test-run.test.sh +++ b/tests/fm-test-run.test.sh @@ -13,6 +13,7 @@ set -u RUNNER="$ROOT/bin/fm-test-run.sh" CI="$ROOT/.github/workflows/ci.yml" CONTRIB="$ROOT/CONTRIBUTING.md" +SHARD_DOC="$ROOT/docs/fm-test-portable-shards.md" assert_present "$RUNNER" "bin/fm-test-run.sh is missing" [ -x "$RUNNER" ] || fail "bin/fm-test-run.sh must be executable" @@ -466,6 +467,35 @@ test_portable_shard_union_and_coverage_guard() { pass "portable shard union, disjointness, and coverage guard hold" } +test_portable_shard_docs_match_lanes() { + python3 - "$RUNNER" "$SHARD_DOC" <<'PY' \ + || fail "portable shard documentation must match lane counts and timing sums" +import re +import subprocess +import sys + +runner, doc_path = sys.argv[1:3] +markdown = open(doc_path, encoding="utf-8").read() +averages = { + path: int(duration) + for duration, path in re.findall(r"^\| (\d+) \| `([^`]+)` \|$", markdown, re.MULTILINE) +} +totals = {} +for lane in ("portable-parallel-1", "portable-parallel-2"): + scripts = subprocess.check_output( + [runner, "--list", "--lane", lane], text=True + ).splitlines() + totals[lane] = (len(scripts), sum(averages[path] for path in scripts)) + +for lane, (count, duration) in totals.items(): + expected = f"| `{lane}` | {count} | {duration} ms (~{duration / 1000:.1f} s) |" + assert expected in markdown +imbalance = abs(totals["portable-parallel-1"][1] - totals["portable-parallel-2"][1]) +assert f"| imbalance | | {imbalance} ms |" in markdown +PY + pass "portable shard documentation matches lane counts and timing sums" +} + test_jobs_requires_proven_isolated() { local tmp rc tmp=$(mktemp -d "${TMPDIR:-/tmp}/fm-test-run-jobs.XXXXXX") @@ -660,6 +690,7 @@ test_fail_on_gate_skip_token test_exclude_family test_ci_and_docs_call_the_owner test_portable_shard_union_and_coverage_guard +test_portable_shard_docs_match_lanes test_jobs_requires_proven_isolated test_jobs_parallel_scheduler_and_failure_propagation test_aggregate_json From c7518d9eaeff5d836f011cb089dc08bda022baee Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Sat, 25 Jul 2026 11:44:25 -0700 Subject: [PATCH 12/52] docs(agents): drop superseded interim quota-window rule (#1039) quota-axi 0.1.13 emits schemaVersion 2 with a quotaSemantics object per provider, so the successor named in the interim rule has landed and the rule's own removal condition is satisfied. Keep the ownership clause so quota-axi remains the single owner of how model or product windows relate to bounding account windows, and drop the interim weakest-headroom instruction. The unknown-semantics case is already covered by the existing requirement to stop and report a candidate whose applicable quota data or interpretation cannot be established. Drop the matching assertion phrase from tests/fm-instruction-owners.test.sh; the retained ownership phrase still asserts. --- AGENTS.md | 2 +- tests/fm-instruction-owners.test.sh | 3 +-- 2 files changed, 2 insertions(+), 3 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 87dfe18f00d..583bda452c3 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -168,7 +168,7 @@ Account for every candidate; if any harness/model/provider relationship, applica Preserve malformed profile configuration as an actionable error rather than selecting around it. When every candidate is tight, preserve the captain's strongest-reasoning class rather than silently downgrading it solely to conserve quota; stop and report the tight choice if that class cannot proceed. Break genuine headroom ties without array-order or harness bias. -`quota-axi` owns how model or product windows relate to bounding account windows; as an explicitly interim rule until successor `quota-axi-interpretation-hints-h3` lands, use the weakest applicable remaining headroom, then remove this interim rule. +`quota-axi` owns how model or product windows relate to bounding account windows. The generic effort fallback and its precedence are owned by `harness-adapters`: explicit captain and standing configured effort win; otherwise use low for well-understood explicit work, xhigh for ambiguous investigation or design, intermediate levels proportionally, and never max without explicit captain preference. Do not add model-specific versions of that policy. diff --git a/tests/fm-instruction-owners.test.sh b/tests/fm-instruction-owners.test.sh index f4fffb8cbe4..5cb26268f7c 100755 --- a/tests/fm-instruction-owners.test.sh +++ b/tests/fm-instruction-owners.test.sh @@ -115,8 +115,7 @@ test_agent_owned_quota_array_dispatch_contract() { 'Preserve malformed profile configuration as an actionable error' \ "preserve the captain's strongest-reasoning class rather than silently downgrading it" \ 'Break genuine headroom ties without array-order or harness bias' \ - '`quota-axi` owns how model or product windows relate to bounding account windows' \ - 'explicitly interim rule until successor `quota-axi-interpretation-hints-h3` lands'; do + '`quota-axi` owns how model or product windows relate to bounding account windows'; do assert_grep "$phrase" "$AGENTS" "array-dispatch contract lost '$phrase'" done From 86e6dd1bb81776a476a1446f38dd99cb1ce16aa9 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Sat, 25 Jul 2026 18:12:52 -0700 Subject: [PATCH 13/52] fix(tmux): scope busy detection and recognize current Claude turns (#1049) * fix(tmux): scope Claude busy detection by harness * no-mistakes(review): Separate verified and fallback busy signatures * no-mistakes(test): Scope busy signatures to supplied harnesses * no-mistakes(document): Document harness-scoped busy detection --- .agents/skills/harness-adapters/SKILL.md | 4 +- bin/fm-tmux-lib.sh | 331 +++++------------------ bin/fm-watch.sh | 71 +---- docs/architecture.md | 6 +- docs/configuration.md | 2 +- docs/herdr-backend.md | 2 +- docs/tmux-backend.md | 4 + tests/fm-tmux-submit-busy.test.sh | 83 +----- 8 files changed, 98 insertions(+), 405 deletions(-) diff --git a/.agents/skills/harness-adapters/SKILL.md b/.agents/skills/harness-adapters/SKILL.md index 036eacdc985..9251130f4c7 100644 --- a/.agents/skills/harness-adapters/SKILL.md +++ b/.agents/skills/harness-adapters/SKILL.md @@ -152,11 +152,11 @@ Natural language is acceptable if uncertain. - pi: no separate verified skill invocation beyond normal command behavior; use natural language if the exact skill command is uncertain. - grok: `/<skill>`, for example `/no-mistakes` (same form as claude). Verified end to end: grok discovers the user-level `no-mistakes` skill, `/no-mistakes` invokes it, and grok drives a real `no-mistakes axi run`. Like codex's `$`/`/` popups, typing `/<skill>` opens grok's slash-autocomplete, so a too-fast Enter selects the popup entry instead of sending, and for an argument-taking command (like `/no-mistakes`'s optional task-first argument) that first Enter only expands the popup selection into an argument-hint placeholder rather than submitting - a genuine second Enter is required (see the grok section below for the 2026-07-03 incident and fix). `fm_tmux_submit_core`'s retried Enter (used by `fm-send` on the tmux backend) already handles this correctly by reading the cursor row; the herdr backend needed a dedicated fix (`fm_backend_herdr_composer_state`, docs/herdr-backend.md) because its prior delta-based verification false-positived on that same popup-close content change. -## claude (VERIFIED) +## claude (VERIFIED; busy signature re-verified 2026-07-25 on Claude Code 2.1.220) | Fact | Value | |---|---| -| Busy-pane signature | `esc to interrupt` | +| Busy-pane signature | Current turns match the harness-scoped `…[[:space:]]+\([0-9]+[smh]` shape after a rotating glyph and word, for example `✢ Pollinating… (16s · ...)`; legacy `esc to interrupt` remains accepted, while `Worked for 31s` is idle. | | Exit command | `/exit` | | Interrupt | single Escape | | Skill invocation | `/<skill>` (e.g. `/no-mistakes`) | diff --git a/bin/fm-tmux-lib.sh b/bin/fm-tmux-lib.sh index cf8c3f7fa5e..84ed3f67f21 100755 --- a/bin/fm-tmux-lib.sh +++ b/bin/fm-tmux-lib.sh @@ -19,15 +19,15 @@ # "suggestion" as dim/faint text inside an otherwise-empty composer. A plain # capture cannot tell it apart from text a human typed, so the old reader saw an # idle pane as holding pending input and the daemon deferred injection / firstmate -# misjudged the pane. The composer reader now captures the visible pane WITH ANSI -# styling (tmux capture-pane -e), locates a bordered composer structurally, and -# extracts the real typed content from every row with the shared, fleet-wide -# fm_composer_strip_ghost (bin/fm-composer-lib.sh), which drops every -# de-emphasised run - dim/faint (SGR 2) AND a dark/muted truecolor foreground - -# so ghost/placeholder text never counts as real input. The styled capture is -# consumed internally and parsed into a boolean here; it is NEVER surfaced -# (fm-peek and every human/LLM-facing path stay plain). This is harness-generic: -# any harness that de-emphasises placeholder/ghost text +# misjudged the pane. The composer reader now captures just the cursor line WITH +# ANSI styling (tmux capture-pane -e) and extracts the real typed content with the +# shared, fleet-wide fm_composer_strip_ghost (bin/fm-composer-lib.sh), which drops +# every de-emphasised run - dim/faint (SGR 2) AND a dark/muted truecolor +# foreground - so ghost/placeholder text never counts as real input. The styled +# capture is consumed internally and parsed into a boolean here; it is NEVER +# surfaced (fm-peek and every human/LLM-facing path stay plain), and only the +# single composer row is captured, so no escape-laden pane bulk is produced. This +# is harness-generic: any harness that de-emphasises placeholder/ghost text # benefits, and the herdr adapter routes through the same owner (task # afk-herdr-false-pending), so the two backends cannot drift. # @@ -66,23 +66,12 @@ # line has an ellipsis followed by a parenthesized elapsed duration. Keep this # signature separate from the shared default because that shape is not generic # enough to classify arbitrary harness output safely. -# Kimi's anchored moon-phase spinner is separate because bare moon glyphs in -# ordinary output must not classify another harness as busy. Leading whitespace is -# OPTIONAL; whitespace on both sides of the separator is REQUIRED because every -# captured spinner row had it. A zero-whitespace form has NEVER been observed and -# is deliberately not matched. The line end is intentionally unanchored because -# rotating tip text follows and is not required to be present. The idle status -# bar's lowercase `thinking` label and independently rotating tip text are not -# busy signals on their own. -# The full moon-phase set remains locale- and emoji-font-sensitive because Kimi -# exposes no stable ASCII busy token. FM_TMUX_BUSY_REGEX_DEFAULT='esc (to )?interrupt|Working\.\.\.|Ctrl\+c:cancel' FM_TMUX_CLAUDE_BUSY_REGEX_DEFAULT='esc to interrupt|…[[:space:]]+\([0-9]+[smh]' FM_TMUX_CODEX_BUSY_REGEX_DEFAULT='esc to interrupt' FM_TMUX_OPENCODE_BUSY_REGEX_DEFAULT='esc interrupt' FM_TMUX_PI_BUSY_REGEX_DEFAULT='Working\.\.\.' FM_TMUX_GROK_BUSY_REGEX_DEFAULT='Ctrl\+c:cancel' -FM_TMUX_KIMI_BUSY_REGEX_DEFAULT='^[[:space:]]*(🌑|🌒|🌓|🌔|🌕|🌖|🌗|🌘)[[:space:]]+·[[:space:]]+' fm_busy_lines_match() { # [harness] local harness=${1:-} lines regex @@ -94,9 +83,8 @@ fm_busy_lines_match() { # [harness] claude) regex=$FM_TMUX_CLAUDE_BUSY_REGEX_DEFAULT ;; codex) regex=$FM_TMUX_CODEX_BUSY_REGEX_DEFAULT ;; opencode) regex=$FM_TMUX_OPENCODE_BUSY_REGEX_DEFAULT ;; - pi|pi-signed) regex=$FM_TMUX_PI_BUSY_REGEX_DEFAULT ;; + pi) regex=$FM_TMUX_PI_BUSY_REGEX_DEFAULT ;; grok) regex=$FM_TMUX_GROK_BUSY_REGEX_DEFAULT ;; - kimi) regex=$FM_TMUX_KIMI_BUSY_REGEX_DEFAULT ;; '') regex=$FM_TMUX_BUSY_REGEX_DEFAULT ;; *) # A supplied harness must never borrow another harness's signature. @@ -117,251 +105,68 @@ fm_busy_lines_match() { # [harness] # so the tmux and herdr adapters cannot drift apart on what counts as ghost text. fm_tmux_strip_ghost() { fm_composer_strip_ghost; } -# fm_tmux_composer_row_state: classify one raw styled candidate row. -# A structural caller forces bordered=1; the compatibility fallback passes 0 -# and may recognize a busy footer. -fm_tmux_composer_row_state() { # <raw-row> [bordered] [allow-busy] -> empty|pending|unknown - local raw=$1 bordered=${2:-0} allow_busy=${3:-1} plain stripped +# fm_tmux_composer_state: classify the cursor/composer line of <target> as +# empty - no pending input (blank, a busy footer, an empty agent composer, or +# only de-emphasised ghost/placeholder text). Safe to inject; also the positive +# acknowledgement that a submit landed. +# pending - real, unsubmitted text on the cursor line (a human mid-typing, or a +# previous injection whose Enter was swallowed). Defer / retry. +# unknown - the pane could not be read (tmux error), OR the cursor line is a +# bare shell prompt (`$`/`%`/`#`/`>`) - a dead shell, not an agent +# composer, so NOT a safe injection target. The caller decides. +# +# The cursor line is captured WITH ANSI styling (capture-pane -e) and bounded to +# the single composer row (-S/-E). The bordered flag (a genuine composer box) is +# read from the PLAIN row (fm_composer_strip_ansi keeps ghost text so the box +# border is still visible), while the real-typed CONTENT is extracted with the +# shared fm_composer_strip_ghost so dim/faint AND dark-truecolor ghost text drops +# out before classification (grok's dark box border drops with the ghost, which +# is why the bordered flag is read from the plain row, not the ghost-stripped +# one). Both are internal only, never surfaced. The detector strips the harness's +# box-drawing composer borders ("│ … │", heavy "┃", or a plain ASCII "|") using +# literal-string substitution (bash 3.2 safe, locale-independent - no \u escapes, +# no multibyte character classes), and delegates the empty/pending/unknown +# decision to the shared owner fm_composer_classify_content +# (bin/fm-composer-lib.sh). The bordered flag is what lets a bordered `│ > │` +# (claude's own idle composer) read empty while a bare, unbordered `$ ` dead-shell +# prompt reads unknown. +fm_tmux_composer_state() { # <target> -> empty|pending|unknown + local target=$1 cy raw plain stripped bordered=0 + cy=$(tmux display-message -p -t "$target" '#{cursor_y}' 2>/dev/null) || { printf 'unknown'; return 0; } + case "$cy" in ''|*[!0-9]*) printf 'unknown'; return 0 ;; esac + raw=$(tmux capture-pane -e -p -t "$target" -S "$cy" -E "$cy" 2>/dev/null) || { printf 'unknown'; return 0; } + # bordered: from the plain row (borders survive an all-ANSI strip). plain=$(printf '%s\n' "$raw" | fm_composer_strip_ansi) plain="${plain#"${plain%%[![:space:]]*}"}" plain="${plain%"${plain##*[![:space:]]}"}" + case "$plain" in + '│'*'│'|'┃'*'┃'|'|'*'|') bordered=1 ;; + esac + # content: from the ghost-stripped row (real typed text only). stripped=$(printf '%s\n' "$raw" | fm_composer_strip_ghost) stripped="${stripped#"${stripped%%[![:space:]]*}"}" stripped="${stripped%"${stripped##*[![:space:]]}"}" case "$stripped" in '│'*'│') stripped=${stripped#│}; stripped=${stripped%│} ;; '┃'*'┃') stripped=${stripped#┃}; stripped=${stripped%┃} ;; - '║'*'║') stripped=${stripped#║}; stripped=${stripped%║} ;; '|'*'|') stripped=${stripped#|}; stripped=${stripped%|} ;; esac stripped="${stripped#"${stripped%%[![:space:]]*}"}" stripped="${stripped%"${stripped##*[![:space:]]}"}" - if [ "$allow_busy" = 1 ] && [ -n "$stripped" ] \ + # A busy footer landing on the cursor line is not pending input (tmux-specific: + # only tmux captures the raw cursor row, which may BE the footer). + if [ -n "$stripped" ] \ && printf '%s' "$stripped" | grep -qiE "${FM_BUSY_REGEX:-$FM_TMUX_BUSY_REGEX_DEFAULT}"; then printf 'empty'; return 0 fi fm_composer_classify_content "$bordered" "$stripped" "${FM_COMPOSER_IDLE_RE:-}" insensitive "$plain" } -fm_tmux_row_has_composer_edge() { # <plain-row> - local row=$1 - row="${row#"${row%%[![:space:]]*}"}" - row="${row%"${row##*[![:space:]]}"}" - case "$row" in - '│'*|*'│'|'┃'*|*'┃'|'║'*|*'║'|'╭'*|*'╭'|'╮'*|*'╮'|\ - '┌'*|*'┌'|'┐'*|*'┐'|'╔'*|*'╔'|'╗'*|*'╗'|'┏'*|*'┏'|'┓'*|*'┓'|\ - '╰'*|*'╰'|'╯'*|*'╯'|'└'*|*'└'|'┘'*|*'┘'|'╚'*|*'╚'|'╝'*|*'╝'|\ - '┗'*|*'┗'|'┛'*|*'┛'|'─'*|*'─'|'━'*|*'━'|'═'*|*'═'|'|'*|*'|'|'+'*|*'+') - return 0 - ;; - esac - return 1 -} - -fm_tmux_composer_geometry_spaces() { # <content-inner> -> spaces - local content=$1 probe - probe="${content#"${content%%[![:space:]]*}"}" - case "$probe" in - '>'*) content=${content/>/ } ;; - '❯'*) content=${content/❯/ } ;; - '›'*) content=${content/›/ } ;; - esac - content=$(printf '%s' "$content" | LC_ALL=C sed 's/[!-~]/ /g') - case "$content" in - *[![:space:]]*) return 1 ;; - esac - printf '%s' "$content" -} - -# fm_tmux_find_composer_box: print the zero-based top and bottom rows of the -# complete bordered box that structurally contains the cursor, plus whether its -# geometry is ambiguous. The cursor may be on any content row or on the bottom -# border; no fixed cursor offset is used. -fm_tmux_find_composer_box() { # <cursor-y> <plain-visible-pane> -> "<top> <bottom> <ambiguous>" - local cy=$1 pane=$2 line indent left_stripped trimmed kind family current_family= - local side_family top_inner top_spaces='' geometry_check=0 geometry_ambiguous=0 - local content_inner content_spaces bottom_inner bottom_spaces - local current_indent= - local row=0 top=-1 valid=0 content_rows=0 unsafe=0 cursor_structural=0 - while IFS= read -r line; do - indent=${line%%[![:space:]]*} - left_stripped="${line#"${line%%[![:space:]]*}"}" - trimmed="${left_stripped%"${left_stripped##*[![:space:]]}"}" - kind= - family= - case "$trimmed" in - '╭'*'╮') kind=top; family=rounded ;; - '┌'*'┐') kind=top; family=light ;; - '╔'*'╗') kind=top; family=double ;; - '┏'*'┓') kind=top; family=heavy ;; - '╰'*'╯') kind=bottom; family=rounded ;; - '└'*'┘') kind=bottom; family=light ;; - '╚'*'╝') kind=bottom; family=double ;; - '┗'*'┛') kind=bottom; family=heavy ;; - '+'*'+') kind=ascii; family=ascii ;; - esac - if [ "$row" -eq "$cy" ] && fm_tmux_row_has_composer_edge "$trimmed"; then - cursor_structural=1 - fi - if [ "$kind" = top ] || { [ "$kind" = ascii ] && [ "$top" -lt 0 ]; }; then - if [ "$top" -ge 0 ] && [ "$top" -lt "$cy" ] && [ "$cy" -le "$row" ]; then - unsafe=1 - fi - top=$row - current_family=$family - current_indent=$indent - valid=1 - content_rows=0 - geometry_ambiguous=0 - geometry_check=1 - top_inner=$trimmed - case "$family" in - rounded) top_inner=${top_inner#╭}; top_inner=${top_inner%╮}; top_spaces=${top_inner//─/ } ;; - light) top_inner=${top_inner#┌}; top_inner=${top_inner%┐}; top_spaces=${top_inner//─/ } ;; - double) top_inner=${top_inner#╔}; top_inner=${top_inner%╗}; top_spaces=${top_inner//═/ } ;; - heavy) top_inner=${top_inner#┏}; top_inner=${top_inner%┓}; top_spaces=${top_inner//━/ } ;; - ascii) top_inner=${top_inner#+}; top_inner=${top_inner%+}; top_spaces=${top_inner//-/ } ;; - esac - case "$top_spaces" in - *[![:space:]]*) geometry_check=0; geometry_ambiguous=1 ;; - esac - elif [ "$kind" = bottom ] || { [ "$kind" = ascii ] && [ "$top" -ge 0 ]; }; then - if [ "$top" -ge 0 ] && [ "$family" = "$current_family" ] \ - && [ "$valid" = 1 ] && [ "$content_rows" -gt 0 ] \ - && [ "$top" -lt "$cy" ] && [ "$cy" -le "$row" ]; then - [ "$indent" = "$current_indent" ] || geometry_ambiguous=1 - if [ "$geometry_check" = 1 ]; then - bottom_inner=$trimmed - case "$family" in - rounded) bottom_inner=${bottom_inner#╰}; bottom_inner=${bottom_inner%╯}; bottom_spaces=${bottom_inner//─/ } ;; - light) bottom_inner=${bottom_inner#└}; bottom_inner=${bottom_inner%┘}; bottom_spaces=${bottom_inner//─/ } ;; - double) bottom_inner=${bottom_inner#╚}; bottom_inner=${bottom_inner%╝}; bottom_spaces=${bottom_inner//═/ } ;; - heavy) bottom_inner=${bottom_inner#┗}; bottom_inner=${bottom_inner%┛}; bottom_spaces=${bottom_inner//━/ } ;; - ascii) bottom_inner=${bottom_inner#+}; bottom_inner=${bottom_inner%+}; bottom_spaces=${bottom_inner//-/ } ;; - esac - [ "$bottom_spaces" = "$top_spaces" ] || geometry_ambiguous=1 - fi - printf '%s %s %s' "$top" "$row" "$geometry_ambiguous" - return 0 - fi - if { [ "$top" -ge 0 ] && [ "$top" -lt "$cy" ] && [ "$cy" -le "$row" ]; } \ - || [ "$row" -eq "$cy" ]; then - unsafe=1 - fi - top=-1 - current_family= - current_indent= - valid=0 - content_rows=0 - elif [ "$top" -ge 0 ]; then - side_family= - case "$trimmed" in - '│'*'│') side_family=single ;; - '┃'*'┃') side_family=heavy ;; - '║'*'║') side_family=double ;; - '|'*'|') side_family=ascii ;; - esac - case "$current_family:$side_family" in - rounded:single|light:single|heavy:heavy|double:double|ascii:ascii) - content_rows=$((content_rows + 1)) - [ "$indent" = "$current_indent" ] || geometry_ambiguous=1 - if [ "$geometry_check" = 1 ]; then - content_inner=$trimmed - case "$side_family" in - single) content_inner=${content_inner#│}; content_inner=${content_inner%│} ;; - heavy) content_inner=${content_inner#┃}; content_inner=${content_inner%┃} ;; - double) content_inner=${content_inner#║}; content_inner=${content_inner%║} ;; - ascii) content_inner=${content_inner#|}; content_inner=${content_inner%|} ;; - esac - if content_spaces=$(fm_tmux_composer_geometry_spaces "$content_inner"); then - [ "$content_spaces" = "$top_spaces" ] || geometry_ambiguous=1 - else - geometry_ambiguous=1 - fi - fi - ;; - *) valid=0 ;; - esac - fi - row=$((row + 1)) - done <<EOF -$pane -EOF - if [ "$top" -ge 0 ] && [ "$top" -lt "$cy" ]; then - unsafe=1 - fi - if [ "$unsafe" = 1 ] || [ "$cursor_structural" = 1 ]; then - return 2 - fi - return 1 -} - -# fm_tmux_composer_state classification contract: -# A row is structural only when its first or last non-whitespace character is a -# composer edge. A complete box has matching border families and bounded top and -# bottom rows. The proof-carrying verdict is empty for proven emptiness, pending -# for proven text in established structure, pending-unproven for text in -# ambiguous structure, and unknown for unreadable state. Consumers that can -# overwrite input or confirm delivery must accept only the exact positive proof -# they require, so unrecognized future verdicts fail safe by default. Empty -# requires positive proof: a genuinely empty composer, an all-empty unambiguous -# box, an empty non-bordered fallback row, or the submit core's proven -# busy-queued Enter conversion. -fm_tmux_composer_state() { # <target> -> empty|pending|pending-unproven|unknown - local target=$1 cy raw pane plain box box_status top bottom geometry_ambiguous - local row row_raw state unknown_seen=0 - cy=$(tmux display-message -p -t "$target" '#{cursor_y}' 2>/dev/null) || { printf 'unknown'; return 0; } - case "$cy" in ''|*[!0-9]*) printf 'unknown'; return 0 ;; esac - pane=$(tmux capture-pane -e -p -t "$target" -S 0 -E - 2>/dev/null) || { printf 'unknown'; return 0; } - plain=$(printf '%s\n' "$pane" | fm_composer_strip_ansi) - if box=$(fm_tmux_find_composer_box "$cy" "$plain"); then - top=${box%% *} - box=${box#* } - bottom=${box%% *} - geometry_ambiguous=${box#* } - row=$((top + 1)) - while [ "$row" -lt "$bottom" ]; do - row_raw=$(printf '%s\n' "$pane" | sed -n "$((row + 1))p") - state=$(fm_tmux_composer_row_state "$row_raw" 1 0) - case "$state" in - pending) - if [ "$geometry_ambiguous" = 1 ]; then - printf 'pending-unproven' - else - printf 'pending' - fi - return 0 - ;; - unknown) unknown_seen=1 ;; - esac - row=$((row + 1)) - done - if [ "$unknown_seen" = 1 ] || [ "$geometry_ambiguous" = 1 ]; then - printf 'unknown' - else - printf 'empty' - fi - return 0 - else - box_status=$? - if [ "$box_status" -eq 2 ]; then - printf 'unknown' - return 0 - fi - fi - raw=$(tmux capture-pane -e -p -t "$target" -S "$cy" -E "$cy" 2>/dev/null) \ - || { printf 'unknown'; return 0; } - if fm_tmux_row_has_composer_edge "$(printf '%s\n' "$raw" | fm_composer_strip_ansi)"; then - printf 'unknown' - return 0 - fi - fm_tmux_composer_row_state "$raw" 0 -} - -# fm_pane_input_pending: 0 when the composer is not proven empty, so pending -# text, ambiguous structure, unreadable state, and future verdicts all defer. +# fm_pane_input_pending: 0 (pending) if the cursor line holds real unsubmitted +# text, 1 otherwise. An unreadable pane is treated as NOT pending (fail-safe: +# the same bias the old daemon used — an unknown pane defers nothing here). fm_pane_input_pending() { # <target> - [ "$(fm_tmux_composer_state "$1")" != empty ] + [ "$(fm_tmux_composer_state "$1")" = pending ] } # fm_pane_is_busy: 0 if the pane's last few non-blank lines show a busy footer @@ -376,34 +181,32 @@ fm_pane_is_busy() { # <target> [harness] # fm_tmux_submit_core: type <text> into <target> ONCE, then submit with Enter, # verifying the composer cleared. Retries Enter ONLY — never retypes, because a # swallowed Enter leaves our text in the composer and retyping would duplicate -# it. Echoes the final proof-carrying verdict on stdout so callers can require -# exact `empty` before treating submission as confirmed. +# it. Echoes the final verdict on stdout (empty|pending|unknown|send-failed) so callers can +# pick their own success policy: +# - the daemon clears its buffer only on "empty" (strict: an unknown pane must +# not be mistaken for a delivered escalation). +# - fm-send fails only on "pending" (lenient: a positively-confirmed swallow), +# so an unreadable pane never turns a normal steer into a false error. # Busy-queued Enter (opencode 1.18.4): the harness accepts Enter while mid-turn # and queues it for after the current turn, but keeps the typed text visible in -# the composer. Once the Enter-retry budget is spent and a structurally proven -# composer still reads "pending", the submit core falls back to -# `fm_pane_is_busy`: a busy pane means the Enter was accepted and queued (report -# `empty` so the caller does not re-send), while an idle pane keeps `pending` as -# a genuine swallow. Pending-unproven receives the same Enter retry budget but -# never reaches this exception. +# the composer. Once the Enter-retry budget is spent and the composer still +# reads "pending", the submit core falls back to `fm_pane_is_busy`: a busy pane +# means the Enter was accepted and queued (report `empty` so the caller does +# not re-send), while an idle pane keeps `pending` as a genuine swallow. This +# is the only place that exception lives, so the daemon's strict and +# fm-send's lenient success policies both treat a busy-queued Enter as +# delivered. fm_tmux_submit_enter_core() { # <target> <retries> <enter-sleep> local target=$1 retries=$2 sleep_s=$3 i=0 state while :; do tmux send-keys -t "$target" Enter 2>/dev/null || true sleep "$sleep_s" state=$(fm_tmux_composer_state "$target") - case "$state" in - pending|pending-unproven) ;; - *) printf '%s' "$state"; return 0 ;; - esac + [ "$state" = pending ] || { printf '%s' "$state"; return 0; } i=$((i + 1)) [ "$i" -lt "$retries" ] || break done - if [ "$state" != pending ]; then - printf '%s' "$state" - return 0 - fi - # Retries exhausted, composer still shows proven pending. + # Retries exhausted, composer still shows pending. # If the pane is busy (agent mid-turn), the harness accepted the Enter # and queued the message for processing when the current turn ends. # Treat it as submitted so the caller does not re-send. diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index e5501f852b3..d77118c3a2a 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -30,16 +30,7 @@ # also carries a "demand-deep-inspection" marker so the # wake payload itself, not just repetition, forces a # closer look instead of another routine supervision -# resume. Unless afk is active. A genuinely busy pane -# (window_is_busy true) is exempt from the above, but -# only up to BUSY_TURN_MAX_SECS with no completed turn -# (state/<id>.turn-ended, or the spawn record before any -# turn completes); past that bound busy_turn_over_age -# routes it through the same wedge timer, so it surfaces -# with the identical "stale: ..." reason, escalation -# count, and demand-deep-inspection marker, for human -# inspection only - never an automatic interrupt, -# signal, or restart of the worker or its tool process. +# resume. Unless afk is active. # check: <script>: <out> authenticated check output, always actionable # check: rejected unauthenticated state checks: <paths> # unsafe state checks were refused without execution @@ -114,8 +105,7 @@ SIGNAL_GRACE=${FM_SIGNAL_GRACE:-30} # seconds to linger after a signal so trai # claude/codex: "esc to interrupt"; opencode: "esc interrupt"; pi: "Working..."; # grok: "Ctrl+c:cancel". Claude's current spinner signature is matched only for # a recorded Claude task because an ellipsis followed by elapsed time is not a -# safe shared signature for arbitrary harness output. Kimi's moon-plus-middot -# spinner signature is likewise matched only for a recorded Kimi task. +# safe shared signature for arbitrary harness output. BUSY_REGEX=${FM_BUSY_REGEX:-'esc (to )?interrupt|Working\.\.\.|Ctrl\+c:cancel'} # Always-on wake triage: most wakes during a long crew validation are benign (a # working: note or turn-end while a pipeline runs, a no-change heartbeat). Rather @@ -136,19 +126,6 @@ BUSY_REGEX=${FM_BUSY_REGEX:-'esc (to )?interrupt|Working\.\.\.|Ctrl\+c:cancel'} # daemon owns triage, so this watcher reverts to one-shot (enqueue + exit on every # wake) and never double-triages - and never runs the costly provably-working read. STALE_ESCALATE_SECS=${FM_STALE_ESCALATE_SECS:-240} # idle secs before a provably-working stale escalates as a possible wedge -# A busy pane is unconditional proof of liveness with no built-in duration bound, -# so a hung foreground call can remain hidden even while its rendered busy -# footer changes every poll. BUSY_TURN_MAX_SECS bounds how long any busy pane -# may go with no completed turn: once its task's -# state/<id>.turn-ended marker (or, before any turn has completed, the task's -# spawn record) is this old, busy_turn_over_age routes the pane through the -# same STALE_ESCALATE_SECS-paced wedge_timer_check used for a provably-working -# non-busy stale, so it escalates via the existing stale reason, escalation -# counter, and demand-deep-inspection marker for human inspection only - never -# an automatic interrupt, signal, or restart. A completed turn touches -# turn-ended and resets the age. Set generously above any legitimate interval -# between completed turns, including long tool calls, builds, or test runs. -BUSY_TURN_MAX_SECS=${FM_BUSY_TURN_MAX_SECS:-3600} # A crew that declared a pause is idling on a known external wait, so its stale # pane is absorbed rather than wedge-escalated. # A captain-held or paused crew whose agent has confidently exited uses the same @@ -303,20 +280,6 @@ wedge_timer_check() { # <window> <since-file> <triage-label> <escalation-count- esac } -# busy_turn_over_age: 0 iff <task>'s latest completed-turn marker is at least -# BUSY_TURN_MAX_SECS old. Ages the per-task turn-ended marker, the harness-neutral -# signal every verified harness's turn-end hook touches; before any turn has -# completed, ages the task's spawn record instead so a fresh task still gets a -# bound. The caller checks that the pane is busy and routes a crossed bound -# through the existing wedge_timer_check, never anything that touches the -# worker itself. -busy_turn_over_age() { # <task> - local task=$1 f - f="$STATE/$task.turn-ended" - [ -e "$f" ] || f="$STATE/$task.meta" - [ "$(age_of "$f")" -ge "$BUSY_TURN_MAX_SECS" ] -} - # Absorb a stale pane under a declared external-wait pause (paused:) or a # dead-agent captain-held transfer, and re-surface it once every # PAUSE_RESURFACE_SECS for a recheck so it cannot rot invisibly. Called on any @@ -887,16 +850,14 @@ EOF ewf="$STATE/.wedge-escalations-$key" pf="$STATE/.paused-$key" # flag: this key's stale is using the bounded pause cadence prev=$(cat "$hf" 2>/dev/null || true) - # Busy match: a backend's native semantic state when available (herdr), else - # the last 6 non-blank lines only (the TUI footer area, where every verified - # harness renders its busy indicator) so busy-looking strings in displayed - # content cannot suppress stale detection. Read once per window per poll and - # reused below so a busy verdict is consistent within one cycle. - if window_is_busy "$w" "$tail40"; then busy_now=0; else busy_now=1; fi if [ "$h" = "$prev" ]; then n=$(( $(cat "$cf" 2>/dev/null || echo 0) + 1 )) echo "$n" > "$cf" - if [ "$n" -ge 2 ] && [ "$busy_now" -ne 0 ]; then + # Busy match: a backend's native semantic state when available (herdr), + # else the last 6 non-blank lines only (the TUI footer area, where every + # verified harness renders its busy indicator) so busy-looking strings + # in displayed content cannot suppress stale detection. + if [ "$n" -ge 2 ] && ! window_is_busy "$w" "$tail40"; then # The pane is idle/stale at hash $h. Triage decides whether this wakes # firstmate. Detection itself is unchanged from above. if [ "$kind" = secondmate ]; then @@ -996,14 +957,8 @@ EOF fi fi else - # Pane busy or not yet stably stale: reset pending escalation bookkeeping, - # unless a genuinely busy pane has gone too long with no completed turn - - # then route it through the same wedge timer instead of erasing it. - if [ "$busy_now" -eq 0 ] && busy_turn_over_age "$task"; then - wedge_timer_check "$w" "$ssf" "busy (no completed turn)" "$ewf" - else - rm -f "$ssf" "$ewf" - fi + # Pane busy or not yet stably stale: reset pending escalation bookkeeping. + rm -f "$ssf" "$ewf" if [ -e "$pf" ] && { [ "$n" -ge 2 ] || ! status_is_paused_or_captain_held "$(last_status_line "$STATE/$(window_to_task "$w" "$STATE").status")"; }; then clear_pause_tracking "$w" fi @@ -1011,13 +966,9 @@ EOF else printf '%s' "$h" > "$hf" echo 0 > "$cf" - if [ "$busy_now" -eq 0 ] && busy_turn_over_age "$task"; then - wedge_timer_check "$w" "$ssf" "busy (no completed turn)" "$ewf" - else - rm -f "$ssf" "$ewf" - fi + rm -f "$ssf" "$ewf" task=$(window_to_task "$w" "$STATE") - if ! afk_present && status_is_paused_or_captain_held "$(last_status_line "$STATE/$task.status")" && [ "$busy_now" -ne 0 ]; then + if ! afk_present && status_is_paused_or_captain_held "$(last_status_line "$STATE/$task.status")" && ! window_is_busy "$w" "$tail40"; then case "$(pause_state_class "$w" "$task")" in paused) handle_paused_stale "$w" "$task" "$h" ;; *) clear_pause_tracking "$w" ;; diff --git a/docs/architecture.md b/docs/architecture.md index e572757dae1..f183ee91384 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -35,7 +35,7 @@ The most recent recognized ci log marker wins, so checks-green monitoring report Only when no matching run exists does it fall back to the pane busy-signature and then a status-log event whose verb maps to a recognized run-state; a dead pane without a run reports unknown instead of trusting a stale log. Decision-only events such as `resolved` never become current state or leak their prose into the current-state detail. In that status-log fallback, a declared external wait reports the distinct `paused` state with its reason. -For herdr, that pane fallback trusts a native `busy` verdict outright, but corroborates native `idle` or unknown verdicts against the rendered busy signature before deciding the crew is not working. +For herdr, that pane fallback trusts a native `busy` verdict outright, but corroborates native `idle` or unknown verdicts against the recorded harness's rendered busy signature before deciding the crew is not working. For whole-fleet read-only review, `bin/fm-fleet-snapshot.sh --json` emits schema `fm-fleet-snapshot.v1` from the backlog, task metadata, current crew state, endpoint probes, PR/report pointers, scout reports, bounded current summaries from registered secondmate homes, and secondmate return-channel guidance. `bin/fm-fleet-view.sh` renders that snapshot as Markdown for humans, while `bin/fm-bearings-snapshot.sh` provides the bounded bearings projection, so both views consume one structured contract instead of reparsing raw fleet files. The script header owns the exact JSON schema. @@ -94,7 +94,9 @@ New spawns select a backend from `--backend`, then `FM_BACKEND`, then local `con Runtime auto-detection is innermost-first: `$TMUX` wins over `HERDR_ENV=1`, which wins over cmux's primary `CMUX_WORKSPACE_ID` marker and documented fallback signals; auto-detected herdr or cmux prints a one-time opt-out notice, auto-detected tmux stays silent, and zellij and orca are never auto-detected (only explicit selection). Unknown backend names fail loudly. For compatibility, default tmux tasks do not write `backend=tmux`; every reader treats a missing `backend=` field as `tmux`. -`fm-watch.sh` polls each window's backend for a busy state: tmux, zellij, orca, and cmux have no native primitive and always report unknown, preserving the original pane-tail-regex detection unchanged; herdr's `agent.get` semantic state (working/idle/done/blocked) is consulted first for stale detection, with unknown native states falling back to the same regex. +`fm-watch.sh` polls each window's backend for a busy state: tmux, zellij, orca, and cmux have no native primitive and always report unknown, so their pane-tail fallback matches only the recorded harness's verified signature; herdr's `agent.get` semantic state (working/idle/done/blocked) is consulted first for stale detection, with unknown native states using the same harness-scoped fallback. +This scope prevents cross-harness false positives such as Kimi's rotating idle tip `ctrl+c: cancel` borrowing Grok's busy token, and keeps Claude's broader elapsed-spinner shape from matching ordinary output in other panes. +Unknown supplied harnesses match no default signature, while callers that have no harness metadata retain the historical combined-pattern compatibility fallback. That poll loop is the default event source for backends with no native push events, so this stays an extraction of the abstraction rather than a watcher rewrite. For capable Herdr sessions, the same watcher replaces its terminal sleep with a bounded native event wait that immediately surfaces `blocked`; [Push events and polling fallback](herdr-backend.md#push-events-and-polling-fallback) owns the current mechanism and capability gates, while [runtime backend verification](verification/runtime-backends.md#native-blocked-event) owns the active evidence. The deeper session-start agent-process liveness probe is separate from that busy-state poll: tmux and Herdr have verified classifiers for secondmate recovery, Zellij remains unverified, and Orca and cmux do not support secondmate spawns. diff --git a/docs/configuration.md b/docs/configuration.md index fba8a6ff335..823a5e18f2a 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -431,7 +431,7 @@ FM_STALE_WORKTREE_LOCK_RETRY_WAIT_SECS= # legacy alias for FM_TREEHOUSE_RETURN FM_FLEET_SYNC_PACKED_REFS_LOCK_RETRIES=3 # fetch retries after fm-fleet-sync.sh hits the orphaned .git/packed-refs.lock signature FM_FLEET_SYNC_PACKED_REFS_LOCK_RETRY_WAIT_SECS=1 # seconds fm-fleet-sync.sh waits before each of those retries FM_FLEET_SYNC_PACKED_REFS_LOCK_AGE_SECS=30 # min mtime age before fm-fleet-sync.sh treats a leftover packed-refs.lock as provably stale -FM_BUSY_REGEX='esc (to )?interrupt|Working\.\.\.|Ctrl\+c:cancel' # busy-pane signatures, shared by watcher, fm-crew-state pane fallback, and tmux helper +FM_BUSY_REGEX= # optional global override for every harness-scoped busy-pane matcher; unset uses each recorded harness's verified signature FM_COMPOSER_IDLE_RE= # optional empty-composer regex, applied after ghost and border stripping FM_COMPOSER_GHOST_LUMA_MAX=128 # fleet-wide: max perceived luminance (0.299R+0.587G+0.114B, 0-255) for a TRUECOLOR foreground to count as de-emphasised ghost/placeholder text and be stripped; dim/faint (SGR 2) is stripped regardless. Assumes a dark terminal theme (bin/fm-composer-lib.sh's fm_composer_strip_ghost, shared by the tmux and herdr composer readers) GROK_HOME= # optional Grok config home for firstmate's global grok turn-end hook; defaults to ~/.grok diff --git a/docs/herdr-backend.md b/docs/herdr-backend.md index c9e7c59ff0a..91047bcc6f3 100644 --- a/docs/herdr-backend.md +++ b/docs/herdr-backend.md @@ -176,7 +176,7 @@ The capture owner requests at least 200 lines from Herdr and trims locally to th This generous floor is required for small composer and peek reads. Herdr's native agent state can read idle while a harness waits on its own long foreground tool. -The shared crew-state path therefore corroborates every native non-busy or unreadable result with the rendered busy regex before concluding that a pane is not working. +The shared crew-state path therefore corroborates every native non-busy or unreadable result with the recorded harness's rendered busy signature before concluding that a pane is not working. A human-blocked permission dialog has no busy banner and still surfaces. ## Composer and injection safety diff --git a/docs/tmux-backend.md b/docs/tmux-backend.md index 50cc14ce9d2..730fa702937 100644 --- a/docs/tmux-backend.md +++ b/docs/tmux-backend.md @@ -57,6 +57,10 @@ Agent liveness and composer safety are separate checks. The shared classifier in `bin/fm-composer-lib.sh` accepts a shell glyph as an empty agent composer only inside a verified bordered composer. A bare shell prompt is `unknown`, so away-mode escalation is never injected into a dead shell. +Rendered busy detection is also harness-scoped. +Task metadata selects only that harness's verified signature, so output from one harness cannot make another harness appear busy. +The exact selection contract and safety rationale live in [architecture](architecture.md#runtime-session-backends), while the signatures live in [the harness-adapters skill](../.agents/skills/harness-adapters/SKILL.md). + `bin/fm-tmux-lib.sh` owns exact type-and-submit mechanics. It types a message once and retries Enter only until the composer clears. A cleared composer is the positive delivery acknowledgement; text left in the composer remains `pending`, and `fm-send.sh` reports the failure instead of retyping. diff --git a/tests/fm-tmux-submit-busy.test.sh b/tests/fm-tmux-submit-busy.test.sh index f3eb49a7eb7..aafdbbae6ca 100755 --- a/tests/fm-tmux-submit-busy.test.sh +++ b/tests/fm-tmux-submit-busy.test.sh @@ -27,7 +27,7 @@ COMPOSER="${FM_FAKE_COMPOSER:?}" case "${1:-}" in display-message) for a in "$@"; do - case "$a" in *cursor_y*) printf '1\n'; exit 0 ;; esac + case "$a" in *cursor_y*) printf '0\n'; exit 0 ;; esac done exit 0 ;; capture-pane) cat "$COMPOSER" 2>/dev/null; exit 0 ;; @@ -37,11 +37,10 @@ case "${1:-}" in case "$1" in -t) shift ;; -l) ;; Enter) is_enter=1 ;; esac; shift done if [ "$is_enter" = 1 ]; then - [ -z "${FM_FAKE_SENT:-}" ] || printf 'Enter\n' >> "$FM_FAKE_SENT" if [ -n "${FM_FAKE_SWALLOW:-}" ] && [ -f "$FM_FAKE_SWALLOW" ]; then [ "${FM_FAKE_PERSIST_SWALLOW:-0}" = 1 ] || rm -f "$FM_FAKE_SWALLOW" else - printf '╭─────╮\n│ > │\n╰─────╯\n' > "$COMPOSER" + printf '│ > │\n' > "$COMPOSER" fi fi exit 0 ;; @@ -60,7 +59,7 @@ test_busy_pane_pending_returns_empty() { composer="$dir/composer" sent="$dir/sent.log" vfile="$dir/verdict" - printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$composer" + printf '│ > fix findings 1 and 3 │\n' > "$composer" : > "$sent" touch "$dir/.swallow" # Pre-check: composer state should be pending (via function, not $()). @@ -71,8 +70,8 @@ test_busy_pane_pending_returns_empty() { FM_FAKE_SWALLOW="$dir/.swallow" FM_FAKE_PERSIST_SWALLOW=1 FM_FAKE_PANE_BUSY=1 \ fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null [ "$(cat "$vfile")" = empty ] || fail "busy-pane pending should return empty, got '$(cat "$vfile")'" - [ "$(grep -c '^Enter$' "$sent" 2>/dev/null || true)" -eq 3 ] \ - || fail "proven pending should consume the configured Enter retry budget" + [ "$(grep -c 'fix findings' "$sent" 2>/dev/null || true)" -eq 0 ] \ + || fail "busy-pane should not retype text" pass "fm_tmux_submit_enter_core: busy pane + pending composer returns empty (message queued)" } @@ -83,7 +82,7 @@ test_idle_pane_pending_returns_pending() { composer="$dir/composer" sent="$dir/sent.log" vfile="$dir/verdict" - printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$composer" + printf '│ > fix findings 1 and 3 │\n' > "$composer" : > "$sent" touch "$dir/.swallow" PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" FM_FAKE_SENT="$sent" \ @@ -100,7 +99,7 @@ test_busy_pane_composer_clears_first_try() { composer="$dir/composer" sent="$dir/sent.log" vfile="$dir/verdict" - printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$composer" + printf '│ > fix findings 1 and 3 │\n' > "$composer" : > "$sent" PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" FM_FAKE_SENT="$sent" FM_FAKE_PANE_BUSY=1 \ fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null @@ -115,7 +114,7 @@ test_idle_pane_composer_clears_first_try() { composer="$dir/composer" sent="$dir/sent.log" vfile="$dir/verdict" - printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$composer" + printf '│ > fix findings 1 and 3 │\n' > "$composer" : > "$sent" PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" FM_FAKE_SENT="$sent" FM_FAKE_PANE_BUSY=0 \ fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null @@ -123,68 +122,6 @@ test_idle_pane_composer_clears_first_try() { pass "fm_tmux_submit_enter_core: idle pane clears composer on first Enter - returns empty as before" } -test_busy_pane_unknown_stays_unknown() { - local dir fakebin composer vfile - dir="$TMP_ROOT/busy-unknown" - fakebin=$(make_submit_mock "$dir") - composer="$dir/composer" - vfile="$dir/verdict" - printf '│ > unbounded\n' > "$composer" - touch "$dir/.swallow" - PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" FM_FAKE_PANE_BUSY=1 \ - FM_FAKE_SWALLOW="$dir/.swallow" FM_FAKE_PERSIST_SWALLOW=1 \ - fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null - [ "$(cat "$vfile")" = unknown ] \ - || fail "a busy pane must not convert an unsafe composer to empty, got '$(cat "$vfile")'" - pass "fm_tmux_submit_enter_core: busy conversion is limited to proven pending input" -} - -test_busy_pane_ambiguous_pending_retries_without_conversion() { - local dir fakebin composer sent vfile - dir="$TMP_ROOT/busy-ambiguous-pending" - fakebin=$(make_submit_mock "$dir") - composer="$dir/composer" - sent="$dir/sent.log" - vfile="$dir/verdict" - : > "$sent" - printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$composer" - touch "$dir/.swallow" - PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" fm_tmux_composer_state "win" > "$vfile" 2>/dev/null - [ "$(cat "$vfile")" = pending-unproven ] \ - || fail "ambiguous composer text should be pending-unproven, got '$(cat "$vfile")'" - PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" FM_FAKE_SENT="$sent" FM_FAKE_PANE_BUSY=1 \ - FM_FAKE_SWALLOW="$dir/.swallow" FM_FAKE_PERSIST_SWALLOW=1 \ - fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null - [ "$(cat "$vfile")" = pending-unproven ] \ - || fail "a busy pane must not convert pending-unproven to empty, got '$(cat "$vfile")'" - [ "$(grep -c '^Enter$' "$sent" 2>/dev/null || true)" -eq 3 ] \ - || fail "pending-unproven should consume the configured Enter retry budget" - pass "fm_tmux_submit_enter_core: pending-unproven retries without busy conversion" -} - -test_unrecognized_state_skips_busy_conversion() { - local dir fakebin composer busy_called vfile - dir="$TMP_ROOT/unrecognized-state" - fakebin=$(make_submit_mock "$dir") - composer="$dir/composer" - busy_called="$dir/busy-called" - vfile="$dir/verdict" - printf '╭─────╮\n│ > │\n╰─────╯\n' > "$composer" - ( - # shellcheck disable=SC2329 - fm_tmux_composer_state() { printf 'future-state'; } - # shellcheck disable=SC2329 - fm_pane_is_busy() { touch "$busy_called"; return 0; } - PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" \ - fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null - ) || fail "unrecognized-state submit check failed" - [ "$(cat "$vfile")" = future-state ] \ - || fail "unrecognized state should be preserved, got '$(cat "$vfile")'" - [ ! -e "$busy_called" ] \ - || fail "unrecognized state must not trigger busy conversion" - pass "fm_tmux_submit_enter_core: unrecognized states skip busy conversion" -} - test_claude_busy_signature_uses_real_capture_shapes() { local dir fakebin composer dir="$TMP_ROOT/claude-signature" @@ -251,7 +188,6 @@ test_claude_busy_signature_uses_real_capture_shapes() { pane_busy old-claude claude || fail "older Claude escape footer should be busy" printf 'Working...\n' > "$composer" pane_busy pi pi || fail "Pi Working footer should be busy" - pane_busy pi-signed pi-signed || fail "pi-signed should share Pi's exact Working footer" printf 'Ctrl+c:cancel\n' > "$composer" pane_busy grok grok || fail "Grok cancel footer should be busy" pass "fm_pane_is_busy: Claude spinner is scoped, multi-frame, and backward-compatible" @@ -261,7 +197,4 @@ test_busy_pane_pending_returns_empty test_idle_pane_pending_returns_pending test_busy_pane_composer_clears_first_try test_idle_pane_composer_clears_first_try -test_busy_pane_unknown_stays_unknown -test_busy_pane_ambiguous_pending_retries_without_conversion -test_unrecognized_state_skips_busy_conversion test_claude_busy_signature_uses_real_capture_shapes From 9f9238bbba89ed593d2e4dfb631b32b785bf9024 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Sat, 25 Jul 2026 19:38:54 -0700 Subject: [PATCH 14/52] feat: add verified Kimi crewmate adapter (#1047) * Add verified Kimi crewmate harness adapter * no-mistakes(review): Scope Kimi moon detection to spinner lines * no-mistakes(review): Match only complete Kimi spinner rows * no-mistakes(review): Resolve Kimi binary portably before pane creation * no-mistakes(document): Align Kimi adapter documentation * no-mistakes(lint): Suppress false-positive ShellCheck warning for sourced watcher override * Fix Kimi busy spinner detection * no-mistakes(review): Recognize Kimi session-lock ancestry and holders * no-mistakes(review): Scope pending-reply Kimi busy detection by harness * no-mistakes(document): Correct Kimi spinner capture documentation * no-mistakes(document): Clarify optional Kimi spinner whitespace * no-mistakes(lint): Silence intentional pending-reply test stub warnings * test: align rebased Kimi busy fixtures * no-mistakes: apply CI fixes * Reconcile Kimi busy detection after per-harness scoping * no-mistakes(review): Clarify observed Kimi spinner whitespace contract * no-mistakes(document): Clarify Kimi harness documentation --- .agents/skills/afk/SKILL.md | 2 +- .agents/skills/firstmate-orca/SKILL.md | 2 +- .agents/skills/harness-adapters/SKILL.md | 53 +++- AGENTS.md | 2 +- README.md | 2 +- bin/backends/tmux.sh | 20 +- bin/fm-bootstrap.sh | 6 +- bin/fm-harness.sh | 10 +- bin/fm-session-lock-lib.sh | 2 +- bin/fm-spawn.sh | 110 ++----- bin/fm-test-run.sh | 2 +- bin/fm-tmux-lib.sh | 11 + bin/fm-watch.sh | 3 +- docs/architecture.md | 2 +- docs/configuration.md | 6 +- docs/tmux-backend.md | 2 +- docs/turnend-guard.md | 4 +- docs/verification/runtime-backends.md | 1 + tests/fm-kimi-harness.test.sh | 374 ++++------------------- tests/fm-pending-reply.test.sh | 8 +- tests/fm-secondmate-liveness.test.sh | 26 +- tests/lib.sh | 11 +- 22 files changed, 181 insertions(+), 478 deletions(-) diff --git a/.agents/skills/afk/SKILL.md b/.agents/skills/afk/SKILL.md index 95f64b11e03..21b7cdd3111 100644 --- a/.agents/skills/afk/SKILL.md +++ b/.agents/skills/afk/SKILL.md @@ -84,7 +84,7 @@ The daemon constructs every current injection as the `away-supervisor` kind owne The bare `FM_INJECT_MARK` form remains accepted for legacy daemon escalations during rollout. U+2063 has no normal keyboard keystroke and survives terminal transport as UTF-8 text. This is how firstmate tells a daemon escalation apart from a real message in the same pane. -The operational prefix travels with the message text; it does not rely on harness-level typed-vs-injected detection, which is not portable across claude, codex, opencode, pi, pi-signed, grok, and kimi. +The operational prefix travels with the message text; it does not rely on harness-level typed-vs-injected detection, which is not portable across claude, codex, opencode, pi, grok, and kimi. ## Busy-guard and composer guard diff --git a/.agents/skills/firstmate-orca/SKILL.md b/.agents/skills/firstmate-orca/SKILL.md index d8d50b07b47..3db6d22c98c 100644 --- a/.agents/skills/firstmate-orca/SKILL.md +++ b/.agents/skills/firstmate-orca/SKILL.md @@ -13,7 +13,7 @@ It does not replace `AGENTS.md`, `docs/orca-backend.md`, or `harness-adapters`. Orca is a runtime backend, not an agent harness. The runtime backend owns the task endpoint and, for Orca, the task worktree. -The harness is the agent process launched inside that endpoint, such as `claude`, `codex`, `opencode`, `pi`, `pi-signed`, `grok`, or `kimi`. +The harness is the agent process launched inside that endpoint, such as `claude`, `codex`, `opencode`, `pi`, `grok`, or `kimi`. Load `harness-adapters` for harness-specific launch, interrupt, resume, trust-dialog, and skill-invocation facts. Implementation details, metadata fields, teardown guarantees, and limitations live in `docs/orca-backend.md`. diff --git a/.agents/skills/harness-adapters/SKILL.md b/.agents/skills/harness-adapters/SKILL.md index 9251130f4c7..0251462ec44 100644 --- a/.agents/skills/harness-adapters/SKILL.md +++ b/.agents/skills/harness-adapters/SKILL.md @@ -1,6 +1,6 @@ --- name: harness-adapters -description: Agent-only reference for firstmate harness operations. Use before spawning or recovering a crewmate or secondmate, handling a trust dialog, sending a harness-specific skill invocation, interrupting or exiting an agent, resuming an exited agent, or verifying a new harness adapter. Contains verified facts for claude, codex, opencode, pi, and grok. +description: Agent-only reference for firstmate harness operations. Use before spawning or recovering a crewmate or secondmate, handling a trust dialog, sending a harness-specific skill invocation, interrupting or exiting an agent, resuming an exited agent, or verifying a new harness adapter. Contains verified facts for claude, codex, opencode, pi, grok, and kimi. user-invocable: false metadata: internal: true @@ -25,7 +25,7 @@ If `config/crew-harness` is unset or `default`, there is no concrete value to in Inheritance also copies the literal `config/crew-dispatch.json` file, so secondmates apply the same best-fit profile rules for their own crewmates. Each adapter splits into mechanics and knowledge. -The per-task mechanics, including launch command, autonomy flag, and crewmate turn-end hook, live in `bin/fm-spawn.sh`. +The per-task mechanics, including launch command, autonomy flag, and any enabled crewmate turn-end hook, live in `bin/fm-spawn.sh`. The primary-session "no turn ends blind" guard contract and harness hook installation paths live in `docs/turnend-guard.md`. The primary-session watcher wake protocols are rendered from `docs/supervision-protocols/` by `bin/fm-supervision-instructions.sh`. The supervision knowledge lives here: busy signature, exit command, interrupt, dialogs, resume behavior, skill invocation, and quirks. @@ -50,16 +50,17 @@ Use that value for interrupt, exit, resume, and skill-invocation facts. ## Primary turn-end guard -Every verified primary harness has an empirically validated hook path for the "no turn ends blind" guard. +The primary integrations for `claude`, `codex`, `opencode`, `pi`, and `grok` have empirically validated hook paths for the "no turn ends blind" guard. `claude` and `codex` block directly through Stop hooks that preserve exit status 2 and stderr from `bin/fm-turnend-guard.sh`. `opencode`, `pi`, and `grok` expose passive lifecycle callbacks for this purpose, so their tracked primary adapters force one bounded follow-up or resume when the shared predicate blocks. +Kimi is outside the current turn-end integration scope; `docs/turnend-guard.md` owns the global-configuration boundary. The exact hook files, commands, scoping rules, and fail-open tradeoffs are owned by `docs/turnend-guard.md`. `docs/verification/supervision.md` "Turn-end guard" owns active validation evidence. When changing any primary turn-end hook, validate the real harness behavior in a scratch project or throwaway home before trusting it, then update that doc and the relevant concise fact below. ## Primary pre-arm (PreToolUse) seatbelt -Every verified primary harness also has a wired PreToolUse-equivalent hook that denies a watcher-arm anti-pattern (shell `&`, truncating pipe, bundling, broad `pkill -f fm-watch`) before it runs. +The primary integrations for `claude`, `codex`, `opencode`, `pi`, and `grok` also have wired PreToolUse-equivalent hooks that deny a watcher-arm anti-pattern (shell `&`, truncating pipe, bundling, broad `pkill -f fm-watch`) before it runs. `claude` and `codex` block directly through PreToolUse hooks; `grok` blocks the same way but requires every `$VAR` reference in its hook `command` string to carry an inline `:-default` or it fails to launch the hook entirely. `opencode` and `pi` block by throwing from `tool.execute.before` / returning `{block: true}` from `tool_call`. The exact hook files, commands, output-shaping quirks (Claude Code only honors the deny when stdout is empty), and validation transcripts are owned by `docs/arm-pretool-check.md`. @@ -121,6 +122,7 @@ The supported launch-profile flags below are verified locally; each row records | grok | `--model <model>` | `--reasoning-effort <low\|medium\|high>` | Verified on grok 0.2.99 (2026-07-13). `--effort` is an alias, but firstmate's profile axis is reasoning effort. As of 0.2.99 the ceiling is `high`; both `xhigh` and `max` are rejected with `use one of: high, medium, low`, so firstmate omits them. | | pi | `--model <model>` | `--thinking <low\|medium\|high\|xhigh\|max>` | Verified 2026-07-13 on Pi 0.80.6. `pi --help` advertises `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, and `max`; `pi --print --model openai-codex/gpt-5.6-sol --thinking max 'Reply with exactly OK.'` completed successfully. | | opencode | `--model <provider/model>` | none for firstmate's interactive launch | Verified on opencode 1.17.6. `opencode run` has `--variant`, but firstmate launches the interactive `opencode --prompt` path, which has no verified effort flag. | +| kimi | `--model <model>` | none | Verified 2026-07-25 on Kimi Code CLI 0.29.1. | ### Model support discovery @@ -134,6 +136,7 @@ Use the discovery surface in the current authenticated environment because suppo | opencode | Run `opencode models [provider]`, which lists available provider/model identifiers. | | pi | Run `pi --list-models [search]`; Pi's installed `docs/models.md` owns how built-in, extension-registered, and custom provider/model entries reach that list. | | grok | Run `grok models`, which lists the models available to the current Grok installation and account. | +| kimi | Run `kimi provider list --json`, which lists the current provider and model configuration. | For an unfamiliar harness or model namespace, establish support and provider identity from that harness's authoritative CLI help, model listing, or current documentation rather than guessing from a name or prefix. If those sources do not establish the relationship needed for dispatch, fail loudly and report the unresolved candidate. @@ -151,6 +154,13 @@ Natural language is acceptable if uncertain. - opencode: no separate verified skill invocation beyond normal slash-command behavior; use natural language if the exact skill command is uncertain. - pi: no separate verified skill invocation beyond normal command behavior; use natural language if the exact skill command is uncertain. - grok: `/<skill>`, for example `/no-mistakes` (same form as claude). Verified end to end: grok discovers the user-level `no-mistakes` skill, `/no-mistakes` invokes it, and grok drives a real `no-mistakes axi run`. Like codex's `$`/`/` popups, typing `/<skill>` opens grok's slash-autocomplete, so a too-fast Enter selects the popup entry instead of sending, and for an argument-taking command (like `/no-mistakes`'s optional task-first argument) that first Enter only expands the popup selection into an argument-hint placeholder rather than submitting - a genuine second Enter is required (see the grok section below for the 2026-07-03 incident and fix). `fm_tmux_submit_core`'s retried Enter (used by `fm-send` on the tmux backend) already handles this correctly by reading the cursor row; the herdr backend needed a dedicated fix (`fm_backend_herdr_composer_state`, docs/herdr-backend.md) because its prior delta-based verification false-positived on that same popup-close content change. +- kimi: `/<skill>`, for example `/no-mistakes`. + +## Submission acknowledgement hazards + +A send or key action reporting success is not proof that the intended action happened. +OpenCode can accept and queue an Enter while leaving text visible, Grok can consume Enter in its slash popup without submitting, and Kimi can silently drop a message sent before readiness even though the send returns success. +The shared symptom is a healthy-looking pane with no work in progress, so each adapter must verify the observable postcondition that is specific to its TUI. ## claude (VERIFIED; busy signature re-verified 2026-07-25 on Claude Code 2.1.220) @@ -334,3 +344,38 @@ The adapter therefore runs the shared predicate and, when it returns 2, forces o It does not pass `--permission-mode`, so the passive hook cannot escalate the primary session's tool permissions. Project-local Grok hooks require folder trust, verified with launch-time `--trust`; if the primary firstmate checkout is not trusted for Grok hooks, this primary guard fails open and `fm-guard.sh` remains the next-command alarm. Grok's primary watcher protocol is Claude-shaped background-notify around `bin/fm-watch-arm.sh`; the passive Stop hook is only a backstop for blind turn ends. + +## kimi (VERIFIED 2026-07-25, kimi 0.29.1) + +Kimi Code CLI launches from the absolute path resolved from `PATH`, falling back to the executable `$HOME/.kimi-code/bin/kimi`. + +| Fact | Value | +|---|---| +| Binary | Executable `kimi` from `PATH`, then executable `$HOME/.kimi-code/bin/kimi`; spawning refuses if neither exists. | +| Launch | Bare interactive TUI with `--auto`, followed by readiness-gated pointer delivery; positional prompts are rejected. | +| Models | `kimi-code/kimi-for-coding` (default), `kimi-code/kimi-for-coding-highspeed`, `kimi-code/k3`, and `kimi-code/k3-256k`. | +| Busy-pane signature | A transient line with optional leading whitespace, a rotating moon-phase glyph, optional whitespace around `·`, and optional trailing content; the line is absent when idle. | +| Exit command | `/exit` | +| Interrupt | Single Escape, which prints `Interrupted by user`. | +| Skill invocation | `/<skill>`, for example `/no-mistakes`; firstmate skills are discovered. | +| Autonomy | `--auto`; `-y` and `--yolo` are weaker and are not used. | +| Trust dialog | None on a clean first launch in a fresh pooled worktree. | +| Slash submission | One Enter submits, with no popup swallow or settle hazard. | +| Environment marker | None; detection relies on process ancestry command name `kimi`. | +| Composer | Bordered box with a bare `>` prompt glyph and no observed ghost or placeholder text. | +| Effort | No reasoning-effort flag exists, so requested effort is recorded in task metadata but omitted from launch. | + +`fm-spawn.sh` launches Kimi bare, waits for the composer box or `Welcome to Kimi Code!`, sends only `Read the brief at <absolute-path> and follow it exactly.`, and requires a cleared composer plus either the echoed `✨` submission or nonzero context before accepting delivery. +This launch-then-send shape is mandatory because Kimi rejects a positional brief as an unknown command. +Sending before readiness was reproduced as a silent drop with a zero exit status, an empty composer, `context: 0%`, no echoed user message, and a healthy-looking idle pane. +The brief path must be absolute because the brief lives outside the task worktree, and Kimi reads it there without `--add-dir`. + +Observed live spinner captures included optional leading whitespace, a moon-phase glyph, whitespace around `·`, and rotating tip text, with the same shape observed during tool execution. +Because those prose examples illustrate the spinner shape rather than define exact bytes, the matcher permits zero whitespace around `·` and does not require trailing tip text. +Kimi's footer tip rotates independently and can display `ctrl+c: cancel` while completely idle, so tip text is never used as its busy signature without the leading moon-plus-middot spinner structure. +The idle status bar can contain lowercase `thinking`, which is the model's effort label rather than a busy signal. +The spinner match covers the full moon-phase glyph set rather than one frame, but it remains locale- and emoji-font-sensitive because Kimi exposes no stable ASCII busy token. + +[`docs/turnend-guard.md`](../../../docs/turnend-guard.md) owns Kimi's verified global hook surface, approval boundary, and absence from the enabled integrations. +The current adapter falls back to idle detection. +That fallback is the weakest idle detection of any supported adapter because Kimi has no stable ASCII busy token, so turn completion can only be inferred from the fragile moon spinner disappearing and the pane becoming stable. diff --git a/AGENTS.md b/AGENTS.md index 583bda452c3..063207215a6 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -157,7 +157,7 @@ A silent bootstrap section needs no action; for any printed actionable diagnosti ## 4. Harness and runtime dispatch Load `harness-adapters` before every spawn or recovery and before trust handling, skill invocation, interrupt, exit, resume, or adapter verification. -The verified harnesses are `claude`, `codex`, `opencode`, `pi`, and `grok`; never dispatch on an unverified adapter. +The verified harnesses are `claude`, `codex`, `opencode`, `pi`, `grok`, and `kimi`; never dispatch on an unverified adapter. If static `config/crew-harness` or `config/secondmate-harness` names an unverified adapter, report it and fall back only to a verified adapter rather than launching it. `docs/configuration.md` owns dispatch-profile and runtime-backend schemas, `bin/fm-harness.sh` owns static resolution, and `bin/fm-spawn.sh` owns launch flags and fail-closed validation. diff --git a/README.md b/README.md index c62e47bba48..05465762659 100644 --- a/README.md +++ b/README.md @@ -58,7 +58,7 @@ Full detail on every feature lives in [docs/architecture.md](docs/architecture.m ### Requirements -- A verified agent harness: Claude Code, Grok, Pi, Codex, or OpenCode. +- A verified primary agent harness: Claude Code, Grok, Pi, Codex, or OpenCode. - Git and the GitHub CLI, authenticated through `gh auth login`. - The CLI and dependencies for your selected runtime backend; tmux is the reference default. diff --git a/bin/backends/tmux.sh b/bin/backends/tmux.sh index f8da21bf0de..b618e055bc4 100644 --- a/bin/backends/tmux.sh +++ b/bin/backends/tmux.sh @@ -117,22 +117,10 @@ fm_backend_tmux_send_literal() { # <target> <text> tmux send-keys -t "$1" -l "$2" } -# fm_backend_tmux_kill: remove one explicitly named task window, best-effort. -# Empty, omitted, and malformed targets return nonzero before invoking tmux so -# tmux can never interpret an empty target as the caller's current window. +# fm_backend_tmux_kill: remove the task's window, best-effort. Mirrors +# fm-teardown.sh's `tmux kill-window -t "$T" 2>/dev/null || true`. fm_backend_tmux_kill() { # <target> - local target=${1:-} session window - case "$target" in - *:*) - session=${target%%:*} - window=${target#*:} - ;; - *) return 1 ;; - esac - case "$session:$window" in - :*|*:|*:*:*) return 1 ;; - esac - tmux kill-window -t "=$session:=$window" 2>/dev/null || true + tmux kill-window -t "$1" 2>/dev/null || true } # fm_backend_tmux_current_command: <target>'s live foreground process name - @@ -193,7 +181,7 @@ fm_backend_tmux_agent_state() { # <target> } comm=${comm#-} case "$comm" in - *claude*|*codex*|*opencode*|*grok*|*kimi*|pi|pi-signed|pi-launcher|Pi) printf 'alive' ;; + *claude*|*codex*|*opencode*|*grok*|*kimi*) printf 'alive' ;; zsh|bash|sh|dash|ash|ksh|mksh|tcsh|csh|fish) printf 'dead' ;; '') printf 'unreadable' ;; *) printf 'ambiguous' ;; diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index 903532c480c..12001223353 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -436,7 +436,7 @@ secondmate_liveness_sweep() { [ -n "$target" ] || target="$window" agent_state=$(fm_backend_agent_state "$backend" "$target" 2>/dev/null) || agent_state=unreadable case "$harness" in - claude|codex|opencode|pi|grok) ;; + claude|codex|opencode|pi|grok|kimi) ;; *) case "$agent_state" in dead|missing) agent_state=unverified-harness ;; esac ;; @@ -713,7 +713,7 @@ crew_dispatch_validate() { return 0 fi err=$(jq -r ' - def verified($h): ["claude","codex","opencode","pi","grok"] | index($h); + def verified($h): ["claude","codex","opencode","pi","grok","kimi"] | index($h); def effort_ok($h; $e): if $e == null then true elif ($e | type) != "string" then false @@ -721,7 +721,7 @@ crew_dispatch_validate() { elif $h == "codex" then (["low","medium","high","xhigh"] | index($e)) elif $h == "grok" then (["low","medium","high"] | index($e)) elif $h == "pi" then (["low","medium","high","xhigh","max"] | index($e)) - elif $h == "opencode" then false + elif $h == "opencode" or $h == "kimi" then false else true end; def profiles($value): diff --git a/bin/fm-harness.sh b/bin/fm-harness.sh index 824b95804de..e9c1e1c24a3 100755 --- a/bin/fm-harness.sh +++ b/bin/fm-harness.sh @@ -1,6 +1,6 @@ #!/usr/bin/env bash # Detect the agent harness this process tree runs on. -# Usage: fm-harness.sh print own harness: claude|codex|opencode|pi|pi-signed|grok|kimi|unknown +# Usage: fm-harness.sh print own harness: claude|codex|opencode|pi|grok|kimi|unknown # fm-harness.sh crew print the effective CREWMATE harness # (config/crew-harness; "default" resolves to own) # fm-harness.sh secondmate print the harness the PRIMARY uses to launch @@ -36,10 +36,7 @@ detect_own() { # ancestry is consulted. This is a precedence hazard, not evidence that # CLAUDECODE inheritance into a kimi child was observed; it was not observed. [ "${CLAUDECODE:-}" = "1" ] && { echo claude; return; } - if [ "${PI_CODING_AGENT:-}" = "true" ]; then - if [ "${FM_PI_HARNESS:-}" = pi-signed ]; then echo pi-signed; else echo pi; fi - return - fi + [ "${PI_CODING_AGENT:-}" = "true" ] && { echo pi; return; } # grok sets GROK_AGENT=1 for its child/tool processes (verified, grok 0.2.73). # It does NOT set CLAUDECODE despite being Claude-Code-compatible, so this marker # is unambiguous when firstmate runs natively on grok. @@ -48,13 +45,12 @@ detect_own() { local pid=$$ comm args for _ in 1 2 3 4 5 6 7 8; do comm=$(ps -o comm= -p "$pid" 2>/dev/null) || break - case "$(basename -- "$comm")" in + case "$(basename "$comm")" in *claude*) echo claude; return ;; *codex*) echo codex; return ;; *opencode*) echo opencode; return ;; *grok*) echo grok; return ;; kimi) echo kimi; return ;; - pi-signed) echo pi; return ;; pi) echo pi; return ;; node*|python*) # Bare interpreter: match the harness name in its script path. diff --git a/bin/fm-session-lock-lib.sh b/bin/fm-session-lock-lib.sh index abc54e62853..73aab2f2136 100644 --- a/bin/fm-session-lock-lib.sh +++ b/bin/fm-session-lock-lib.sh @@ -9,7 +9,7 @@ # This file is sourced by scripts and has no side effects on source. # Known harness command names; extend when a new adapter is verified. -FM_HARNESS_RE='claude|codex|opencode|grok|^pi$' +FM_HARNESS_RE='claude|codex|opencode|grok|kimi|^pi$' # Walk the current process ancestry (up to 8 hops) and print the first pid whose # command looks like a verified harness. The harness pid lives as long as the diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index 3526572550c..0c47c0b9ab1 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -59,11 +59,10 @@ # profile consultation. A --secondmate spawn is exempt and resolves the SECONDMATE # harness (config/secondmate-harness -> config/crew-harness -> own), so the # secondmate-vs-crewmate split is DURABLE across every respawn (recovery, -# /updatefirstmate, restart). A bare adapter name (claude|codex|opencode|pi|pi-signed|grok|kimi) +# /updatefirstmate, restart). A bare adapter name (claude|codex|opencode|pi|grok|kimi) # overrides it for this spawn (either kind). A non-flag string containing # whitespace is treated as a RAW launch command - the escape hatch for verifying -# new adapters. pi-signed launches that exact executable name from PATH and -# refuses before endpoint creation when it is unavailable; it never falls back to pi. +# new adapters. # config/secondmate-harness may also carry an optional model and effort as extra # whitespace-separated tokens ("<harness> [<model>] [<effort>]"). For a # --secondmate spawn, those tokens apply only when this spawn also resolves its @@ -103,8 +102,7 @@ # __PIWATCH__ absolute path to .pi/extensions/fm-primary-pi-watch.ts in a pi secondmate home # __OPINPUT__ absolute path to the canonical operational-input encoder # Verified per-harness turn-end hooks are installed automatically where enabled; some live outside the worktree. -# Kimi uses one surgically installed Firstmate region in $HOME/.kimi-code/config.toml, -# a firstmate-owned global hook and registry, and a gitignored per-task pointer. +# Kimi has no enabled hook because its only verified Stop-hook configuration is global. # grok uses a firstmate-owned global hook under ${GROK_HOME:-$HOME/.grok}/hooks # plus a gitignored .fm-grok-turnend worktree pointer and a state token. # On success prints: spawned <id> harness=<name> kind=<ship|scout|secondmate> mode=<mode> yolo=<on|off> window=<backend-target> worktree=<path> @@ -124,26 +122,6 @@ esac FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" - -resolve_directory_input() { - local name=$1 path=$2 resolved - case "$path" in - /*) printf '%s\n' "$path"; return 0 ;; - esac - resolved=$(CDPATH='' cd -- "$path" 2>/dev/null && pwd -P) || { - echo "error: $name directory cannot be resolved: $path" >&2 - return 1 - } - printf '%s\n' "$resolved" -} - -FM_HOME=$(resolve_directory_input FM_HOME "$FM_HOME") || exit 1 -if [ -n "${FM_STATE_OVERRIDE:-}" ]; then - FM_STATE_OVERRIDE=$(resolve_directory_input FM_STATE_OVERRIDE "$FM_STATE_OVERRIDE") || exit 1 -fi -if [ -n "${FM_DATA_OVERRIDE:-}" ]; then - FM_DATA_OVERRIDE=$(resolve_directory_input FM_DATA_OVERRIDE "$FM_DATA_OVERRIDE") || exit 1 -fi STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" PROJECTS="${FM_PROJECTS_OVERRIDE:-$FM_HOME/projects}" @@ -409,7 +387,7 @@ FIRSTMATE_HOME= if [ "$KIND" = secondmate ]; then case "${POS[1]:-}" in - ''|claude|codex|opencode|pi|pi-signed|grok|kimi) + ''|claude|codex|opencode|pi|grok|kimi) ARG3=${POS[1]:-} ;; *' '*) @@ -455,11 +433,11 @@ launch_template() { fi ;; opencode) printf '%s' 'OPENCODE_CONFIG_CONTENT='\''{"permission":{"*":"allow"}}'\'' opencode __MODELFLAG__--prompt "$(__OPINPUT__ encode launch-brief < __BRIEF__)"' ;; - pi|pi-signed) + pi) if [ "$kind" = secondmate ]; then - printf '%s%s' "$harness" ' __MODELFLAG____EFFORTFLAG__-e __PITURNEND__ -e __PIWATCH__ "$(__OPINPUT__ encode launch-brief < __BRIEF__)"' + printf '%s' 'pi __MODELFLAG____EFFORTFLAG__-e __PITURNEND__ -e __PIWATCH__ "$(__OPINPUT__ encode launch-brief < __BRIEF__)"' else - printf '%s%s' "$harness" ' __MODELFLAG____EFFORTFLAG__-e __PIEXT__ "$(__OPINPUT__ encode launch-brief < __BRIEF__)"' + printf '%s' 'pi __MODELFLAG____EFFORTFLAG__-e __PIEXT__ "$(__OPINPUT__ encode launch-brief < __BRIEF__)"' fi ;; # grok (Grok Build TUI): a positional prompt starts the supervised interactive @@ -472,8 +450,8 @@ launch_template() { grok) printf '%s' 'grok --always-approve __MODELFLAG____EFFORTFLAG__"$(__OPINPUT__ encode launch-brief < __BRIEF__)"' ;; # Kimi Code rejects a positional prompt, so it launches bare and receives # only an absolute brief pointer after the TUI readiness gate below. - # Its turn-end signal is a globally configured Stop hook plus a guarded - # per-task worktree token, so no launch placeholder belongs here. + # No turn-end placeholder belongs here because its only verified Stop hook + # is global configuration and is not enabled without captain approval. kimi) printf '%s' '__KIMIBIN__ __MODELFLAG__--auto' ;; *) return 1 ;; esac @@ -515,18 +493,6 @@ case "$ARG3" in ;; esac -case "$HARNESS" in - pi|pi-signed) LAUNCH="FM_PI_HARNESS=$HARNESS $LAUNCH" ;; -esac - -# pi-signed is an explicitly selected executable identity, not an alias that may -# silently fall back to pi. Resolve it from PATH before creating an endpoint and -# retain the literal name in the launch command and task metadata. -if [ "$HARNESS" = pi-signed ] && ! command -v pi-signed >/dev/null 2>&1; then - echo "error: pi-signed executable not found on PATH; install the signed Pi wrapper or select a different verified harness" >&2 - exit 1 -fi - # config/secondmate-harness may carry optional model/effort tokens alongside the # harness ("<harness> [<model>] [<effort>]"). They apply only when this is a # --secondmate spawn and no explicit per-spawn harness/raw launch was supplied, so @@ -598,7 +564,7 @@ model_flag_for_harness() { local harness=$1 model=$2 [ -n "$model" ] && [ "$model" != default ] || return 0 case "$harness" in - claude|codex|opencode|pi|pi-signed|grok|kimi) + claude|codex|opencode|pi|grok|kimi) printf -- '--model %s ' "$(shell_quote "$model")" ;; esac @@ -630,7 +596,7 @@ effort_flag_for_harness() { low|medium|high) printf -- '--reasoning-effort %s ' "$(shell_quote "$effort")" ;; esac ;; - pi|pi-signed) + pi) # Pi 0.80.6 accepts the full shared effort vocabulary, including max, through # its --thinking flag. case "$effort" in @@ -649,12 +615,6 @@ case "$LAUNCH" in *__KIMIBIN__*) KIMI_BIN=$(resolve_kimi_binary) || exit 1 LAUNCH=${LAUNCH//__KIMIBIN__/$(shell_quote "$KIMI_BIN")} - if [ "$KIND" != secondmate ]; then - "$FM_ROOT/bin/fm-kimi-turnend-hook.sh" install || { - echo "error: refusing Kimi spawn because the global turn-end hook could not be installed safely" >&2 - exit 1 - } - fi ;; esac @@ -1318,11 +1278,12 @@ mkdir -p "$TASK_TMP/gotmp" # Per-harness turn-end hook where enabled: a file that touches # state/<id>.turn-ended when the agent finishes a turn. Worktree-resident hooks -# and token pointers stay out of git's view so they never block teardown's dirty -# check or leak into a commit. +# are kept out of git's view so they never block teardown's dirty check or leak +# into a commit. Kimi has no path because its global-only hook is not enabled. mkdir -p "$STATE" STATE_REAL=$(cd "$STATE" && pwd -P) -TURNEND="$STATE_REAL/$ID.turn-ended" +TURNEND= +[ "$HARNESS" = kimi ] || TURNEND="$STATE_REAL/$ID.turn-ended" exclude_path() { local rel=$1 EXCL EXCL=$(git -C "$WT" rev-parse --git-path info/exclude 2>/dev/null || true) @@ -1350,7 +1311,7 @@ export const FmTurnEnd = async ({ \$ }) => ({ EOF exclude_path '.opencode/plugins/fm-turn-end.js' ;; - pi|pi-signed) + pi*) # Written OUTSIDE the worktree: pi's project-trust gate fires on any extension # loaded from inside the project (verified live), but an explicit -e path # elsewhere loads without a dialog. Lives in state/, cleaned by teardown. @@ -1417,21 +1378,6 @@ EOF printf 'token=%s\n' "${auth_file##*/}" > "$WT/.fm-grok-turnend" exclude_path '.fm-grok-turnend' ;; - kimi*) - # Kimi's Stop hook is global, but it is inert unless cwd contains this - # task's token pointer and the token resolves through Firstmate's private - # registry. The installer above owns the format-preserving config edit and - # the always-zero, silent hook script. - KIMI_AUTH_DIR="$HOME/.kimi-code/fm-turn-end.d" - old_umask=$(umask) - umask 077 - auth_file=$(mktemp "$KIMI_AUTH_DIR/fm.XXXXXXXXXXXX") - umask "$old_umask" - printf '%s\n' "$TURNEND" > "$auth_file" - printf '%s\n' "${auth_file##*/}" > "$STATE/$ID.kimi-turnend-token" - printf 'token=%s\n' "${auth_file##*/}" > "$WT/.fm-kimi-turnend" - exclude_path '.fm-kimi-turnend' - ;; esac fi @@ -1455,7 +1401,6 @@ META_WINDOW=$T [ "$BACKEND" = orca ] && META_WINDOW=$W { echo "window=$META_WINDOW" - echo "endpoint_task_id=$ID" echo "worktree=$WT" echo "project=$PROJ_ABS" echo "harness=$HARNESS" @@ -1511,16 +1456,6 @@ LAUNCH=${LAUNCH//__PIEXT__/$sq_piext} LAUNCH=${LAUNCH//__PITURNEND__/$sq_piturnend} LAUNCH=${LAUNCH//__PIWATCH__/$sq_piwatch} LAUNCH=${LAUNCH//__OPINPUT__/$sq_opinput} -# Crewmate panes are created by a long-lived tmux/herdr daemon that does not -# inherit firstmate's current environment, so a bare `claude` in the pane falls -# back to the default ~/.claude store even when firstmate itself runs under a -# different CLAUDE_CONFIG_DIR (for example a work-vs-personal subscription split). -# Forward firstmate's own resolved store onto the claude launch so the crewmate -# uses the same credential/config firstmate is authenticated with. Only when set; -# an unset value is the single-store default and needs no prefix. -if [ "$HARNESS" = claude ] && [ -n "${CLAUDE_CONFIG_DIR:-}" ]; then - LAUNCH="CLAUDE_CONFIG_DIR=$(shell_quote "$CLAUDE_CONFIG_DIR") $LAUNCH" -fi if [ "$KIND" = secondmate ]; then sq_home=$(shell_quote "$PROJ_ABS") LAUNCH="FM_ROOT_OVERRIDE= FM_STATE_OVERRIDE= FM_DATA_OVERRIDE= FM_PROJECTS_OVERRIDE= FM_CONFIG_OVERRIDE= FM_HOME=$sq_home $LAUNCH" @@ -1543,16 +1478,11 @@ if [ "$HARNESS" = kimi ]; then exit 1 fi KIMI_POINTER="Read the brief at $BRIEF_REAL and follow it exactly." - KIMI_SUBMIT_RETRIES=${FM_KIMI_SUBMIT_RETRIES:-3} - KIMI_SUBMIT_SLEEP=${FM_KIMI_SUBMIT_SLEEP:-${FM_KIMI_POLL_INTERVAL:-0.5}} - KIMI_SUBMIT_SETTLE=${FM_KIMI_SUBMIT_SETTLE:-0} - KIMI_SUBMIT_VERDICT=$(fm_backend_send_text_submit \ - "$BACKEND" "$T" "$KIMI_POINTER" "$KIMI_SUBMIT_RETRIES" \ - "$KIMI_SUBMIT_SLEEP" "$KIMI_SUBMIT_SETTLE" "$W") || { - kimi_spawn_fail "kimi brief pointer could not be submitted" + if ! spawn_send_literal "$T" "$KIMI_POINTER"; then + kimi_spawn_fail "kimi brief pointer could not be typed" exit 1 - } - if [ "$KIMI_SUBMIT_VERDICT" = send-failed ]; then + fi + if ! spawn_send_key "$T" Enter; then kimi_spawn_fail "kimi brief pointer could not be submitted" exit 1 fi diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index c1e8ade8c79..c90d759c0db 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -122,7 +122,7 @@ family_for_basename() { fm-composer-ghost.test.sh|fm-composer-lib.test.sh|\ fm-crew-state.test.sh|fm-decision-hold-lifecycle.test.sh|\ fm-documentation-audiences.test.sh|fm-ensure-agents-md.test.sh|fm-grok-harness.test.sh|\ - fm-herdr-lab.test.sh|fm-instruction-owners.test.sh|fm-lint.test.sh|\ + fm-kimi-harness.test.sh|fm-herdr-lab.test.sh|fm-instruction-owners.test.sh|fm-lint.test.sh|\ fm-install-herdr.test.sh|fm-nm-test-contract.test.sh|fm-no-mistakes-ownership.test.sh|\ fm-operational-input.test.sh|fm-pi-primary-types.test.sh|\ fm-send-popup-settle.test.sh|fm-send-settle.test.sh|fm-stow-contract.test.sh|\ diff --git a/bin/fm-tmux-lib.sh b/bin/fm-tmux-lib.sh index 84ed3f67f21..350ff885f17 100755 --- a/bin/fm-tmux-lib.sh +++ b/bin/fm-tmux-lib.sh @@ -66,12 +66,22 @@ # line has an ellipsis followed by a parenthesized elapsed duration. Keep this # signature separate from the shared default because that shape is not generic # enough to classify arbitrary harness output safely. +# Kimi's anchored moon-phase spinner is separate because bare moon glyphs in +# ordinary output must not classify another harness as busy. Leading whitespace +# and whitespace around the middot are optional because prose examples preserve +# the spinner's shape rather than defining byte-exact spacing. The line end stays +# unanchored because rotating tip text follows but is not required. The idle +# status bar's lowercase `thinking` label and independently rotating tip text are +# not busy signals on their own. +# The full moon-phase set remains locale- and emoji-font-sensitive because Kimi +# exposes no stable ASCII busy token. FM_TMUX_BUSY_REGEX_DEFAULT='esc (to )?interrupt|Working\.\.\.|Ctrl\+c:cancel' FM_TMUX_CLAUDE_BUSY_REGEX_DEFAULT='esc to interrupt|…[[:space:]]+\([0-9]+[smh]' FM_TMUX_CODEX_BUSY_REGEX_DEFAULT='esc to interrupt' FM_TMUX_OPENCODE_BUSY_REGEX_DEFAULT='esc interrupt' FM_TMUX_PI_BUSY_REGEX_DEFAULT='Working\.\.\.' FM_TMUX_GROK_BUSY_REGEX_DEFAULT='Ctrl\+c:cancel' +FM_TMUX_KIMI_BUSY_REGEX_DEFAULT='^[[:space:]]*(🌑|🌒|🌓|🌔|🌕|🌖|🌗|🌘)[[:space:]]*·[[:space:]]*' fm_busy_lines_match() { # [harness] local harness=${1:-} lines regex @@ -85,6 +95,7 @@ fm_busy_lines_match() { # [harness] opencode) regex=$FM_TMUX_OPENCODE_BUSY_REGEX_DEFAULT ;; pi) regex=$FM_TMUX_PI_BUSY_REGEX_DEFAULT ;; grok) regex=$FM_TMUX_GROK_BUSY_REGEX_DEFAULT ;; + kimi) regex=$FM_TMUX_KIMI_BUSY_REGEX_DEFAULT ;; '') regex=$FM_TMUX_BUSY_REGEX_DEFAULT ;; *) # A supplied harness must never borrow another harness's signature. diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index d77118c3a2a..b7006e6362a 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -105,7 +105,8 @@ SIGNAL_GRACE=${FM_SIGNAL_GRACE:-30} # seconds to linger after a signal so trai # claude/codex: "esc to interrupt"; opencode: "esc interrupt"; pi: "Working..."; # grok: "Ctrl+c:cancel". Claude's current spinner signature is matched only for # a recorded Claude task because an ellipsis followed by elapsed time is not a -# safe shared signature for arbitrary harness output. +# safe shared signature for arbitrary harness output. Kimi's moon-plus-middot +# spinner signature is likewise matched only for a recorded Kimi task. BUSY_REGEX=${FM_BUSY_REGEX:-'esc (to )?interrupt|Working\.\.\.|Ctrl\+c:cancel'} # Always-on wake triage: most wakes during a long crew validation are benign (a # working: note or turn-end while a pipeline runs, a no-change heartbeat). Rather diff --git a/docs/architecture.md b/docs/architecture.md index f183ee91384..d1bbb1c1ff5 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -148,7 +148,7 @@ The session-start bootstrap step keeps valid dispatch configuration silent unles When the file exists, `fm-spawn.sh` refuses crewmate and scout launches without an explicit harness, so `config/crew-harness` is only automatic when no dispatch profile file is active. Secondmate launches are exempt because they resolve the secondmate harness and any optional secondmate model or effort tokens instead. Unsupported effort values are still recorded in task meta when passed to `fm-spawn.sh`, but the launch template omits any effort flag that the selected harness does not accept. -That keeps spawn launch compatible across claude, codex, grok, pi, and opencode while preserving the requested profile for later audit. +That keeps spawn launch compatible across claude, codex, grok, pi, opencode, and kimi while preserving the requested profile for later audit. ## Optional secondmates diff --git a/docs/configuration.md b/docs/configuration.md index 823a5e18f2a..8843b4740b0 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -174,10 +174,12 @@ The full cmux home label also includes a short hash of the resolved `FM_ROOT` pa ## Harness support -claude, codex, opencode, pi, and grok are all empirically verified; new harnesses get verified through a supervised trial task before joining the set. +claude, codex, opencode, pi, grok, and kimi are empirically verified for crewmate and secondmate launches; [README requirements](../README.md#requirements) own the narrower set supported for the primary session. +New harnesses get verified through a supervised trial task before joining the set. The verified adapter knowledge - busy signatures, interrupt and exit commands, skill-invocation syntax, and per-harness quirks - lives in [`.agents/skills/harness-adapters/SKILL.md`](../.agents/skills/harness-adapters/SKILL.md). Launch mechanics, including the verified command templates, live in [`bin/fm-spawn.sh`](../bin/fm-spawn.sh). -Primary-session turn-end guard integrations for verified harnesses are tracked as repo-level hook files and documented in [`docs/turnend-guard.md`](turnend-guard.md). +Enabled primary-session turn-end guard integrations are tracked as repo-level hook files and documented in [`docs/turnend-guard.md`](turnend-guard.md). +Kimi is outside the enabled integration scope; [`docs/turnend-guard.md`](turnend-guard.md#compatibility-limits) owns its verified hook boundary. Primary-session watcher wake protocols are rendered at session start by [`bin/fm-supervision-instructions.sh`](../bin/fm-supervision-instructions.sh) from [`docs/supervision-protocols/`](supervision-protocols/). Claude's Stop `asyncRewake` hook owns tokenless re-arm cycles, Grok uses background-notify cycles, Codex uses bounded foreground checkpoints, Pi uses its two tracked primary extensions, and OpenCode uses its TUI plugin. `config/crew-harness` is a local, gitignored file containing one adapter name for crewmate and scout launches. diff --git a/docs/tmux-backend.md b/docs/tmux-backend.md index 730fa702937..a33b385c142 100644 --- a/docs/tmux-backend.md +++ b/docs/tmux-backend.md @@ -46,7 +46,7 @@ Verify setup by spawning a small task and confirming its `fm-<id>` window appear A target-existence check proves only that the pane exists. The deeper tmux agent-liveness probe first verifies exact window membership, then reads `#{pane_current_command}` to distinguish a running harness process from a bare idle shell. -It classifies recognized Claude, Codex, OpenCode, and Grok process names as `alive`, common shells as `dead`, an authoritatively absent window as `missing`, unreadable state as `unreadable`, and every other process as `ambiguous`. +It classifies recognized Claude, Codex, OpenCode, Grok, and Kimi process names as `alive`, common shells as `dead`, an authoritatively absent window as `missing`, unreadable state as `unreadable`, and every other process as `ambiguous`. Only `dead` and `missing` authorize recovery because a false dead result could launch a duplicate agent. Pi runs through a generic `node` process name and cannot be attributed confidently from the tmux foreground-process field. diff --git a/docs/turnend-guard.md b/docs/turnend-guard.md index 08e1dc177f9..2fe72b0d2c1 100644 --- a/docs/turnend-guard.md +++ b/docs/turnend-guard.md @@ -3,7 +3,7 @@ This is the authoritative current contract for the "no turn ends blind" primary backstop referenced from AGENTS.md section 8. The predicate lives in `bin/fm-turnend-guard.sh`. Primary scope lives in `bin/fm-primary-scope-lib.sh`, shared with the native session-start nudge in [`sessionstart-nudge.md`](sessionstart-nudge.md). -Harness hook files only adapt each verified harness's turn-end mechanism to that shared predicate. +Harness hook files adapt each enabled primary harness integration's turn-end mechanism to that shared predicate. Related PreToolUse guards deny unsafe commands before execution rather than detecting a blind turn end afterward. Their separate owners are [`arm-pretool-check.md`](arm-pretool-check.md), [`cd-guard.md`](cd-guard.md), and [`subagent-guard.md`](subagent-guard.md). @@ -72,6 +72,8 @@ That warning uses `bin/fm-supervision-instructions.sh --repair-line`, so it alwa - A valid secondmate home is in scope; an idle secondmate endpoint with no X-mode relay poll remains healthy because it has no supervision need. - Claude and Codex block directly, while OpenCode, Pi, and Grok use bounded passive follow-ups. - OpenCode headless mode and untrusted Grok project hooks remain fail-open at the host boundary. +- Kimi Code CLI 0.29.1 exposes only global `[[hooks]]` configuration in `~/.kimi-code/config.toml`, including a `Stop` event with snake_case payload fields `hook_event_name`, `session_id`, `cwd`, and `stop_hook_active`. +- Kimi has no project-level hook configuration, so Firstmate does not modify that global file without captain approval, installs no Kimi hook, and writes no Kimi turn-end marker. - Missing `jq` or unreadable hook input remains fail-open. - No harness adapter uses a shell ampersand to manufacture supervision. diff --git a/docs/verification/runtime-backends.md b/docs/verification/runtime-backends.md index 046c12a02d0..b19135c3599 100644 --- a/docs/verification/runtime-backends.md +++ b/docs/verification/runtime-backends.md @@ -29,6 +29,7 @@ zsh A persistent parent shell waiting for a child remained reported as the parent process, while a shell that directly execed a simple command changed identity with the process itself. Claude, Codex, OpenCode, and Grok were observed under their own process names. +Kimi Code CLI 0.29.1 was observed under `kimi` on 2026-07-25. Pi remained a generic `node` process and is intentionally inconclusive. The OpenCode 1.18.4 busy-queue behavior and the tmux fallback are pinned by: diff --git a/tests/fm-kimi-harness.test.sh b/tests/fm-kimi-harness.test.sh index 8e27052d8ce..93ac12b4853 100755 --- a/tests/fm-kimi-harness.test.sh +++ b/tests/fm-kimi-harness.test.sh @@ -6,20 +6,31 @@ set -u . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" SPAWN="$ROOT/bin/fm-spawn.sh" -TEARDOWN="$ROOT/bin/fm-teardown.sh" -KIMI_HOOK="$ROOT/bin/fm-kimi-turnend-hook.sh" TMP_ROOT=$(fm_test_tmproot fm-kimi-harness) -KIMI_RUNTIME_TASK_TMP= -PYTHON_BIN=$(command -v python3) || fail "test needs python3" -PYTHON_BIN_DIR=$(dirname "$PYTHON_BIN") -JQ_BIN=$(command -v jq) || fail "test needs jq" -BASE_PATH=${FM_TEST_BASE_PATH:-$PYTHON_BIN_DIR:/usr/bin:/bin:/usr/sbin:/sbin} +BASE_PATH=${FM_TEST_BASE_PATH:-/usr/bin:/bin:/usr/sbin:/sbin} -cleanup_kimi_harness() { - [ -z "$KIMI_RUNTIME_TASK_TMP" ] || rm -rf "$KIMI_RUNTIME_TASK_TMP" - rm -rf "$TMP_ROOT" +assert_source_line() { + local line=$1 + grep -Fqx -- "$line" "$SPAWN" || fail "existing launch template changed: $line" +} + +test_existing_launch_templates_are_byte_pinned() { + assert_source_line " claude) printf '%s' 'CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false claude --dangerously-skip-permissions __MODELFLAG____EFFORTFLAG__\"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"' ;;" + assert_source_line " printf '%s' 'codex __MODELFLAG____EFFORTFLAG__--dangerously-bypass-approvals-and-sandbox \"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"'" + assert_source_line " printf '%s' 'codex __MODELFLAG____EFFORTFLAG__--dangerously-bypass-approvals-and-sandbox -c \"notify=[\\\"bash\\\",\\\"-c\\\",\\\"touch __TURNEND__\\\"]\" \"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"'" + assert_source_line " opencode) printf '%s' 'OPENCODE_CONFIG_CONTENT='\\''{\"permission\":{\"*\":\"allow\"}}'\\'' opencode __MODELFLAG__--prompt \"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"' ;;" + assert_source_line " printf '%s' 'pi __MODELFLAG____EFFORTFLAG__-e __PITURNEND__ -e __PIWATCH__ \"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"'" + assert_source_line " printf '%s' 'pi __MODELFLAG____EFFORTFLAG__-e __PIEXT__ \"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"'" + assert_source_line " grok) printf '%s' 'grok --always-approve __MODELFLAG____EFFORTFLAG__\"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"' ;;" + pass "fm-spawn: the five pre-existing adapters' launch templates stay byte-pinned" +} + +test_tracked_files_have_no_user_absolute_paths() { + local pattern="/""Users/" matches + matches=$(git -C "$ROOT" grep -n -F "$pattern" -- . || true) + [ -z "$matches" ] || fail "tracked files contain user-specific absolute paths: $matches" + pass "repository: tracked files contain no user-specific absolute paths" } -trap cleanup_kimi_harness EXIT make_spawn_fakebin() { local dir=$1 fakebin @@ -29,32 +40,9 @@ make_spawn_fakebin() { set -u printf '%s\n' "$*" >> "$FM_FAKE_TMUX_CALL_LOG" state=$(cat "$FM_FAKE_KIMI_STATE" 2>/dev/null || true) -fake_screen() { - case "$state" in - ready) - printf 'Welcome to Kimi Code!\ncontext: 0%% (0/256k)\n╭────────────────────────────────╮\n│ > │\n╰────────────────────────────────╯\n' - ;; - pointer-typed) - printf 'context: 0%% (0/256k)\n╭────────────────────────────────╮\n│ > Read the brief and follow it │\n│ │\n╰────────────────────────────────╯\n' - ;; - delivered) - printf '✨ Read the brief at %s and follow it exactly.\ncontext: 1%% (2k/256k)\n╭────────────────────────────────╮\n│ > │\n╰────────────────────────────────╯\n' "$FM_FAKE_BRIEF_REAL" - ;; - *) - printf 'shell starting\n$ \n' - ;; - esac -} -fake_cursor_y() { - case "$state" in - pointer-typed) printf '3\n' ;; - ready|delivered) printf '3\n' ;; - *) printf '1\n' ;; - esac -} case "$*" in *"#{pane_current_path}"*) printf '%s\n' "$FM_FAKE_PANE_PATH"; exit 0 ;; - *"#{cursor_y}"*) fake_cursor_y; exit 0 ;; + *"#{cursor_y}"*) printf '0\n'; exit 0 ;; esac case "${1:-}" in display-message) printf 'firstmate\n'; exit 0 ;; @@ -90,12 +78,7 @@ case "${1:-}" in ;; pointer-typed) if [ "${FM_FAKE_KIMI_DELIVERY:-yes}" = yes ]; then - if [ "${FM_FAKE_KIMI_SWALLOW_FIRST:-no}" = yes ] \ - && [ ! -f "$FM_FAKE_KIMI_SWALLOWED" ]; then - : > "$FM_FAKE_KIMI_SWALLOWED" - else - printf 'delivered\n' > "$FM_FAKE_KIMI_STATE" - fi + printf 'delivered\n' > "$FM_FAKE_KIMI_STATE" else printf 'ready\n' > "$FM_FAKE_KIMI_STATE" fi @@ -106,18 +89,19 @@ case "${1:-}" in exit 0 ;; capture-pane) - start= end= prev= - for arg in "$@"; do - case "$prev" in - -S) start=$arg ;; - -E) end=$arg ;; - esac - case "$arg" in -S|-E) prev=$arg ;; *) prev= ;; esac - done - case "$start:$end" in - *[!0-9:]*|'':*|*:'') fake_screen ;; - *) fake_screen | awk -v start="$start" -v end="$end" \ - 'NR - 1 >= start && NR - 1 <= end' ;; + case "$state" in + ready) + printf 'Welcome to Kimi Code!\ncontext: 0%% (0/256k)\n│ > │\n' + ;; + pointer-typed) + printf 'context: 0%% (0/256k)\n│ > pending │\n' + ;; + delivered) + printf '✨ Read the brief at %s and follow it exactly.\ncontext: 1%% (2k/256k)\n│ > │\n' "$FM_FAKE_BRIEF_REAL" + ;; + *) + printf 'shell starting\n$ \n' + ;; esac exit 0 ;; @@ -125,9 +109,8 @@ esac exit 0 SH chmod +x "$fakebin/tmux" - fm_fake_exit0 "$fakebin" treehouse gh-axi gh + fm_fake_exit0 "$fakebin" treehouse fm_fake_exit0 "$fakebin" kimi - ln -s "$JQ_BIN" "$fakebin/jq" printf '%s\n' "$fakebin" } @@ -138,8 +121,7 @@ make_spawn_case() { proj="$case_dir/project" wt="$case_dir/wt" fakebin=$(make_spawn_fakebin "$case_dir/fake") - mkdir -p "$home/data/$id" "$home/projects" "$home/state" "$home/config" "$home/.kimi-code" - printf '# Kimi test config\ndefault_model = "test"\n' > "$home/.kimi-code/config.toml" + mkdir -p "$home/data/$id" "$home/projects" "$home/state" "$home/config" printf 'brief for kimi\n' > "$home/data/$id/brief.md" printf 'kimi\n' > "$home/config/crew-harness" fm_git_worktree "$proj" "$wt" "wt-$name" @@ -161,8 +143,6 @@ run_spawn() { FM_FAKE_LAUNCH_LOG="$case_dir/launch.log" \ FM_FAKE_POINTER_LOG="$case_dir/pointer.log" \ FM_FAKE_KIMI_STATE="$case_dir/kimi.state" \ - FM_FAKE_KIMI_SWALLOWED="$case_dir/kimi.swallowed" \ - FM_FAKE_KIMI_SWALLOW_FIRST="${FM_FAKE_KIMI_SWALLOW_FIRST:-no}" \ FM_FAKE_TMUX_CALL_LOG="$case_dir/tmux-calls.log" \ FM_FAKE_BRIEF_REAL="$(cd "$home/data/$id" && pwd -P)/brief.md" \ FM_KIMI_READY_POLLS=2 FM_KIMI_DELIVERY_POLLS=2 FM_KIMI_POLL_INTERVAL=0 \ @@ -177,15 +157,11 @@ EOF } test_kimi_launch_then_send_is_verified() { - local id rec out rc launch pointer brief_real meta task_tmp - id="kimi-success-z1-$$" - task_tmp="/tmp/fm-$id" - KIMI_RUNTIME_TASK_TMP=$task_tmp - rm -rf "$task_tmp" + local id rec out rc launch pointer brief_real meta + id=kimi-success-z1 rec=$(make_spawn_case success "$id") read_spawn_record "$rec" - out=$(FM_FAKE_KIMI_SWALLOW_FIRST=yes run_spawn \ - "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id" \ + out=$(run_spawn "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id" \ --model kimi-code/k3 --effort high) rc=$? expect_code 0 "$rc" "verified kimi launch-then-send should succeed" @@ -195,7 +171,7 @@ test_kimi_launch_then_send_is_verified() { [ "$launch" = "'$FAKEBIN_DIR/kimi' --model 'kimi-code/k3' --auto" ] \ || fail "kimi launch did not use the absolute binary, model, and --auto only: $launch" assert_not_contains "$launch" "--effort" "kimi launch emitted a nonexistent effort flag" - assert_not_contains "$launch" "turn-ended" "kimi launch embedded a turn-end path" + assert_not_contains "$launch" "turn-ended" "kimi launch implied a turn-end marker" assert_not_contains "$launch" "__TURNEND__" "kimi launch retained a turn-end placeholder" brief_real="$(cd "$HOME_DIR/data/$id" && pwd -P)/brief.md" @@ -205,235 +181,8 @@ test_kimi_launch_then_send_is_verified() { meta="$HOME_DIR/state/$id.meta" assert_grep 'model=kimi-code/k3' "$meta" "kimi meta lost the requested model" assert_grep 'effort=high' "$meta" "kimi meta did not retain the unsupported effort axis" - assert_grep "tasktmp=$task_tmp" "$meta" "kimi meta did not record its task temp root" - assert_present "$task_tmp/gotmp" "kimi spawn did not create its Go temp directory" - assert_grep "export GOTMPDIR=$task_tmp/gotmp" "$CASE_DIR/tmux-calls.log" \ - "kimi spawn did not export its Go temp directory into the pane" - assert_grep 'BEGIN FIRSTMATE KIMI TURN-END HOOK' "$HOME_DIR/.kimi-code/config.toml" \ - "kimi spawn did not install its guarded global hook region" - assert_grep 'token=' "$WT_DIR/.fm-kimi-turnend" "kimi spawn did not write its token pointer" - assert_present "$HOME_DIR/state/$id.kimi-turnend-token" "kimi spawn did not record its token" - pass "fm-spawn: kimi launches, delivers its brief, and registers a guarded turn-end token" -} - -test_kimi_hook_install_is_surgical_idempotent_and_removable() { - local home config original once stripped count - home="$TMP_ROOT/config-surgery" - config="$home/.kimi-code/config.toml" - original="$home/original.toml" - once="$home/once.toml" - stripped="$home/stripped.toml" - mkdir -p "$home/.kimi-code" - cat > "$config" <<'EOF' -# Captain's leading comment stays exactly here. - -[ui] -theme = "night" # inline comment -show_usage = true - -# Foreign hook with intentionally unusual key ordering. -[[hooks]] -timeout=17 -command = "printf foreign" -matcher="" -event = "Stop" - -[providers.example] -model = "some/model" -# Final comment and blank line follow. - -EOF - cp "$config" "$original" - - HOME="$home" "$KIMI_HOOK" install || fail "Kimi hook install refused a realistic config" - cp "$config" "$once" - HOME="$home" "$KIMI_HOOK" install || fail "second Kimi hook install failed" - cmp -s "$once" "$config" || fail "second Kimi hook install changed config bytes" - count=$(grep -c '^# BEGIN FIRSTMATE KIMI TURN-END HOOK' "$config") - [ "$count" -eq 1 ] || fail "idempotent install left $count Firstmate regions" - - HOME="$home" "$KIMI_HOOK" remove || fail "Kimi hook removal failed" - cp "$config" "$stripped" - cmp -s "$original" "$stripped" \ - || fail "config with the Firstmate region excised was not byte-identical to the original" - assert_absent "$home/.kimi-code/fm-turn-end.sh" "removal left the Firstmate hook script" - assert_absent "$home/.kimi-code/fm-turn-end.d" "removal left the Firstmate registry" - pass "Kimi hook install is idempotent and removal restores every foreign config byte" -} - -test_kimi_hook_remove_preserves_owned_newline_boundary() { - local appended config expected home original - home="$TMP_ROOT/config-owned-newline" - config="$home/.kimi-code/config.toml" - original="$home/original.toml" - expected="$home/expected.toml" - appended="$home/appended.toml" - mkdir -p "$home/.kimi-code" - printf 'default_model = "test"' > "$config" - cp "$config" "$original" - - HOME="$home" "$KIMI_HOOK" install || fail "Kimi hook install refused config without a final newline" - HOME="$home" "$KIMI_HOOK" remove || fail "Kimi hook removal failed without appended config" - cmp -s "$original" "$config" \ - || fail "pristine removal did not restore the absent final newline byte-identically" - - HOME="$home" "$KIMI_HOOK" install || fail "second Kimi hook install refused config without a final newline" - printf '[captain]\nenabled = true\n' > "$appended" - cat "$appended" >> "$config" - HOME="$home" "$KIMI_HOOK" remove || fail "Kimi hook removal joined config appended after its region" - { - cat "$original" - printf '\n' - cat "$appended" - } > "$expected" - cmp -s "$expected" "$config" \ - || fail "removal did not preserve appended captain config on its own line" - "$PYTHON_BIN" - "$config" <<'PY' || fail "config with appended captain TOML did not parse after removal" -import sys -import tomllib - -with open(sys.argv[1], "rb") as stream: - tomllib.load(stream) -PY - pass "Kimi hook removal preserves owned newline boundaries and pristine bytes" -} - -test_kimi_hook_fails_closed_on_missing_malformed_or_partial_config() { - local missing malformed partial out rc - missing="$TMP_ROOT/config-missing" - malformed="$TMP_ROOT/config-malformed" - partial="$TMP_ROOT/config-partial" - mkdir -p "$missing/.kimi-code" "$malformed/.kimi-code" "$partial/.kimi-code" - - rc=0 - out=$(HOME="$missing" "$KIMI_HOOK" install 2>&1) || rc=$? - [ "$rc" -ne 0 ] || fail "missing Kimi config was accepted" - assert_contains "$out" "Kimi config is missing" "missing config refusal lacked its concrete reason" - assert_absent "$missing/.kimi-code/fm-turn-end.sh" "missing config refusal wrote the hook script" - - printf '[broken\n' > "$malformed/.kimi-code/config.toml" - cp "$malformed/.kimi-code/config.toml" "$malformed/before" - rc=0 - out=$(HOME="$malformed" "$KIMI_HOOK" install 2>&1) || rc=$? - [ "$rc" -ne 0 ] || fail "malformed Kimi config was accepted" - assert_contains "$out" "malformed TOML" "malformed config refusal lacked its concrete reason" - cmp -s "$malformed/before" "$malformed/.kimi-code/config.toml" \ - || fail "malformed config refusal changed config bytes" - assert_absent "$malformed/.kimi-code/fm-turn-end.sh" "malformed config refusal wrote the hook script" - - printf '# BEGIN FIRSTMATE KIMI TURN-END HOOK\n' > "$partial/.kimi-code/config.toml" - cp "$partial/.kimi-code/config.toml" "$partial/before" - rc=0 - out=$(HOME="$partial" "$KIMI_HOOK" install 2>&1) || rc=$? - [ "$rc" -ne 0 ] || fail "partial Firstmate marker was accepted" - assert_contains "$out" "partial, duplicated, or altered" "partial marker refusal lacked its concrete reason" - cmp -s "$partial/before" "$partial/.kimi-code/config.toml" \ - || fail "partial marker refusal changed config bytes" - pass "Kimi hook install refuses missing, malformed, and surprising config without writing" -} - -test_kimi_hook_install_refuses_without_jq() { - local home config before fakebin out rc - home="$TMP_ROOT/config-no-jq" - config="$home/.kimi-code/config.toml" - before="$home/config-before.toml" - fakebin=$(fm_fakebin "$home/no-jq") - mkdir -p "$home/.kimi-code" - printf '# Captain config\nmodel = "test"\n' > "$config" - cp "$config" "$before" - ln -s "$(command -v bash)" "$fakebin/bash" - ln -s "$(command -v python3)" "$fakebin/python3" - - rc=0 - out=$(HOME="$home" PATH="$fakebin" "$KIMI_HOOK" install 2>&1) || rc=$? - [ "$rc" -ne 0 ] || fail "Kimi hook install succeeded without jq" - assert_contains "$out" "jq is required" "missing-jq refusal did not name jq" - cmp -s "$before" "$config" || fail "missing-jq refusal changed config bytes" - assert_absent "$home/.kimi-code/fm-turn-end.sh" "missing-jq refusal wrote the hook script" - assert_absent "$home/.kimi-code/fm-turn-end.d" "missing-jq refusal wrote the registry" - pass "Kimi hook install refuses without jq before any config write" -} - -test_kimi_hook_is_silent_and_requires_registered_workspace_token() { - local id rec out rc hook target token no_token snapshot_before snapshot_after fakebin - id=kimi-hook-auth-z6 - rec=$(make_spawn_case hook-auth "$id") - read_spawn_record "$rec" - out=$(run_spawn "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id") - rc=$? - expect_code 0 "$rc" "Kimi spawn should succeed before hook authentication checks" - hook="$HOME_DIR/.kimi-code/fm-turn-end.sh" - target="$HOME_DIR/state/$id.turn-ended" - token=$(sed -n 's/^token=//p' "$WT_DIR/.fm-kimi-turnend") - assert_present "$HOME_DIR/.kimi-code/fm-turn-end.d/$token" "Kimi registry token is missing" - - no_token="$CASE_DIR/no-token-workspace" - mkdir -p "$no_token" - snapshot_before=$(find "$no_token" -mindepth 1 -print) - out=$(printf '{"hook_event_name":"Stop","session_id":"ordinary","cwd":"%s","stop_hook_active":false}\n' "$no_token" \ - | HOME="$HOME_DIR" bash "$hook" 2>&1) - rc=$? - expect_code 0 "$rc" "Kimi hook must never block a tokenless session" - [ -z "$out" ] || fail "Kimi hook printed into a tokenless session: $out" - snapshot_after=$(find "$no_token" -mindepth 1 -print) - [ "$snapshot_before" = "$snapshot_after" ] || fail "Kimi hook wrote inside a tokenless workspace" - assert_absent "$target" "tokenless Kimi hook invocation touched a task marker" - - printf 'token=%s\n' "$token" > "$WT_DIR/.fm-kimi-turnend" - out=$(printf '{"hook_event_name":"Stop","session_id":"crew","cwd":"%s","stop_hook_active":false}\n' "$WT_DIR" \ - | HOME="$HOME_DIR" bash "$hook" 2>&1) - rc=$? - expect_code 0 "$rc" "registered Kimi hook invocation did not exit zero" - [ -z "$out" ] || fail "registered Kimi hook invocation printed output: $out" - assert_present "$target" "registered Kimi hook invocation did not touch the turn-end marker" - - rm "$target" - fakebin=$(fm_fakebin "$CASE_DIR/no-jq") - ln -s "$(command -v bash)" "$fakebin/bash" - out=$(printf '{"hook_event_name":"Stop","session_id":"crew","cwd":"%s","stop_hook_active":false}\n' "$WT_DIR" \ - | HOME="$HOME_DIR" PATH="$fakebin" "$hook" 2>&1) - rc=$? - expect_code 0 "$rc" "Kimi hook without jq must still exit zero" - [ -z "$out" ] || fail "Kimi hook without jq printed output: $out" - assert_absent "$target" "Kimi hook without jq touched the turn-end marker" - pass "Kimi hook stays silent and inert without a Firstmate registry token" -} - -test_kimi_spawn_refuses_unsafe_global_config_before_pane_creation() { - local id rec out rc - id=kimi-config-refuse-z7 - rec=$(make_spawn_case config-refuse "$id") - read_spawn_record "$rec" - printf '[malformed\n' > "$HOME_DIR/.kimi-code/config.toml" - rc=0 - out=$(run_spawn "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id") || rc=$? - [ "$rc" -ne 0 ] || fail "Kimi spawn accepted malformed global config" - assert_contains "$out" "malformed TOML" "Kimi spawn omitted the concrete config refusal" - if grep -Eq '(^| )new-(session|window)( |$)' "$CASE_DIR/tmux-calls.log"; then - fail "unsafe Kimi config refusal created a tmux container or pane" - fi - pass "fm-spawn: unsafe Kimi global config refuses before pane creation" -} - -test_kimi_teardown_removes_pointer_and_registry_token() { - local id rec out rc token - id=kimi-teardown-z8 - rec=$(make_spawn_case teardown "$id") - read_spawn_record "$rec" - out=$(run_spawn "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id") - rc=$? - expect_code 0 "$rc" "Kimi spawn should succeed before teardown" - token=$(sed -n 's/^token=//p' "$WT_DIR/.fm-kimi-turnend") - - HOME="$HOME_DIR" FM_ROOT_OVERRIDE="$ROOT" FM_HOME="$HOME_DIR" \ - FM_STATE_OVERRIDE="$HOME_DIR/state" FM_DATA_OVERRIDE="$HOME_DIR/data" \ - FM_PROJECTS_OVERRIDE="$HOME_DIR/projects" FM_CONFIG_OVERRIDE="$HOME_DIR/config" \ - FM_SPAWN_NO_GUARD=1 PATH="$FAKEBIN_DIR:$BASE_PATH" \ - "$TEARDOWN" "$id" --force >/dev/null 2>&1 || fail "Kimi teardown failed" - assert_absent "$WT_DIR/.fm-kimi-turnend" "Kimi token pointer survived teardown" - assert_absent "$HOME_DIR/.kimi-code/fm-turn-end.d/$token" "Kimi registry token survived teardown" - assert_absent "$HOME_DIR/state/$id.kimi-turnend-token" "Kimi token state survived teardown" - pass "fm-teardown: Kimi task pointer and registry token are removed" + assert_absent "$HOME_DIR/.kimi-code/config.toml" "kimi spawn wrote a global config file" + pass "fm-spawn: kimi launches bare, waits for readiness, sends an absolute brief pointer, and confirms delivery" } test_kimi_falls_back_to_expanded_home_binary() { @@ -566,7 +315,7 @@ SH } test_kimi_busy_signature_is_scoped_to_spinner_lines() { - local capture + local capture phase kimi_regex_lines # shellcheck source=/dev/null . "$ROOT/bin/fm-tmux-lib.sh" unset FM_BUSY_REGEX @@ -578,13 +327,11 @@ test_kimi_busy_signature_is_scoped_to_spinner_lines() { esac } # These fixtures reproduce the observed spinner shape rather than byte-exact - # transcriptions. Leading whitespace is deliberately varied; separator whitespace - # follows the captured contract. - local phase - for phase in 🌑 🌒 🌓 🌔 🌕 🌖 🌗 🌘; do - printf ' %s · Tip: Kimi is working\n│ > │\n' "$phase" > "$capture" - fm_pane_is_busy fake kimi || fail "Kimi spinner phase $phase was not recognized as busy" - done + # transcriptions. Leading and separator whitespace are deliberately varied. + printf ' 🌑 · Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"\n│ > │\n' > "$capture" + fm_pane_is_busy fake kimi || fail "the first real Kimi spinner shape was not recognized as busy" + printf ' 🌗·Tip: /plugins: manage plugins ...\n│ > │\n' > "$capture" + fm_pane_is_busy fake kimi || fail "the tool-execution Kimi spinner shape was not recognized as busy" printf 'ordinary response ending with 🌕\n│ > │\n' > "$capture" if fm_pane_is_busy fake kimi; then fail "a moon outside Kimi's spinner-line shape was misread as busy" @@ -609,11 +356,19 @@ test_kimi_busy_signature_is_scoped_to_spinner_lines() { if fm_pane_is_busy fake kimi; then fail "Kimi's idle thinking-effort status label was misread as busy" fi + kimi_regex_lines=$(grep 'KIMI_BUSY_REGEX' "$ROOT/bin/fm-tmux-lib.sh" "$ROOT/bin/fm-watch.sh") + if printf '%s\n' "$kimi_regex_lines" | grep -qi thinking; then + fail "Kimi busy regex still depends on a Thinking or thinking token" + fi + for phase in 🌑 🌒 🌓 🌔 🌕 🌖 🌗 🌘; do + grep -Fq "$phase" "$ROOT/bin/fm-tmux-lib.sh" \ + || fail "shared Kimi matcher is missing moon phase $phase" + done pass "busy detection: real Kimi moon-plus-middot captures require its harness while idle labels stay idle" } test_watcher_scopes_moon_spinner_to_recorded_kimi_task() ( - local state="$TMP_ROOT/watch-state" busy_capture=' 🌑 · Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"' + local state="$TMP_ROOT/watch-state" busy_capture=' 🌑· Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"' mkdir -p "$state" printf 'window=fake\nharness=kimi\n' > "$state/kimi-watch.meta" unset FM_BUSY_REGEX @@ -657,14 +412,9 @@ test_kimi_bordered_prompt_needs_no_override() { pass "composer classifier: kimi's existing bordered > shape is already safe without an override" } -test_kimi_hook_install_is_surgical_idempotent_and_removable -test_kimi_hook_remove_preserves_owned_newline_boundary -test_kimi_hook_fails_closed_on_missing_malformed_or_partial_config -test_kimi_hook_install_refuses_without_jq +test_tracked_files_have_no_user_absolute_paths +test_existing_launch_templates_are_byte_pinned test_kimi_launch_then_send_is_verified -test_kimi_hook_is_silent_and_requires_registered_workspace_token -test_kimi_spawn_refuses_unsafe_global_config_before_pane_creation -test_kimi_teardown_removes_pointer_and_registry_token test_kimi_falls_back_to_expanded_home_binary test_kimi_missing_binary_refuses_before_pane_creation test_kimi_unconfirmed_delivery_fails_loudly diff --git a/tests/fm-pending-reply.test.sh b/tests/fm-pending-reply.test.sh index 325125eeeb2..254b03a6317 100755 --- a/tests/fm-pending-reply.test.sh +++ b/tests/fm-pending-reply.test.sh @@ -59,9 +59,9 @@ case "${1:-}" in fi exit 0 ;; display-message) - for a in "$@"; do case "$a" in *cursor_y*) printf '1\n'; exit 0 ;; esac; done + for a in "$@"; do case "$a" in *cursor_y*) printf '0\n'; exit 0 ;; esac; done printf 'fakepane\n'; exit 0 ;; - capture-pane) printf '╭────╮\n│ │\n╰────╯\n'; exit 0 ;; + capture-pane) printf '\xe2\x94\x82 \xe2\x94\x82\n'; exit 0 ;; list-windows) exit 0 ;; esac exit 0 @@ -725,14 +725,14 @@ test_kimi_capture_fallback_uses_recorded_harness() ( fm_write_secondmate_meta "$state/hibit.meta" "$sm_home" "session:fm-hibit" alpha kimi fm_backend_busy_state() { printf 'unknown'; } fm_backend_capture() { printf '%s' "$FM_PENDING_KIMI_CAPTURE"; } - export FM_PENDING_KIMI_CAPTURE=' 🌑 · Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"' + export FM_PENDING_KIMI_CAPTURE=' 🌑 ·Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"' [ "$(fm_pending_reply_backend_observation tmux session:fm-hibit fm-hibit codex)" = fallback-idle ] \ || fail "Kimi spinner leaked into another harness" export FM_PENDING_KIMI_CAPTURE='Ctrl+c:cancel' [ "$(fm_pending_reply_backend_observation tmux session:fm-hibit fm-hibit kimi)" = fallback-idle ] \ || fail "Grok's exact busy token leaked into Kimi pending-reply observation" - export FM_PENDING_KIMI_CAPTURE=' 🌑 · Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"' + export FM_PENDING_KIMI_CAPTURE=' 🌑 ·Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"' fm_pending_reply_tick "$state" rec=$(fm_pending_reply_path "$state" "$corr") [ "$(fm_pending_reply_get "$rec" turn_seen_busy)" = 1 ] \ diff --git a/tests/fm-secondmate-liveness.test.sh b/tests/fm-secondmate-liveness.test.sh index ed356638962..2b572681572 100755 --- a/tests/fm-secondmate-liveness.test.sh +++ b/tests/fm-secondmate-liveness.test.sh @@ -97,7 +97,7 @@ SH test_tmux_agent_state_classifies() { local fb out - for harness in claude codex opencode grok kimi pi pi-signed pi-launcher Pi; do + for harness in claude codex opencode grok kimi; do fb=$(make_probe_tmux "$TMP_ROOT/tmux-$harness" "$harness") out=$(PATH="$fb:$BASE_PATH" bash -c '. "$0/bin/fm-backend.sh"; fm_backend_agent_state tmux sess:win' "$ROOT") [ "$out" = alive ] || fail "a live $harness foreground process should classify as alive, got '$out'" @@ -206,7 +206,7 @@ test_agent_state_dispatcher_and_compatibility() { make_toolchain() { local dir=$1 fakebin fakebin=$(fm_fakebin "$dir") - fm_fake_exit0 "$fakebin" node gh-axi chrome-devtools-axi lavish-axi pi-signed + fm_fake_exit0 "$fakebin" node gh-axi chrome-devtools-axi lavish-axi cat > "$fakebin/gh" <<'SH' #!/usr/bin/env bash exit 0 @@ -349,7 +349,7 @@ test_sweep_respawns_confirmed_dead_secondmate() { assert_not_contains "$out" "SECONDMATE_LIVENESS: secondmate sm1: respawned" \ "a successfully respawned secondmate should be handled silently" - assert_contains "$(cat "$log")" "kill-window -t =firstmate:=fm-sm1" \ + assert_contains "$(cat "$log")" "kill-window -t firstmate:fm-sm1" \ "the stale endpoint must be killed before respawn (tmux refuses a same-named window over a live one)" assert_contains "$(cat "$log")" "new-window" \ "a confirmed-dead secondmate should actually be relaunched" @@ -391,25 +391,6 @@ test_sweep_respawns_authoritatively_missing_pi_secondmate() { pass "sweep: an authoritatively missing Pi secondmate window is relaunched" } -test_sweep_respawns_authoritatively_missing_pi_signed_secondmate() { - local w fb tmuxfb log out - w=$(new_world sweep-missing-pi-signed) - printf '%s\n' pi-signed > "$w/home/config/secondmate-harness" - add_sm_home "$w" sm1 firstmate:fm-sm1 pi-signed - fb=$(make_toolchain "$w"); tmuxfb=$(make_liveness_tmux "$w") - log="$w/calls.log"; : > "$log" - - out=$(run_bootstrap "$tmuxfb:$fb" "$w/home" missing "$log") - - assert_not_contains "$out" "unverified for recovery" \ - "a recorded pi-signed secondmate should be verified for recovery" - assert_contains "$(cat "$log")" "new-window" \ - "an authoritatively missing pi-signed secondmate should be relaunched" - assert_not_contains "$(cat "$log")" "kill-window" \ - "an absent pi-signed window should not need a destructive pre-kill" - pass "sweep: an authoritatively missing pi-signed secondmate window is relaunched" -} - test_sweep_never_acts_on_ambiguous_existing_process() { local w fb tmuxfb log out w=$(new_world sweep-ambiguous) @@ -533,7 +514,6 @@ test_agent_state_dispatcher_and_compatibility test_sweep_respawns_confirmed_dead_secondmate test_sweep_leaves_alive_secondmate_untouched test_sweep_respawns_authoritatively_missing_pi_secondmate -test_sweep_respawns_authoritatively_missing_pi_signed_secondmate test_sweep_never_acts_on_ambiguous_existing_process test_sweep_never_acts_on_transient_unreadability test_sweep_reports_missing_endpoint_relaunch_failure diff --git a/tests/lib.sh b/tests/lib.sh index ee3b1d1476c..d33062915ff 100644 --- a/tests/lib.sh +++ b/tests/lib.sh @@ -152,16 +152,13 @@ fm_write_meta() { } # fm_write_secondmate_meta <file> <home> [window] [projects] [harness]: write the -# standard kind=secondmate meta block used across the secondmate suites. Window -# defaults to firstmate:fm-<id>, projects defaults to alpha, and harness defaults -# to echo to match the common case. +# standard kind=secondmate meta block used across the secondmate suites. window +# is explicit and defaults to firstmate:fm-domain, projects defaults to alpha, +# and harness defaults to echo to match the common case. fm_write_secondmate_meta() { - local file=$1 home=$2 id window projects=${4:-alpha} harness=${5:-echo} - id=$(basename "$file" .meta) - window=${3:-firstmate:fm-$id} + local file=$1 home=$2 window=${3:-firstmate:fm-domain} projects=${4:-alpha} harness=${5:-echo} fm_write_meta "$file" \ "window=$window" \ - "endpoint_task_id=$id" \ "worktree=$home" \ "project=$home" \ "harness=$harness" \ From 8870fc810615b29a75e2cc487d0f5749fed7fa12 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Sat, 25 Jul 2026 20:57:57 -0700 Subject: [PATCH 15/52] fix: harden Kimi submission and spinner matching (#1058) * fix kimi pointer submission and spinner conformance * no-mistakes(review): Preserve Kimi submit target ownership guard --- .agents/skills/harness-adapters/SKILL.md | 6 ++++-- bin/fm-spawn.sh | 13 +++++++++---- bin/fm-tmux-lib.sh | 15 ++++++++------- tests/fm-kimi-harness.test.sh | 19 ++++++++++++++----- 4 files changed, 35 insertions(+), 18 deletions(-) diff --git a/.agents/skills/harness-adapters/SKILL.md b/.agents/skills/harness-adapters/SKILL.md index 0251462ec44..9da5ed25f52 100644 --- a/.agents/skills/harness-adapters/SKILL.md +++ b/.agents/skills/harness-adapters/SKILL.md @@ -354,7 +354,7 @@ Kimi Code CLI launches from the absolute path resolved from `PATH`, falling back | Binary | Executable `kimi` from `PATH`, then executable `$HOME/.kimi-code/bin/kimi`; spawning refuses if neither exists. | | Launch | Bare interactive TUI with `--auto`, followed by readiness-gated pointer delivery; positional prompts are rejected. | | Models | `kimi-code/kimi-for-coding` (default), `kimi-code/kimi-for-coding-highspeed`, `kimi-code/k3`, and `kimi-code/k3-256k`. | -| Busy-pane signature | A transient line with optional leading whitespace, a rotating moon-phase glyph, optional whitespace around `·`, and optional trailing content; the line is absent when idle. | +| Busy-pane signature | A transient line with optional leading whitespace, a rotating moon-phase glyph, required whitespace on both sides of `·`, and optional trailing content; the line is absent when idle. | | Exit command | `/exit` | | Interrupt | Single Escape, which prints `Interrupted by user`. | | Skill invocation | `/<skill>`, for example `/no-mistakes`; firstmate skills are discovered. | @@ -371,7 +371,9 @@ Sending before readiness was reproduced as a silent drop with a zero exit status The brief path must be absolute because the brief lives outside the task worktree, and Kimi reads it there without `--add-dir`. Observed live spinner captures included optional leading whitespace, a moon-phase glyph, whitespace around `·`, and rotating tip text, with the same shape observed during tool execution. -Because those prose examples illustrate the spinner shape rather than define exact bytes, the matcher permits zero whitespace around `·` and does not require trailing tip text. +Because every captured spinner row had whitespace on both sides of `·`, the matcher requires that whitespace, deliberately does not match the never-observed zero-whitespace form, and does not require trailing tip text. +The startup input-readiness window is the established cause of Kimi's first-Enter delivery defect: the banner is not the cause, and Grok's cursor-row quirk does not apply. +No rendering signal is trustworthy for proving that Kimi will accept input during this window, so delivery retries Enter through the shared submit core and retains the existing postcondition verification rather than relaxing readiness or delivery checks. Kimi's footer tip rotates independently and can display `ctrl+c: cancel` while completely idle, so tip text is never used as its busy signature without the leading moon-plus-middot spinner structure. The idle status bar can contain lowercase `thinking`, which is the model's effort label rather than a busy signal. The spinner match covers the full moon-phase glyph set rather than one frame, but it remains locale- and emoji-font-sensitive because Kimi exposes no stable ASCII busy token. diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index 0c47c0b9ab1..c941034eda8 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -1478,11 +1478,16 @@ if [ "$HARNESS" = kimi ]; then exit 1 fi KIMI_POINTER="Read the brief at $BRIEF_REAL and follow it exactly." - if ! spawn_send_literal "$T" "$KIMI_POINTER"; then - kimi_spawn_fail "kimi brief pointer could not be typed" + KIMI_SUBMIT_RETRIES=${FM_KIMI_SUBMIT_RETRIES:-3} + KIMI_SUBMIT_SLEEP=${FM_KIMI_SUBMIT_SLEEP:-${FM_KIMI_POLL_INTERVAL:-0.5}} + KIMI_SUBMIT_SETTLE=${FM_KIMI_SUBMIT_SETTLE:-0} + KIMI_SUBMIT_VERDICT=$(fm_backend_send_text_submit \ + "$BACKEND" "$T" "$KIMI_POINTER" "$KIMI_SUBMIT_RETRIES" \ + "$KIMI_SUBMIT_SLEEP" "$KIMI_SUBMIT_SETTLE" "$W") || { + kimi_spawn_fail "kimi brief pointer could not be submitted" exit 1 - fi - if ! spawn_send_key "$T" Enter; then + } + if [ "$KIMI_SUBMIT_VERDICT" = send-failed ]; then kimi_spawn_fail "kimi brief pointer could not be submitted" exit 1 fi diff --git a/bin/fm-tmux-lib.sh b/bin/fm-tmux-lib.sh index 350ff885f17..d06d8a3b303 100755 --- a/bin/fm-tmux-lib.sh +++ b/bin/fm-tmux-lib.sh @@ -67,12 +67,13 @@ # signature separate from the shared default because that shape is not generic # enough to classify arbitrary harness output safely. # Kimi's anchored moon-phase spinner is separate because bare moon glyphs in -# ordinary output must not classify another harness as busy. Leading whitespace -# and whitespace around the middot are optional because prose examples preserve -# the spinner's shape rather than defining byte-exact spacing. The line end stays -# unanchored because rotating tip text follows but is not required. The idle -# status bar's lowercase `thinking` label and independently rotating tip text are -# not busy signals on their own. +# ordinary output must not classify another harness as busy. Leading whitespace is +# OPTIONAL; whitespace on both sides of the separator is REQUIRED because every +# captured spinner row had it. A zero-whitespace form has NEVER been observed and +# is deliberately not matched. The line end is intentionally unanchored because +# rotating tip text follows and is not required to be present. The idle status +# bar's lowercase `thinking` label and independently rotating tip text are not +# busy signals on their own. # The full moon-phase set remains locale- and emoji-font-sensitive because Kimi # exposes no stable ASCII busy token. FM_TMUX_BUSY_REGEX_DEFAULT='esc (to )?interrupt|Working\.\.\.|Ctrl\+c:cancel' @@ -81,7 +82,7 @@ FM_TMUX_CODEX_BUSY_REGEX_DEFAULT='esc to interrupt' FM_TMUX_OPENCODE_BUSY_REGEX_DEFAULT='esc interrupt' FM_TMUX_PI_BUSY_REGEX_DEFAULT='Working\.\.\.' FM_TMUX_GROK_BUSY_REGEX_DEFAULT='Ctrl\+c:cancel' -FM_TMUX_KIMI_BUSY_REGEX_DEFAULT='^[[:space:]]*(🌑|🌒|🌓|🌔|🌕|🌖|🌗|🌘)[[:space:]]*·[[:space:]]*' +FM_TMUX_KIMI_BUSY_REGEX_DEFAULT='^[[:space:]]*(🌑|🌒|🌓|🌔|🌕|🌖|🌗|🌘)[[:space:]]+·[[:space:]]+' fm_busy_lines_match() { # [harness] local harness=${1:-} lines regex diff --git a/tests/fm-kimi-harness.test.sh b/tests/fm-kimi-harness.test.sh index 93ac12b4853..4075247a1ee 100755 --- a/tests/fm-kimi-harness.test.sh +++ b/tests/fm-kimi-harness.test.sh @@ -78,7 +78,12 @@ case "${1:-}" in ;; pointer-typed) if [ "${FM_FAKE_KIMI_DELIVERY:-yes}" = yes ]; then - printf 'delivered\n' > "$FM_FAKE_KIMI_STATE" + if [ "${FM_FAKE_KIMI_SWALLOW_FIRST:-no}" = yes ] \ + && [ ! -f "$FM_FAKE_KIMI_SWALLOWED" ]; then + : > "$FM_FAKE_KIMI_SWALLOWED" + else + printf 'delivered\n' > "$FM_FAKE_KIMI_STATE" + fi else printf 'ready\n' > "$FM_FAKE_KIMI_STATE" fi @@ -143,6 +148,8 @@ run_spawn() { FM_FAKE_LAUNCH_LOG="$case_dir/launch.log" \ FM_FAKE_POINTER_LOG="$case_dir/pointer.log" \ FM_FAKE_KIMI_STATE="$case_dir/kimi.state" \ + FM_FAKE_KIMI_SWALLOWED="$case_dir/kimi.swallowed" \ + FM_FAKE_KIMI_SWALLOW_FIRST="${FM_FAKE_KIMI_SWALLOW_FIRST:-no}" \ FM_FAKE_TMUX_CALL_LOG="$case_dir/tmux-calls.log" \ FM_FAKE_BRIEF_REAL="$(cd "$home/data/$id" && pwd -P)/brief.md" \ FM_KIMI_READY_POLLS=2 FM_KIMI_DELIVERY_POLLS=2 FM_KIMI_POLL_INTERVAL=0 \ @@ -161,7 +168,8 @@ test_kimi_launch_then_send_is_verified() { id=kimi-success-z1 rec=$(make_spawn_case success "$id") read_spawn_record "$rec" - out=$(run_spawn "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id" \ + out=$(FM_FAKE_KIMI_SWALLOW_FIRST=yes run_spawn \ + "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id" \ --model kimi-code/k3 --effort high) rc=$? expect_code 0 "$rc" "verified kimi launch-then-send should succeed" @@ -327,10 +335,11 @@ test_kimi_busy_signature_is_scoped_to_spinner_lines() { esac } # These fixtures reproduce the observed spinner shape rather than byte-exact - # transcriptions. Leading and separator whitespace are deliberately varied. + # transcriptions. Leading whitespace is deliberately varied; separator whitespace + # follows the captured contract. printf ' 🌑 · Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"\n│ > │\n' > "$capture" fm_pane_is_busy fake kimi || fail "the first real Kimi spinner shape was not recognized as busy" - printf ' 🌗·Tip: /plugins: manage plugins ...\n│ > │\n' > "$capture" + printf ' 🌗 · Tip: /plugins: manage plugins ...\n│ > │\n' > "$capture" fm_pane_is_busy fake kimi || fail "the tool-execution Kimi spinner shape was not recognized as busy" printf 'ordinary response ending with 🌕\n│ > │\n' > "$capture" if fm_pane_is_busy fake kimi; then @@ -368,7 +377,7 @@ test_kimi_busy_signature_is_scoped_to_spinner_lines() { } test_watcher_scopes_moon_spinner_to_recorded_kimi_task() ( - local state="$TMP_ROOT/watch-state" busy_capture=' 🌑· Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"' + local state="$TMP_ROOT/watch-state" busy_capture=' 🌑 · Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"' mkdir -p "$state" printf 'window=fake\nharness=kimi\n' > "$state/kimi-watch.meta" unset FM_BUSY_REGEX From 2e0c1a900900b667ea9f4f6f3fbb6de5a24a59a4 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Sat, 25 Jul 2026 21:49:59 -0700 Subject: [PATCH 16/52] feat(bin): add guarded Kimi turn-end wake (#1059) * Add guarded Kimi turn-end hook * no-mistakes(review): Require jq before installing Kimi turn-end hook * no-mistakes(review): Expose jq inside isolated Kimi test fixtures * no-mistakes(review): Preserve Kimi config boundaries during hook removal * no-mistakes(review): Document Kimi removal newline safeguard * no-mistakes(document): Document Kimi shared-home preservation --- .agents/skills/harness-adapters/SKILL.md | 10 +- AGENTS.md | 1 + bin/fm-spawn.sh | 35 +++- docs/configuration.md | 6 +- docs/turnend-guard.md | 11 +- tests/fm-kimi-harness.test.sh | 249 ++++++++++++++++++++++- 6 files changed, 291 insertions(+), 21 deletions(-) diff --git a/.agents/skills/harness-adapters/SKILL.md b/.agents/skills/harness-adapters/SKILL.md index 9da5ed25f52..6cf41d8bff8 100644 --- a/.agents/skills/harness-adapters/SKILL.md +++ b/.agents/skills/harness-adapters/SKILL.md @@ -53,7 +53,7 @@ Use that value for interrupt, exit, resume, and skill-invocation facts. The primary integrations for `claude`, `codex`, `opencode`, `pi`, and `grok` have empirically validated hook paths for the "no turn ends blind" guard. `claude` and `codex` block directly through Stop hooks that preserve exit status 2 and stderr from `bin/fm-turnend-guard.sh`. `opencode`, `pi`, and `grok` expose passive lifecycle callbacks for this purpose, so their tracked primary adapters force one bounded follow-up or resume when the shared predicate blocks. -Kimi is outside the current turn-end integration scope; `docs/turnend-guard.md` owns the global-configuration boundary. +Kimi is outside the primary turn-end guard scope, while `docs/turnend-guard.md` owns its separate guarded global hook for crew wake signals. The exact hook files, commands, scoping rules, and fail-open tradeoffs are owned by `docs/turnend-guard.md`. `docs/verification/supervision.md` "Turn-end guard" owns active validation evidence. When changing any primary turn-end hook, validate the real harness behavior in a scratch project or throwaway home before trusting it, then update that doc and the relevant concise fact below. @@ -378,6 +378,8 @@ Kimi's footer tip rotates independently and can display `ctrl+c: cancel` while c The idle status bar can contain lowercase `thinking`, which is the model's effort label rather than a busy signal. The spinner match covers the full moon-phase glyph set rather than one frame, but it remains locale- and emoji-font-sensitive because Kimi exposes no stable ASCII busy token. -[`docs/turnend-guard.md`](../../../docs/turnend-guard.md) owns Kimi's verified global hook surface, approval boundary, and absence from the enabled integrations. -The current adapter falls back to idle detection. -That fallback is the weakest idle detection of any supported adapter because Kimi has no stable ASCII busy token, so turn completion can only be inferred from the fragile moon spinner disappearing and the pane becoming stable. +[`docs/turnend-guard.md`](../../../docs/turnend-guard.md) owns Kimi's verified global hook surface and captain-approved crew wake integration. +`fm-spawn.sh` installs one marker-delimited Firstmate entry in `$HOME/.kimi-code/config.toml`, one silent always-zero hook script, and one private token registry under `$HOME/.kimi-code/fm-turn-end.d/`. +Each Kimi crew worktree receives a gitignored `.fm-kimi-turnend` token pointer, and the global hook touches that task's `state/<id>.turn-ended` only when the Stop payload's `cwd`, pointer, and registry entry all agree. +A guarded silent hook cannot be verified from absence of effect, so prove invocation with an unguarded probe before concluding that the hook did not fire. +The guarded turn-end signal supplements the pane busy signature, whose locale- and emoji-font-sensitive limits still apply while a turn is running. diff --git a/AGENTS.md b/AGENTS.md index 063207215a6..382413e8214 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -87,6 +87,7 @@ state/ volatile runtime signals; gitignored <id>.status appended by crewmates: "<state>: <note>" wake-event lines, not current-state truth <id>.turn-ended touched by turn-end hooks <id>.grok-turnend-token firstmate-owned grok hook registry token for the task; removed by teardown + <id>.kimi-turnend-token firstmate-owned Kimi hook registry token for the task; removed by teardown <id>.meta written by fm-spawn: window=, worktree=, project=, harness=, model=, effort=, kind=, mode=, yolo=, tasktmp=; kind=secondmate also records home= and projects=; a non-default runtime backend records further backend-specific fields (docs/configuration.md "Runtime backend"; bin/fm-backend.sh, section 8); fm-pr-check, including through fm-pr-merge, records one canonical pr= and the forge's pr_head= when available (GitHub pull requests and GitLab merge requests; docs/gitlab-merge-watch.md); fm-x-link appends x_request=, x_request_ts=, x_followups=, and optional x_platform=/x_reply_max_chars= for an X-mode-originated task (section 14) <id>.herdr-presentation quarantinable attempt and restart-binding journal for Herdr's optional visual projection; never task or endpoint authority; see docs/herdr-backend.md "Optional presentation spaces" <id>.check.sh authenticated slow poll; the watcher dispatches validated PR data and the byte-identified X shim through trusted repository scripts, runs registered custom checks from hash-validated private snapshots, and rejects every other state check without execution diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index c941034eda8..00a2262ef4d 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -102,7 +102,8 @@ # __PIWATCH__ absolute path to .pi/extensions/fm-primary-pi-watch.ts in a pi secondmate home # __OPINPUT__ absolute path to the canonical operational-input encoder # Verified per-harness turn-end hooks are installed automatically where enabled; some live outside the worktree. -# Kimi has no enabled hook because its only verified Stop-hook configuration is global. +# Kimi uses one surgically installed Firstmate region in $HOME/.kimi-code/config.toml, +# a firstmate-owned global hook and registry, and a gitignored per-task pointer. # grok uses a firstmate-owned global hook under ${GROK_HOME:-$HOME/.grok}/hooks # plus a gitignored .fm-grok-turnend worktree pointer and a state token. # On success prints: spawned <id> harness=<name> kind=<ship|scout|secondmate> mode=<mode> yolo=<on|off> window=<backend-target> worktree=<path> @@ -450,8 +451,8 @@ launch_template() { grok) printf '%s' 'grok --always-approve __MODELFLAG____EFFORTFLAG__"$(__OPINPUT__ encode launch-brief < __BRIEF__)"' ;; # Kimi Code rejects a positional prompt, so it launches bare and receives # only an absolute brief pointer after the TUI readiness gate below. - # No turn-end placeholder belongs here because its only verified Stop hook - # is global configuration and is not enabled without captain approval. + # Its turn-end signal is a globally configured Stop hook plus a guarded + # per-task worktree token, so no launch placeholder belongs here. kimi) printf '%s' '__KIMIBIN__ __MODELFLAG__--auto' ;; *) return 1 ;; esac @@ -615,6 +616,12 @@ case "$LAUNCH" in *__KIMIBIN__*) KIMI_BIN=$(resolve_kimi_binary) || exit 1 LAUNCH=${LAUNCH//__KIMIBIN__/$(shell_quote "$KIMI_BIN")} + if [ "$KIND" != secondmate ]; then + "$FM_ROOT/bin/fm-kimi-turnend-hook.sh" install || { + echo "error: refusing Kimi spawn because the global turn-end hook could not be installed safely" >&2 + exit 1 + } + fi ;; esac @@ -1278,12 +1285,11 @@ mkdir -p "$TASK_TMP/gotmp" # Per-harness turn-end hook where enabled: a file that touches # state/<id>.turn-ended when the agent finishes a turn. Worktree-resident hooks -# are kept out of git's view so they never block teardown's dirty check or leak -# into a commit. Kimi has no path because its global-only hook is not enabled. +# and token pointers stay out of git's view so they never block teardown's dirty +# check or leak into a commit. mkdir -p "$STATE" STATE_REAL=$(cd "$STATE" && pwd -P) -TURNEND= -[ "$HARNESS" = kimi ] || TURNEND="$STATE_REAL/$ID.turn-ended" +TURNEND="$STATE_REAL/$ID.turn-ended" exclude_path() { local rel=$1 EXCL EXCL=$(git -C "$WT" rev-parse --git-path info/exclude 2>/dev/null || true) @@ -1378,6 +1384,21 @@ EOF printf 'token=%s\n' "${auth_file##*/}" > "$WT/.fm-grok-turnend" exclude_path '.fm-grok-turnend' ;; + kimi*) + # Kimi's Stop hook is global, but it is inert unless cwd contains this + # task's token pointer and the token resolves through Firstmate's private + # registry. The installer above owns the format-preserving config edit and + # the always-zero, silent hook script. + KIMI_AUTH_DIR="$HOME/.kimi-code/fm-turn-end.d" + old_umask=$(umask) + umask 077 + auth_file=$(mktemp "$KIMI_AUTH_DIR/fm.XXXXXXXXXXXX") + umask "$old_umask" + printf '%s\n' "$TURNEND" > "$auth_file" + printf '%s\n' "${auth_file##*/}" > "$STATE/$ID.kimi-turnend-token" + printf 'token=%s\n' "${auth_file##*/}" > "$WT/.fm-kimi-turnend" + exclude_path '.fm-kimi-turnend' + ;; esac fi diff --git a/docs/configuration.md b/docs/configuration.md index 8843b4740b0..896b98ae31a 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -179,7 +179,7 @@ New harnesses get verified through a supervised trial task before joining the se The verified adapter knowledge - busy signatures, interrupt and exit commands, skill-invocation syntax, and per-harness quirks - lives in [`.agents/skills/harness-adapters/SKILL.md`](../.agents/skills/harness-adapters/SKILL.md). Launch mechanics, including the verified command templates, live in [`bin/fm-spawn.sh`](../bin/fm-spawn.sh). Enabled primary-session turn-end guard integrations are tracked as repo-level hook files and documented in [`docs/turnend-guard.md`](turnend-guard.md). -Kimi is outside the enabled integration scope; [`docs/turnend-guard.md`](turnend-guard.md#compatibility-limits) owns its verified hook boundary. +Kimi remains outside the primary turn-end guard integrations; [`docs/turnend-guard.md`](turnend-guard.md#compatibility-limits) owns its separate captain-approved crew wake hook. Primary-session watcher wake protocols are rendered at session start by [`bin/fm-supervision-instructions.sh`](../bin/fm-supervision-instructions.sh) from [`docs/supervision-protocols/`](supervision-protocols/). Claude's Stop `asyncRewake` hook owns tokenless re-arm cycles, Grok uses background-notify cycles, Codex uses bounded foreground checkpoints, Pi uses its two tracked primary extensions, and OpenCode uses its TUI plugin. `config/crew-harness` is a local, gitignored file containing one adapter name for crewmate and scout launches. @@ -196,6 +196,10 @@ The inherited-local-material contract is owned by [`secondmate-provisioning`](.. Those inherited values are defaults and rules only; `fm-spawn` still permits a consciously chosen explicit runtime outside the config. `config/secondmate-harness` is not inherited because secondmates do not launch secondmates. For grok, `fm-spawn.sh` installs one firstmate-owned global turn-end hook under `$GROK_HOME/hooks/`, or `~/.grok/hooks/` when `GROK_HOME` is unset, and drops a per-task `.fm-grok-turnend` pointer in the worktree, with teardown removing the task token and pointer. +For Kimi crews, `fm-spawn.sh` runs `fm-kimi-turnend-hook.sh install`, drops a per-task `.fm-kimi-turnend` pointer in the worktree, and records the matching private registry token for teardown. +Kimi continues to use the captain's normal Kimi home, including the existing config, skills, and memory; Firstmate does not create an isolated Kimi home. +The Kimi installer requires an existing regular non-symlink `~/.kimi-code/config.toml`, `python3` with `tomllib`, and `jq`; it validates but never serializes the captain's TOML and refuses before writing when the config is missing, malformed, or surprising or when either tool requirement is unavailable. +Its `remove` action excises only the marker-delimited Firstmate region and removes Firstmate's hook files. For Pi secondmate launches, `fm-spawn.sh` starts Pi with `-e` pointed at the secondmate home's own tracked `.pi/extensions/fm-primary-pi-watch.ts` and `.pi/extensions/fm-primary-turnend-guard.ts`, both already present from the secondmate home's git worktree. ## Crew dispatch profiles (config/crew-dispatch.json) diff --git a/docs/turnend-guard.md b/docs/turnend-guard.md index 2fe72b0d2c1..5589ea23852 100644 --- a/docs/turnend-guard.md +++ b/docs/turnend-guard.md @@ -73,13 +73,18 @@ That warning uses `bin/fm-supervision-instructions.sh --repair-line`, so it alwa - Claude and Codex block directly, while OpenCode, Pi, and Grok use bounded passive follow-ups. - OpenCode headless mode and untrusted Grok project hooks remain fail-open at the host boundary. - Kimi Code CLI 0.29.1 exposes only global `[[hooks]]` configuration in `~/.kimi-code/config.toml`, including a `Stop` event with snake_case payload fields `hook_event_name`, `session_id`, `cwd`, and `stop_hook_active`. -- Kimi has no project-level hook configuration, so Firstmate does not modify that global file without captain approval, installs no Kimi hook, and writes no Kimi turn-end marker. -- Missing `jq` or unreadable hook input remains fail-open. +- Kimi has no project-level hook configuration and remains outside the primary guard integrations above. +- Captain-approved Kimi crew wake support uses `bin/fm-kimi-turnend-hook.sh` to edit only one marker-delimited Firstmate region in that global config and install a silent always-zero hook. +- The hook remains inert unless the payload `cwd` contains a per-task token pointer that resolves through Firstmate's private registry to one `state/<id>.turn-ended` marker. +- Installation refuses before writing unless `python3` with `tomllib` and `jq` are available. +- If `jq` is removed after installation, the hook remains silent and exits 0, turn-end wakes stop, and Kimi crews fall back to idle detection. +- Unreadable hook input remains fail-open. - No harness adapter uses a shell ampersand to manufacture supervision. ## Regression coverage -`tests/fm-turnend-guard.test.sh` covers the predicate, main and secondmate primary scope, child-worktree exclusion, `FM_HOME` and `FM_STATE_OVERRIDE` precedence, the cooperative `--claude` claim wait, epoch allow, re-block budget, Pi logical-run latching, missing-`jq` behavior, all five registrations, and Grok resume permission and recursion safety. +`tests/fm-turnend-guard.test.sh` covers the predicate, main and secondmate primary scope, child-worktree exclusion, `FM_HOME` and `FM_STATE_OVERRIDE` precedence, the cooperative `--claude` claim wait, epoch allow, re-block budget, Pi logical-run latching, missing-`jq` behavior, all five primary registrations, and Grok resume permission and recursion safety. +`tests/fm-kimi-harness.test.sh` covers the separate Kimi crew hook's format preservation, idempotence, refusal cases, token guard, spawn registration, and teardown cleanup. `tests/fm-supervision-instructions.test.sh` covers recovery-line ownership. `FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh` is the opt-in isolated Pi path. [`verification/supervision.md`](verification/supervision.md#turn-end-guard) records the active cross-harness empirical evidence, including the 2026-07-24 Claude `asyncRewake` revalidation. diff --git a/tests/fm-kimi-harness.test.sh b/tests/fm-kimi-harness.test.sh index 4075247a1ee..6f3ba6745b2 100755 --- a/tests/fm-kimi-harness.test.sh +++ b/tests/fm-kimi-harness.test.sh @@ -6,8 +6,13 @@ set -u . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" SPAWN="$ROOT/bin/fm-spawn.sh" +TEARDOWN="$ROOT/bin/fm-teardown.sh" +KIMI_HOOK="$ROOT/bin/fm-kimi-turnend-hook.sh" TMP_ROOT=$(fm_test_tmproot fm-kimi-harness) -BASE_PATH=${FM_TEST_BASE_PATH:-/usr/bin:/bin:/usr/sbin:/sbin} +PYTHON_BIN=$(command -v python3) || fail "test needs python3" +PYTHON_BIN_DIR=$(dirname "$PYTHON_BIN") +JQ_BIN=$(command -v jq) || fail "test needs jq" +BASE_PATH=${FM_TEST_BASE_PATH:-$PYTHON_BIN_DIR:/usr/bin:/bin:/usr/sbin:/sbin} assert_source_line() { local line=$1 @@ -114,8 +119,9 @@ esac exit 0 SH chmod +x "$fakebin/tmux" - fm_fake_exit0 "$fakebin" treehouse + fm_fake_exit0 "$fakebin" treehouse gh-axi gh fm_fake_exit0 "$fakebin" kimi + ln -s "$JQ_BIN" "$fakebin/jq" printf '%s\n' "$fakebin" } @@ -126,7 +132,8 @@ make_spawn_case() { proj="$case_dir/project" wt="$case_dir/wt" fakebin=$(make_spawn_fakebin "$case_dir/fake") - mkdir -p "$home/data/$id" "$home/projects" "$home/state" "$home/config" + mkdir -p "$home/data/$id" "$home/projects" "$home/state" "$home/config" "$home/.kimi-code" + printf '# Kimi test config\ndefault_model = "test"\n' > "$home/.kimi-code/config.toml" printf 'brief for kimi\n' > "$home/data/$id/brief.md" printf 'kimi\n' > "$home/config/crew-harness" fm_git_worktree "$proj" "$wt" "wt-$name" @@ -179,7 +186,7 @@ test_kimi_launch_then_send_is_verified() { [ "$launch" = "'$FAKEBIN_DIR/kimi' --model 'kimi-code/k3' --auto" ] \ || fail "kimi launch did not use the absolute binary, model, and --auto only: $launch" assert_not_contains "$launch" "--effort" "kimi launch emitted a nonexistent effort flag" - assert_not_contains "$launch" "turn-ended" "kimi launch implied a turn-end marker" + assert_not_contains "$launch" "turn-ended" "kimi launch embedded a turn-end path" assert_not_contains "$launch" "__TURNEND__" "kimi launch retained a turn-end placeholder" brief_real="$(cd "$HOME_DIR/data/$id" && pwd -P)/brief.md" @@ -189,8 +196,231 @@ test_kimi_launch_then_send_is_verified() { meta="$HOME_DIR/state/$id.meta" assert_grep 'model=kimi-code/k3' "$meta" "kimi meta lost the requested model" assert_grep 'effort=high' "$meta" "kimi meta did not retain the unsupported effort axis" - assert_absent "$HOME_DIR/.kimi-code/config.toml" "kimi spawn wrote a global config file" - pass "fm-spawn: kimi launches bare, waits for readiness, sends an absolute brief pointer, and confirms delivery" + assert_grep 'BEGIN FIRSTMATE KIMI TURN-END HOOK' "$HOME_DIR/.kimi-code/config.toml" \ + "kimi spawn did not install its guarded global hook region" + assert_grep 'token=' "$WT_DIR/.fm-kimi-turnend" "kimi spawn did not write its token pointer" + assert_present "$HOME_DIR/state/$id.kimi-turnend-token" "kimi spawn did not record its token" + pass "fm-spawn: kimi launches, delivers its brief, and registers a guarded turn-end token" +} + +test_kimi_hook_install_is_surgical_idempotent_and_removable() { + local home config original once stripped count + home="$TMP_ROOT/config-surgery" + config="$home/.kimi-code/config.toml" + original="$home/original.toml" + once="$home/once.toml" + stripped="$home/stripped.toml" + mkdir -p "$home/.kimi-code" + cat > "$config" <<'EOF' +# Captain's leading comment stays exactly here. + +[ui] +theme = "night" # inline comment +show_usage = true + +# Foreign hook with intentionally unusual key ordering. +[[hooks]] +timeout=17 +command = "printf foreign" +matcher="" +event = "Stop" + +[providers.example] +model = "some/model" +# Final comment and blank line follow. + +EOF + cp "$config" "$original" + + HOME="$home" "$KIMI_HOOK" install || fail "Kimi hook install refused a realistic config" + cp "$config" "$once" + HOME="$home" "$KIMI_HOOK" install || fail "second Kimi hook install failed" + cmp -s "$once" "$config" || fail "second Kimi hook install changed config bytes" + count=$(grep -c '^# BEGIN FIRSTMATE KIMI TURN-END HOOK' "$config") + [ "$count" -eq 1 ] || fail "idempotent install left $count Firstmate regions" + + HOME="$home" "$KIMI_HOOK" remove || fail "Kimi hook removal failed" + cp "$config" "$stripped" + cmp -s "$original" "$stripped" \ + || fail "config with the Firstmate region excised was not byte-identical to the original" + assert_absent "$home/.kimi-code/fm-turn-end.sh" "removal left the Firstmate hook script" + assert_absent "$home/.kimi-code/fm-turn-end.d" "removal left the Firstmate registry" + pass "Kimi hook install is idempotent and removal restores every foreign config byte" +} + +test_kimi_hook_remove_preserves_owned_newline_boundary() { + local appended config expected home original + home="$TMP_ROOT/config-owned-newline" + config="$home/.kimi-code/config.toml" + original="$home/original.toml" + expected="$home/expected.toml" + appended="$home/appended.toml" + mkdir -p "$home/.kimi-code" + printf 'default_model = "test"' > "$config" + cp "$config" "$original" + + HOME="$home" "$KIMI_HOOK" install || fail "Kimi hook install refused config without a final newline" + HOME="$home" "$KIMI_HOOK" remove || fail "Kimi hook removal failed without appended config" + cmp -s "$original" "$config" \ + || fail "pristine removal did not restore the absent final newline byte-identically" + + HOME="$home" "$KIMI_HOOK" install || fail "second Kimi hook install refused config without a final newline" + printf '[captain]\nenabled = true\n' > "$appended" + cat "$appended" >> "$config" + HOME="$home" "$KIMI_HOOK" remove || fail "Kimi hook removal joined config appended after its region" + { + cat "$original" + printf '\n' + cat "$appended" + } > "$expected" + cmp -s "$expected" "$config" \ + || fail "removal did not preserve appended captain config on its own line" + "$PYTHON_BIN" - "$config" <<'PY' || fail "config with appended captain TOML did not parse after removal" +import sys +import tomllib + +with open(sys.argv[1], "rb") as stream: + tomllib.load(stream) +PY + pass "Kimi hook removal preserves owned newline boundaries and pristine bytes" +} + +test_kimi_hook_fails_closed_on_missing_malformed_or_partial_config() { + local missing malformed partial out rc + missing="$TMP_ROOT/config-missing" + malformed="$TMP_ROOT/config-malformed" + partial="$TMP_ROOT/config-partial" + mkdir -p "$missing/.kimi-code" "$malformed/.kimi-code" "$partial/.kimi-code" + + rc=0 + out=$(HOME="$missing" "$KIMI_HOOK" install 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "missing Kimi config was accepted" + assert_contains "$out" "Kimi config is missing" "missing config refusal lacked its concrete reason" + assert_absent "$missing/.kimi-code/fm-turn-end.sh" "missing config refusal wrote the hook script" + + printf '[broken\n' > "$malformed/.kimi-code/config.toml" + cp "$malformed/.kimi-code/config.toml" "$malformed/before" + rc=0 + out=$(HOME="$malformed" "$KIMI_HOOK" install 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "malformed Kimi config was accepted" + assert_contains "$out" "malformed TOML" "malformed config refusal lacked its concrete reason" + cmp -s "$malformed/before" "$malformed/.kimi-code/config.toml" \ + || fail "malformed config refusal changed config bytes" + assert_absent "$malformed/.kimi-code/fm-turn-end.sh" "malformed config refusal wrote the hook script" + + printf '# BEGIN FIRSTMATE KIMI TURN-END HOOK\n' > "$partial/.kimi-code/config.toml" + cp "$partial/.kimi-code/config.toml" "$partial/before" + rc=0 + out=$(HOME="$partial" "$KIMI_HOOK" install 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "partial Firstmate marker was accepted" + assert_contains "$out" "partial, duplicated, or altered" "partial marker refusal lacked its concrete reason" + cmp -s "$partial/before" "$partial/.kimi-code/config.toml" \ + || fail "partial marker refusal changed config bytes" + pass "Kimi hook install refuses missing, malformed, and surprising config without writing" +} + +test_kimi_hook_install_refuses_without_jq() { + local home config before fakebin out rc + home="$TMP_ROOT/config-no-jq" + config="$home/.kimi-code/config.toml" + before="$home/config-before.toml" + fakebin=$(fm_fakebin "$home/no-jq") + mkdir -p "$home/.kimi-code" + printf '# Captain config\nmodel = "test"\n' > "$config" + cp "$config" "$before" + ln -s "$(command -v bash)" "$fakebin/bash" + ln -s "$(command -v python3)" "$fakebin/python3" + + rc=0 + out=$(HOME="$home" PATH="$fakebin" "$KIMI_HOOK" install 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "Kimi hook install succeeded without jq" + assert_contains "$out" "jq is required" "missing-jq refusal did not name jq" + cmp -s "$before" "$config" || fail "missing-jq refusal changed config bytes" + assert_absent "$home/.kimi-code/fm-turn-end.sh" "missing-jq refusal wrote the hook script" + assert_absent "$home/.kimi-code/fm-turn-end.d" "missing-jq refusal wrote the registry" + pass "Kimi hook install refuses without jq before any config write" +} + +test_kimi_hook_is_silent_and_requires_registered_workspace_token() { + local id rec out rc hook target token no_token snapshot_before snapshot_after fakebin + id=kimi-hook-auth-z6 + rec=$(make_spawn_case hook-auth "$id") + read_spawn_record "$rec" + out=$(run_spawn "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id") + rc=$? + expect_code 0 "$rc" "Kimi spawn should succeed before hook authentication checks" + hook="$HOME_DIR/.kimi-code/fm-turn-end.sh" + target="$HOME_DIR/state/$id.turn-ended" + token=$(sed -n 's/^token=//p' "$WT_DIR/.fm-kimi-turnend") + assert_present "$HOME_DIR/.kimi-code/fm-turn-end.d/$token" "Kimi registry token is missing" + + no_token="$CASE_DIR/no-token-workspace" + mkdir -p "$no_token" + snapshot_before=$(find "$no_token" -mindepth 1 -print) + out=$(printf '{"hook_event_name":"Stop","session_id":"ordinary","cwd":"%s","stop_hook_active":false}\n' "$no_token" \ + | HOME="$HOME_DIR" bash "$hook" 2>&1) + rc=$? + expect_code 0 "$rc" "Kimi hook must never block a tokenless session" + [ -z "$out" ] || fail "Kimi hook printed into a tokenless session: $out" + snapshot_after=$(find "$no_token" -mindepth 1 -print) + [ "$snapshot_before" = "$snapshot_after" ] || fail "Kimi hook wrote inside a tokenless workspace" + assert_absent "$target" "tokenless Kimi hook invocation touched a task marker" + + printf 'token=%s\n' "$token" > "$WT_DIR/.fm-kimi-turnend" + out=$(printf '{"hook_event_name":"Stop","session_id":"crew","cwd":"%s","stop_hook_active":false}\n' "$WT_DIR" \ + | HOME="$HOME_DIR" bash "$hook" 2>&1) + rc=$? + expect_code 0 "$rc" "registered Kimi hook invocation did not exit zero" + [ -z "$out" ] || fail "registered Kimi hook invocation printed output: $out" + assert_present "$target" "registered Kimi hook invocation did not touch the turn-end marker" + + rm "$target" + fakebin=$(fm_fakebin "$CASE_DIR/no-jq") + ln -s "$(command -v bash)" "$fakebin/bash" + out=$(printf '{"hook_event_name":"Stop","session_id":"crew","cwd":"%s","stop_hook_active":false}\n' "$WT_DIR" \ + | HOME="$HOME_DIR" PATH="$fakebin" "$hook" 2>&1) + rc=$? + expect_code 0 "$rc" "Kimi hook without jq must still exit zero" + [ -z "$out" ] || fail "Kimi hook without jq printed output: $out" + assert_absent "$target" "Kimi hook without jq touched the turn-end marker" + pass "Kimi hook stays silent and inert without a Firstmate registry token" +} + +test_kimi_spawn_refuses_unsafe_global_config_before_pane_creation() { + local id rec out rc + id=kimi-config-refuse-z7 + rec=$(make_spawn_case config-refuse "$id") + read_spawn_record "$rec" + printf '[malformed\n' > "$HOME_DIR/.kimi-code/config.toml" + rc=0 + out=$(run_spawn "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id") || rc=$? + [ "$rc" -ne 0 ] || fail "Kimi spawn accepted malformed global config" + assert_contains "$out" "malformed TOML" "Kimi spawn omitted the concrete config refusal" + if grep -Eq '(^| )new-(session|window)( |$)' "$CASE_DIR/tmux-calls.log"; then + fail "unsafe Kimi config refusal created a tmux container or pane" + fi + pass "fm-spawn: unsafe Kimi global config refuses before pane creation" +} + +test_kimi_teardown_removes_pointer_and_registry_token() { + local id rec out rc token + id=kimi-teardown-z8 + rec=$(make_spawn_case teardown "$id") + read_spawn_record "$rec" + out=$(run_spawn "$CASE_DIR" "$HOME_DIR" "$PROJ_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$id") + rc=$? + expect_code 0 "$rc" "Kimi spawn should succeed before teardown" + token=$(sed -n 's/^token=//p' "$WT_DIR/.fm-kimi-turnend") + + HOME="$HOME_DIR" FM_ROOT_OVERRIDE="$ROOT" FM_HOME="$HOME_DIR" \ + FM_STATE_OVERRIDE="$HOME_DIR/state" FM_DATA_OVERRIDE="$HOME_DIR/data" \ + FM_PROJECTS_OVERRIDE="$HOME_DIR/projects" FM_CONFIG_OVERRIDE="$HOME_DIR/config" \ + FM_SPAWN_NO_GUARD=1 PATH="$FAKEBIN_DIR:$BASE_PATH" \ + "$TEARDOWN" "$id" --force >/dev/null 2>&1 || fail "Kimi teardown failed" + assert_absent "$WT_DIR/.fm-kimi-turnend" "Kimi token pointer survived teardown" + assert_absent "$HOME_DIR/.kimi-code/fm-turn-end.d/$token" "Kimi registry token survived teardown" + assert_absent "$HOME_DIR/state/$id.kimi-turnend-token" "Kimi token state survived teardown" + pass "fm-teardown: Kimi task pointer and registry token are removed" } test_kimi_falls_back_to_expanded_home_binary() { @@ -423,7 +653,14 @@ test_kimi_bordered_prompt_needs_no_override() { test_tracked_files_have_no_user_absolute_paths test_existing_launch_templates_are_byte_pinned +test_kimi_hook_install_is_surgical_idempotent_and_removable +test_kimi_hook_remove_preserves_owned_newline_boundary +test_kimi_hook_fails_closed_on_missing_malformed_or_partial_config +test_kimi_hook_install_refuses_without_jq test_kimi_launch_then_send_is_verified +test_kimi_hook_is_silent_and_requires_registered_workspace_token +test_kimi_spawn_refuses_unsafe_global_config_before_pane_creation +test_kimi_teardown_removes_pointer_and_registry_token test_kimi_falls_back_to_expanded_home_binary test_kimi_missing_binary_refuses_before_pane_creation test_kimi_unconfirmed_delivery_fails_loudly From c4a89f46585293529fdda7130ab015baca64edf8 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Sun, 26 Jul 2026 01:08:14 -0700 Subject: [PATCH 17/52] fix(tmux): classify bordered composers across all rows (#1066) * Fix structural tmux composer reading * Verify Calm compatibility with Pi 0.82 * no-mistakes(review): Harden structural composer classification boundaries * no-mistakes(review): Refresh composer and Kimi regression fixtures * no-mistakes(review): Fail closed on unbounded composer edges * no-mistakes(review): Enforce aligned composer geometry safely * no-mistakes(review): Make composer ambiguity locale-safe * no-mistakes(review): Preserve ambiguity through composer submission * no-mistakes(review): Carry composer proof through retries * no-mistakes(document): Document structural tmux composer delivery guarantees * no-mistakes: apply CI fixes --- .agents/skills/harness-adapters/SKILL.md | 19 +- .pi/extensions/fm-calm.ts | 25 +- .../lib/fm-calm-operational-user-layout.ts | 25 +- bin/fm-tmux-lib.sh | 317 ++++++++++++++---- docs/calm-mode-feasibility.md | 28 +- docs/tmux-backend.md | 16 +- docs/verification/runtime-backends.md | 7 +- docs/zellij-backend.md | 2 +- tests/fm-calm-pi-extension.test.sh | 254 ++++---------- tests/fm-composer-ghost.test.sh | 6 +- tests/fm-kimi-harness.test.sh | 50 ++- tests/fm-pending-reply.test.sh | 8 +- tests/fm-tmux-submit-busy.test.sh | 82 ++++- 13 files changed, 490 insertions(+), 349 deletions(-) diff --git a/.agents/skills/harness-adapters/SKILL.md b/.agents/skills/harness-adapters/SKILL.md index 6cf41d8bff8..85cd11c35ce 100644 --- a/.agents/skills/harness-adapters/SKILL.md +++ b/.agents/skills/harness-adapters/SKILL.md @@ -153,7 +153,7 @@ Natural language is acceptable if uncertain. - codex: `$<skill>`, for example `$no-mistakes`; `/<skill>` is claude-only and codex rejects it as "Unrecognized command". - opencode: no separate verified skill invocation beyond normal slash-command behavior; use natural language if the exact skill command is uncertain. - pi: no separate verified skill invocation beyond normal command behavior; use natural language if the exact skill command is uncertain. -- grok: `/<skill>`, for example `/no-mistakes` (same form as claude). Verified end to end: grok discovers the user-level `no-mistakes` skill, `/no-mistakes` invokes it, and grok drives a real `no-mistakes axi run`. Like codex's `$`/`/` popups, typing `/<skill>` opens grok's slash-autocomplete, so a too-fast Enter selects the popup entry instead of sending, and for an argument-taking command (like `/no-mistakes`'s optional task-first argument) that first Enter only expands the popup selection into an argument-hint placeholder rather than submitting - a genuine second Enter is required (see the grok section below for the 2026-07-03 incident and fix). `fm_tmux_submit_core`'s retried Enter (used by `fm-send` on the tmux backend) already handles this correctly by reading the cursor row; the herdr backend needed a dedicated fix (`fm_backend_herdr_composer_state`, docs/herdr-backend.md) because its prior delta-based verification false-positived on that same popup-close content change. +- grok: `/<skill>`, for example `/no-mistakes` (same form as claude). Verified end to end: grok discovers the user-level `no-mistakes` skill, `/no-mistakes` invokes it, and grok drives a real `no-mistakes axi run`. Like codex's `$`/`/` popups, typing `/<skill>` opens grok's slash-autocomplete, so a too-fast Enter selects the popup entry instead of sending, and for an argument-taking command (like `/no-mistakes`'s optional task-first argument) that first Enter only expands the popup selection into an argument-hint placeholder rather than submitting - a genuine second Enter is required (see the grok section below for the 2026-07-03 incident and fix). `fm_tmux_submit_core`'s retried Enter (used by `fm-send` on the tmux backend) handles this through the structural composer reader; the herdr backend needed a dedicated fix (`fm_backend_herdr_composer_state`, docs/herdr-backend.md) because its prior delta-based verification false-positived on that same popup-close content change. - kimi: `/<skill>`, for example `/no-mistakes`. ## Submission acknowledgement hazards @@ -306,8 +306,8 @@ For Grok's supported reasoning-effort values and omission behavior, see the [lau **Incident (2026-07-03, herdr backend only, grok 0.2.82):** two grok/herdr crewmates were sent `/no-mistakes` via `fm-send`; both left it fully typed but unsubmitted in the composer for minutes (footer still `Enter:send`), and `fm-send` exited 0 with no error. Reproduced live: the herdr adapter's submit-verification at the time treated ANY pane-content change after Enter as "submitted", and the popup-close-with-placeholder-fill described above IS a visible content change even though nothing was actually sent. -The tmux backend was never affected - `fm_tmux_composer_state` reads the actual cursor row, correctly sees the placeholder text as still-pending, and its retry loop already sends the needed second Enter. -Fixed in the Herdr adapter (`fm_backend_herdr_composer_state`, `bin/backends/herdr.sh`) by classifying the composer's own row structurally instead of diffing raw content; see `docs/herdr-backend.md` "Composer and injection safety" for the current boundary and `tests/fm-backend-herdr.test.sh` for regression coverage. +The tmux backend's structural `fm_tmux_composer_state` read sees placeholder-filled text on any content row as still pending, so its retry loop sends the needed second Enter. +The Herdr adapter (`fm_backend_herdr_composer_state`, `bin/backends/herdr.sh`) classifies the composer's own row structurally instead of diffing raw content; see `docs/herdr-backend.md` "Composer and injection safety" for the current boundary and `tests/fm-backend-herdr.test.sh` for regression coverage. Startup dialog: the "Run Grok Build in a project directory?" project picker appears ONLY when grok is launched from a non-project directory (home, Desktop, Downloads, `/tmp`). `fm-spawn` launches inside the treehouse worktree (a git repo root), so the picker never appears and grok treats the worktree as a trusted project automatically - no post-launch keystroke is needed. @@ -320,11 +320,10 @@ Verified live against grok 0.2.93: real input is the bright `38;2;224;222;244` ( This assumes a dark terminal theme, the fleet reality; the SGR-2 signal stays theme-independent. Regression coverage: `tests/fm-composer-ghost.test.sh` (`test_strip_ghost_drops_dark_truecolor_ghost`, `test_dark_truecolor_ghost_only_composer_is_not_pending`) and `tests/fm-backend-herdr.test.sh` (`test_composer_state_grok_dark_truecolor_placeholder_is_empty`, `test_composer_state_grok_bright_truecolor_real_text_is_pending`). -**Residual gap, tmux-only (unfixed):** -in that same pristine placeholder-only state, tmux's own `#{cursor_y}` points at the composer box's BOTTOM BORDER row, one row below the actual text row (the box appears to render one row lower before any real typing starts); once real text is typed the cursor correctly aligns with the text row again. -This is a row-SELECTION quirk, orthogonal to the styling fix above, and affects only the tmux path (herdr uses a structural composer-row scan, not `cursor_y`, so it is unaffected). -A correct fix needs a row-window read near `cursor_y` rather than the single `cursor_y` row. -In practice `fm-spawn` launches grok with the brief as its initial prompt, so a live task's composer is never observed in this pristine pre-typing state - but this is unverified for every path (e.g. a steer sent before grok's first real turn settles) and needs dedicated investigation before relying on it. +**Tmux bottom-border cursor quirk (fixed):** +In a pristine placeholder-only composer, tmux's `#{cursor_y}` can point at the box's bottom border instead of its text row. +The shared tmux reader now locates the complete box structurally and classifies every content row, so the cursor may sit on a content row or the bottom border without changing the result. +The same structural read covers multi-row composers without fixed cursor offsets, while Herdr retains its own structural composer-row scan. Turn-end hook: grok fires a `Stop` hook at every turn boundary, giving firstmate a precise per-turn wake instead of only stale-pane detection. grok loads PROJECT hooks (`<worktree>/.grok/hooks/`, `<worktree>/.claude/settings.local.json`) only after the folder is granted hook-trust in `~/.grok/trusted_folders.toml`, which is not automatic and which firstmate will not establish by editing grok's own managed trust store. @@ -372,7 +371,9 @@ The brief path must be absolute because the brief lives outside the task worktre Observed live spinner captures included optional leading whitespace, a moon-phase glyph, whitespace around `·`, and rotating tip text, with the same shape observed during tool execution. Because every captured spinner row had whitespace on both sides of `·`, the matcher requires that whitespace, deliberately does not match the never-observed zero-whitespace form, and does not require trailing tip text. -The startup input-readiness window is the established cause of Kimi's first-Enter delivery defect: the banner is not the cause, and Grok's cursor-row quirk does not apply. +The startup input-readiness window is the established cause of Kimi's first-Enter delivery defect, while the banner is not the cause. +An early Enter can expand Kimi's composer to multiple content rows, leaving the pointer text on the first row and the cursor on an empty later row, which is the same single-cursor-row reading defect exposed by Grok's bottom-border cursor quirk. +The shared tmux reader now locates the complete bordered composer and treats real text on any content row as positive evidence that submission is still pending. No rendering signal is trustworthy for proving that Kimi will accept input during this window, so delivery retries Enter through the shared submit core and retains the existing postcondition verification rather than relaxing readiness or delivery checks. Kimi's footer tip rotates independently and can display `ctrl+c: cancel` while completely idle, so tip text is never used as its busy signature without the leading moon-plus-middot spinner structure. The idle status bar can contain lowercase `thinking`, which is the model's effort label rather than a busy signal. diff --git a/.pi/extensions/fm-calm.ts b/.pi/extensions/fm-calm.ts index f78c1b5acd9..eb009fd8e3b 100644 --- a/.pi/extensions/fm-calm.ts +++ b/.pi/extensions/fm-calm.ts @@ -1,13 +1,11 @@ // Firstmate's home-persistent Pi transcript presentation toggle. // -// Verified against Pi 0.81.1 and 0.82.0, which expose built-in ToolDefinitions, per-slot +// Compatibility boundary: Pi 0.81.1 and 0.82.0 expose built-in ToolDefinitions, per-slot // renderers, renderShell: "self", session_start replacement reasons, // ExtensionUIContext.setToolsExpanded(), setWorkingVisible(), and -// setHiddenThinkingLabel(). The focused tests pin those assumptions but never reject a -// newer Pi solely for its version. The collapsed-thinking and operational-user -// presentation adapters probe the exact API they patch and degrade independently with a -// diagnostic (see installCalmPresentationAdapter below) if a future Pi removes it; Pi -// still exposes no global renderer for arbitrary built-in or custom rows. +// setHiddenThinkingLabel(). The focused tests pin those assumptions. Version-bounded +// presentation adapters cover collapsed assistant thinking and operational user rows; +// Pi still exposes no global renderer for arbitrary built-in or custom rows. // docs/configuration.md owns the home-local Calm preference contract. import { randomUUID } from "node:crypto"; import { @@ -76,20 +74,9 @@ const extensionFile = fileURLToPath(import.meta.url); const extensionDir = dirname(extensionFile); const root = resolve(extensionDir, "../.."); -// Each presentation adapter probes the exact Pi API it patches. If a future Pi removes -// that API, only the affected adapter degrades; the rest of Calm keeps working. -function installCalmPresentationAdapter(name: string, install: () => void): void { - try { - install(); - } catch (error) { - const reason = error instanceof Error ? error.message : String(error); - console.error(`Firstmate Calm: ${name} presentation adapter unavailable, skipping. ${reason}`); - } -} - export default function (pi: ExtensionAPI) { - installCalmPresentationAdapter("collapsed-thinking", installCalmAssistantLayout); - installCalmPresentationAdapter("operational-user-row", installCalmOperationalUserLayout); + installCalmAssistantLayout(); + installCalmOperationalUserLayout(); let exportRendering = false; let removeTerminalInputHandler: (() => void) | undefined; diff --git a/.pi/extensions/lib/fm-calm-operational-user-layout.ts b/.pi/extensions/lib/fm-calm-operational-user-layout.ts index ca9b0bbcc0a..82c69eda01f 100644 --- a/.pi/extensions/lib/fm-calm-operational-user-layout.ts +++ b/.pi/extensions/lib/fm-calm-operational-user-layout.ts @@ -1,14 +1,13 @@ -// Verified against Pi 0.81.1 and 0.82.0, which add the ordinary-user spacer and row -// together via InteractiveMode.addMessageToChat. This adapter probes that exact method -// and throws if it is missing; fm-calm.ts catches that and skips only this adapter with a -// diagnostic instead of blocking Calm or Pi. It changes only that presentation and never -// message delivery. -import type { UserMessageComponent as PiUserMessageComponent } from "@earendil-works/pi-coding-agent"; -import * as PiCodingAgent from "@earendil-works/pi-coding-agent"; +// Pi 0.81.1 and 0.82.0 add the ordinary-user spacer and row together. +// This version-bounded adapter changes only that presentation and never message delivery. +import { + InteractiveMode, + UserMessageComponent, +} from "@earendil-works/pi-coding-agent"; import { calmPresentationHides } from "./fm-calm-visibility.ts"; import { classifyFirstmateCurrentOperationalText } from "./fm-operational-input.ts"; -type UserMessageConstructorArgs = ConstructorParameters<typeof PiUserMessageComponent>; +type UserMessageConstructorArgs = ConstructorParameters<typeof UserMessageComponent>; type UserMessageLike = { role: string; content: unknown; @@ -19,7 +18,7 @@ type AddMessageOptions = { type InteractiveModePresentation = { chatContainer: { children: unknown[]; - addChild(component: PiUserMessageComponent): void; + addChild(component: UserMessageComponent): void; }; editor: { addToHistory?(text: string): void; @@ -82,20 +81,12 @@ export function installCalmOperationalUserLayout(): void { hidesOperationalInput, isOperationalInput, }; - const InteractiveMode = PiCodingAgent.InteractiveMode; - if (typeof InteractiveMode !== "function") { - throw new Error("Firstmate Calm requires Pi InteractiveMode"); - } const prototype = InteractiveMode.prototype as unknown as InteractiveModePrototype; const originalAddMessageToChat = prototype.addMessageToChat; if (typeof originalAddMessageToChat !== "function") { throw new Error("Firstmate Calm requires Pi InteractiveMode.addMessageToChat"); } - const UserMessageComponent = PiCodingAgent.UserMessageComponent; - if (typeof UserMessageComponent !== "function") { - throw new Error("Firstmate Calm requires Pi UserMessageComponent"); - } class CalmOperationalUserMessageComponent extends UserMessageComponent { private readonly hasLeadingSpacer: boolean; diff --git a/bin/fm-tmux-lib.sh b/bin/fm-tmux-lib.sh index d06d8a3b303..7eddc323f33 100755 --- a/bin/fm-tmux-lib.sh +++ b/bin/fm-tmux-lib.sh @@ -19,15 +19,15 @@ # "suggestion" as dim/faint text inside an otherwise-empty composer. A plain # capture cannot tell it apart from text a human typed, so the old reader saw an # idle pane as holding pending input and the daemon deferred injection / firstmate -# misjudged the pane. The composer reader now captures just the cursor line WITH -# ANSI styling (tmux capture-pane -e) and extracts the real typed content with the -# shared, fleet-wide fm_composer_strip_ghost (bin/fm-composer-lib.sh), which drops -# every de-emphasised run - dim/faint (SGR 2) AND a dark/muted truecolor -# foreground - so ghost/placeholder text never counts as real input. The styled -# capture is consumed internally and parsed into a boolean here; it is NEVER -# surfaced (fm-peek and every human/LLM-facing path stay plain), and only the -# single composer row is captured, so no escape-laden pane bulk is produced. This -# is harness-generic: any harness that de-emphasises placeholder/ghost text +# misjudged the pane. The composer reader now captures the visible pane WITH ANSI +# styling (tmux capture-pane -e), locates a bordered composer structurally, and +# extracts the real typed content from every row with the shared, fleet-wide +# fm_composer_strip_ghost (bin/fm-composer-lib.sh), which drops every +# de-emphasised run - dim/faint (SGR 2) AND a dark/muted truecolor foreground - +# so ghost/placeholder text never counts as real input. The styled capture is +# consumed internally and parsed into a boolean here; it is NEVER surfaced +# (fm-peek and every human/LLM-facing path stay plain). This is harness-generic: +# any harness that de-emphasises placeholder/ghost text # benefits, and the herdr adapter routes through the same owner (task # afk-herdr-false-pending), so the two backends cannot drift. # @@ -117,68 +117,251 @@ fm_busy_lines_match() { # [harness] # so the tmux and herdr adapters cannot drift apart on what counts as ghost text. fm_tmux_strip_ghost() { fm_composer_strip_ghost; } -# fm_tmux_composer_state: classify the cursor/composer line of <target> as -# empty - no pending input (blank, a busy footer, an empty agent composer, or -# only de-emphasised ghost/placeholder text). Safe to inject; also the positive -# acknowledgement that a submit landed. -# pending - real, unsubmitted text on the cursor line (a human mid-typing, or a -# previous injection whose Enter was swallowed). Defer / retry. -# unknown - the pane could not be read (tmux error), OR the cursor line is a -# bare shell prompt (`$`/`%`/`#`/`>`) - a dead shell, not an agent -# composer, so NOT a safe injection target. The caller decides. -# -# The cursor line is captured WITH ANSI styling (capture-pane -e) and bounded to -# the single composer row (-S/-E). The bordered flag (a genuine composer box) is -# read from the PLAIN row (fm_composer_strip_ansi keeps ghost text so the box -# border is still visible), while the real-typed CONTENT is extracted with the -# shared fm_composer_strip_ghost so dim/faint AND dark-truecolor ghost text drops -# out before classification (grok's dark box border drops with the ghost, which -# is why the bordered flag is read from the plain row, not the ghost-stripped -# one). Both are internal only, never surfaced. The detector strips the harness's -# box-drawing composer borders ("│ … │", heavy "┃", or a plain ASCII "|") using -# literal-string substitution (bash 3.2 safe, locale-independent - no \u escapes, -# no multibyte character classes), and delegates the empty/pending/unknown -# decision to the shared owner fm_composer_classify_content -# (bin/fm-composer-lib.sh). The bordered flag is what lets a bordered `│ > │` -# (claude's own idle composer) read empty while a bare, unbordered `$ ` dead-shell -# prompt reads unknown. -fm_tmux_composer_state() { # <target> -> empty|pending|unknown - local target=$1 cy raw plain stripped bordered=0 - cy=$(tmux display-message -p -t "$target" '#{cursor_y}' 2>/dev/null) || { printf 'unknown'; return 0; } - case "$cy" in ''|*[!0-9]*) printf 'unknown'; return 0 ;; esac - raw=$(tmux capture-pane -e -p -t "$target" -S "$cy" -E "$cy" 2>/dev/null) || { printf 'unknown'; return 0; } - # bordered: from the plain row (borders survive an all-ANSI strip). +# fm_tmux_composer_row_state: classify one raw styled candidate row. +# A structural caller forces bordered=1; the compatibility fallback passes 0 +# and may recognize a busy footer. +fm_tmux_composer_row_state() { # <raw-row> [bordered] [allow-busy] -> empty|pending|unknown + local raw=$1 bordered=${2:-0} allow_busy=${3:-1} plain stripped plain=$(printf '%s\n' "$raw" | fm_composer_strip_ansi) plain="${plain#"${plain%%[![:space:]]*}"}" plain="${plain%"${plain##*[![:space:]]}"}" - case "$plain" in - '│'*'│'|'┃'*'┃'|'|'*'|') bordered=1 ;; - esac - # content: from the ghost-stripped row (real typed text only). stripped=$(printf '%s\n' "$raw" | fm_composer_strip_ghost) stripped="${stripped#"${stripped%%[![:space:]]*}"}" stripped="${stripped%"${stripped##*[![:space:]]}"}" case "$stripped" in '│'*'│') stripped=${stripped#│}; stripped=${stripped%│} ;; '┃'*'┃') stripped=${stripped#┃}; stripped=${stripped%┃} ;; + '║'*'║') stripped=${stripped#║}; stripped=${stripped%║} ;; '|'*'|') stripped=${stripped#|}; stripped=${stripped%|} ;; esac stripped="${stripped#"${stripped%%[![:space:]]*}"}" stripped="${stripped%"${stripped##*[![:space:]]}"}" - # A busy footer landing on the cursor line is not pending input (tmux-specific: - # only tmux captures the raw cursor row, which may BE the footer). - if [ -n "$stripped" ] \ + if [ "$allow_busy" = 1 ] && [ -n "$stripped" ] \ && printf '%s' "$stripped" | grep -qiE "${FM_BUSY_REGEX:-$FM_TMUX_BUSY_REGEX_DEFAULT}"; then printf 'empty'; return 0 fi fm_composer_classify_content "$bordered" "$stripped" "${FM_COMPOSER_IDLE_RE:-}" insensitive "$plain" } -# fm_pane_input_pending: 0 (pending) if the cursor line holds real unsubmitted -# text, 1 otherwise. An unreadable pane is treated as NOT pending (fail-safe: -# the same bias the old daemon used — an unknown pane defers nothing here). +fm_tmux_row_has_composer_edge() { # <plain-row> + local row=$1 + row="${row#"${row%%[![:space:]]*}"}" + row="${row%"${row##*[![:space:]]}"}" + case "$row" in + '│'*|*'│'|'┃'*|*'┃'|'║'*|*'║'|'╭'*|*'╭'|'╮'*|*'╮'|\ + '┌'*|*'┌'|'┐'*|*'┐'|'╔'*|*'╔'|'╗'*|*'╗'|'┏'*|*'┏'|'┓'*|*'┓'|\ + '╰'*|*'╰'|'╯'*|*'╯'|'└'*|*'└'|'┘'*|*'┘'|'╚'*|*'╚'|'╝'*|*'╝'|\ + '┗'*|*'┗'|'┛'*|*'┛'|'─'*|*'─'|'━'*|*'━'|'═'*|*'═'|'|'*|*'|'|'+'*|*'+') + return 0 + ;; + esac + return 1 +} + +fm_tmux_composer_geometry_spaces() { # <content-inner> -> spaces + local content=$1 probe + probe="${content#"${content%%[![:space:]]*}"}" + case "$probe" in + '>'*) content=${content/>/ } ;; + '❯'*) content=${content/❯/ } ;; + '›'*) content=${content/›/ } ;; + esac + content=$(printf '%s' "$content" | LC_ALL=C sed 's/[!-~]/ /g') + case "$content" in + *[![:space:]]*) return 1 ;; + esac + printf '%s' "$content" +} + +# fm_tmux_find_composer_box: print the zero-based top and bottom rows of the +# complete bordered box that structurally contains the cursor, plus whether its +# geometry is ambiguous. The cursor may be on any content row or on the bottom +# border; no fixed cursor offset is used. +fm_tmux_find_composer_box() { # <cursor-y> <plain-visible-pane> -> "<top> <bottom> <ambiguous>" + local cy=$1 pane=$2 line indent left_stripped trimmed kind family current_family= + local side_family top_inner top_spaces='' geometry_check=0 geometry_ambiguous=0 + local content_inner content_spaces bottom_inner bottom_spaces + local current_indent= + local row=0 top=-1 valid=0 content_rows=0 unsafe=0 cursor_structural=0 + while IFS= read -r line; do + indent=${line%%[![:space:]]*} + left_stripped="${line#"${line%%[![:space:]]*}"}" + trimmed="${left_stripped%"${left_stripped##*[![:space:]]}"}" + kind= + family= + case "$trimmed" in + '╭'*'╮') kind=top; family=rounded ;; + '┌'*'┐') kind=top; family=light ;; + '╔'*'╗') kind=top; family=double ;; + '┏'*'┓') kind=top; family=heavy ;; + '╰'*'╯') kind=bottom; family=rounded ;; + '└'*'┘') kind=bottom; family=light ;; + '╚'*'╝') kind=bottom; family=double ;; + '┗'*'┛') kind=bottom; family=heavy ;; + '+'*'+') kind=ascii; family=ascii ;; + esac + if [ "$row" -eq "$cy" ] && fm_tmux_row_has_composer_edge "$trimmed"; then + cursor_structural=1 + fi + if [ "$kind" = top ] || { [ "$kind" = ascii ] && [ "$top" -lt 0 ]; }; then + if [ "$top" -ge 0 ] && [ "$top" -lt "$cy" ] && [ "$cy" -le "$row" ]; then + unsafe=1 + fi + top=$row + current_family=$family + current_indent=$indent + valid=1 + content_rows=0 + geometry_ambiguous=0 + geometry_check=1 + top_inner=$trimmed + case "$family" in + rounded) top_inner=${top_inner#╭}; top_inner=${top_inner%╮}; top_spaces=${top_inner//─/ } ;; + light) top_inner=${top_inner#┌}; top_inner=${top_inner%┐}; top_spaces=${top_inner//─/ } ;; + double) top_inner=${top_inner#╔}; top_inner=${top_inner%╗}; top_spaces=${top_inner//═/ } ;; + heavy) top_inner=${top_inner#┏}; top_inner=${top_inner%┓}; top_spaces=${top_inner//━/ } ;; + ascii) top_inner=${top_inner#+}; top_inner=${top_inner%+}; top_spaces=${top_inner//-/ } ;; + esac + case "$top_spaces" in + *[![:space:]]*) geometry_check=0; geometry_ambiguous=1 ;; + esac + elif [ "$kind" = bottom ] || { [ "$kind" = ascii ] && [ "$top" -ge 0 ]; }; then + if [ "$top" -ge 0 ] && [ "$family" = "$current_family" ] \ + && [ "$valid" = 1 ] && [ "$content_rows" -gt 0 ] \ + && [ "$top" -lt "$cy" ] && [ "$cy" -le "$row" ]; then + [ "$indent" = "$current_indent" ] || geometry_ambiguous=1 + if [ "$geometry_check" = 1 ]; then + bottom_inner=$trimmed + case "$family" in + rounded) bottom_inner=${bottom_inner#╰}; bottom_inner=${bottom_inner%╯}; bottom_spaces=${bottom_inner//─/ } ;; + light) bottom_inner=${bottom_inner#└}; bottom_inner=${bottom_inner%┘}; bottom_spaces=${bottom_inner//─/ } ;; + double) bottom_inner=${bottom_inner#╚}; bottom_inner=${bottom_inner%╝}; bottom_spaces=${bottom_inner//═/ } ;; + heavy) bottom_inner=${bottom_inner#┗}; bottom_inner=${bottom_inner%┛}; bottom_spaces=${bottom_inner//━/ } ;; + ascii) bottom_inner=${bottom_inner#+}; bottom_inner=${bottom_inner%+}; bottom_spaces=${bottom_inner//-/ } ;; + esac + [ "$bottom_spaces" = "$top_spaces" ] || geometry_ambiguous=1 + fi + printf '%s %s %s' "$top" "$row" "$geometry_ambiguous" + return 0 + fi + if { [ "$top" -ge 0 ] && [ "$top" -lt "$cy" ] && [ "$cy" -le "$row" ]; } \ + || [ "$row" -eq "$cy" ]; then + unsafe=1 + fi + top=-1 + current_family= + current_indent= + valid=0 + content_rows=0 + elif [ "$top" -ge 0 ]; then + side_family= + case "$trimmed" in + '│'*'│') side_family=single ;; + '┃'*'┃') side_family=heavy ;; + '║'*'║') side_family=double ;; + '|'*'|') side_family=ascii ;; + esac + case "$current_family:$side_family" in + rounded:single|light:single|heavy:heavy|double:double|ascii:ascii) + content_rows=$((content_rows + 1)) + [ "$indent" = "$current_indent" ] || geometry_ambiguous=1 + if [ "$geometry_check" = 1 ]; then + content_inner=$trimmed + case "$side_family" in + single) content_inner=${content_inner#│}; content_inner=${content_inner%│} ;; + heavy) content_inner=${content_inner#┃}; content_inner=${content_inner%┃} ;; + double) content_inner=${content_inner#║}; content_inner=${content_inner%║} ;; + ascii) content_inner=${content_inner#|}; content_inner=${content_inner%|} ;; + esac + if content_spaces=$(fm_tmux_composer_geometry_spaces "$content_inner"); then + [ "$content_spaces" = "$top_spaces" ] || geometry_ambiguous=1 + else + geometry_ambiguous=1 + fi + fi + ;; + *) valid=0 ;; + esac + fi + row=$((row + 1)) + done <<EOF +$pane +EOF + if [ "$top" -ge 0 ] && [ "$top" -lt "$cy" ]; then + unsafe=1 + fi + if [ "$unsafe" = 1 ] || [ "$cursor_structural" = 1 ]; then + return 2 + fi + return 1 +} + +# fm_tmux_composer_state classification contract: +# A row is structural only when its first or last non-whitespace character is a +# composer edge. A complete box has matching border families and bounded top and +# bottom rows. The proof-carrying verdict is empty for proven emptiness, pending +# for proven text in established structure, pending-unproven for text in +# ambiguous structure, and unknown for unreadable state. Consumers that can +# overwrite input or confirm delivery must accept only the exact positive proof +# they require, so unrecognized future verdicts fail safe by default. Empty +# requires positive proof: a genuinely empty composer, an all-empty unambiguous +# box, an empty non-bordered fallback row, or the submit core's proven +# busy-queued Enter conversion. +fm_tmux_composer_state() { # <target> -> empty|pending|pending-unproven|unknown + local target=$1 cy raw pane plain box box_status top bottom geometry_ambiguous + local row row_raw state unknown_seen=0 + cy=$(tmux display-message -p -t "$target" '#{cursor_y}' 2>/dev/null) || { printf 'unknown'; return 0; } + case "$cy" in ''|*[!0-9]*) printf 'unknown'; return 0 ;; esac + pane=$(tmux capture-pane -e -p -t "$target" -S 0 -E - 2>/dev/null) || { printf 'unknown'; return 0; } + plain=$(printf '%s\n' "$pane" | fm_composer_strip_ansi) + if box=$(fm_tmux_find_composer_box "$cy" "$plain"); then + top=${box%% *} + box=${box#* } + bottom=${box%% *} + geometry_ambiguous=${box#* } + row=$((top + 1)) + while [ "$row" -lt "$bottom" ]; do + row_raw=$(printf '%s\n' "$pane" | sed -n "$((row + 1))p") + state=$(fm_tmux_composer_row_state "$row_raw" 1 0) + case "$state" in + pending) + if [ "$geometry_ambiguous" = 1 ]; then + printf 'pending-unproven' + else + printf 'pending' + fi + return 0 + ;; + unknown) unknown_seen=1 ;; + esac + row=$((row + 1)) + done + if [ "$unknown_seen" = 1 ] || [ "$geometry_ambiguous" = 1 ]; then + printf 'unknown' + else + printf 'empty' + fi + return 0 + else + box_status=$? + if [ "$box_status" -eq 2 ]; then + printf 'unknown' + return 0 + fi + fi + raw=$(tmux capture-pane -e -p -t "$target" -S "$cy" -E "$cy" 2>/dev/null) \ + || { printf 'unknown'; return 0; } + if fm_tmux_row_has_composer_edge "$(printf '%s\n' "$raw" | fm_composer_strip_ansi)"; then + printf 'unknown' + return 0 + fi + fm_tmux_composer_row_state "$raw" 0 +} + +# fm_pane_input_pending: 0 when the composer is not proven empty, so pending +# text, ambiguous structure, unreadable state, and future verdicts all defer. fm_pane_input_pending() { # <target> - [ "$(fm_tmux_composer_state "$1")" = pending ] + [ "$(fm_tmux_composer_state "$1")" != empty ] } # fm_pane_is_busy: 0 if the pane's last few non-blank lines show a busy footer @@ -193,32 +376,34 @@ fm_pane_is_busy() { # <target> [harness] # fm_tmux_submit_core: type <text> into <target> ONCE, then submit with Enter, # verifying the composer cleared. Retries Enter ONLY — never retypes, because a # swallowed Enter leaves our text in the composer and retyping would duplicate -# it. Echoes the final verdict on stdout (empty|pending|unknown|send-failed) so callers can -# pick their own success policy: -# - the daemon clears its buffer only on "empty" (strict: an unknown pane must -# not be mistaken for a delivered escalation). -# - fm-send fails only on "pending" (lenient: a positively-confirmed swallow), -# so an unreadable pane never turns a normal steer into a false error. +# it. Echoes the final proof-carrying verdict on stdout so callers can require +# exact `empty` before treating submission as confirmed. # Busy-queued Enter (opencode 1.18.4): the harness accepts Enter while mid-turn # and queues it for after the current turn, but keeps the typed text visible in -# the composer. Once the Enter-retry budget is spent and the composer still -# reads "pending", the submit core falls back to `fm_pane_is_busy`: a busy pane -# means the Enter was accepted and queued (report `empty` so the caller does -# not re-send), while an idle pane keeps `pending` as a genuine swallow. This -# is the only place that exception lives, so the daemon's strict and -# fm-send's lenient success policies both treat a busy-queued Enter as -# delivered. +# the composer. Once the Enter-retry budget is spent and a structurally proven +# composer still reads "pending", the submit core falls back to +# `fm_pane_is_busy`: a busy pane means the Enter was accepted and queued (report +# `empty` so the caller does not re-send), while an idle pane keeps `pending` as +# a genuine swallow. Pending-unproven receives the same Enter retry budget but +# never reaches this exception. fm_tmux_submit_enter_core() { # <target> <retries> <enter-sleep> local target=$1 retries=$2 sleep_s=$3 i=0 state while :; do tmux send-keys -t "$target" Enter 2>/dev/null || true sleep "$sleep_s" state=$(fm_tmux_composer_state "$target") - [ "$state" = pending ] || { printf '%s' "$state"; return 0; } + case "$state" in + pending|pending-unproven) ;; + *) printf '%s' "$state"; return 0 ;; + esac i=$((i + 1)) [ "$i" -lt "$retries" ] || break done - # Retries exhausted, composer still shows pending. + if [ "$state" != pending ]; then + printf '%s' "$state" + return 0 + fi + # Retries exhausted, composer still shows proven pending. # If the pane is busy (agent mid-turn), the harness accepted the Enter # and queued the message for processing when the current turn ends. # Treat it as submitted so the caller does not re-send. diff --git a/docs/calm-mode-feasibility.md b/docs/calm-mode-feasibility.md index b94b6a6aef9..c4a051b9cc6 100644 --- a/docs/calm-mode-feasibility.md +++ b/docs/calm-mode-feasibility.md @@ -9,17 +9,9 @@ A qualifying implementation must auto-load from the trusted project, persist the The governing presentation policy allows genuine original user prompts, genuine user-facing assistant text, and Pi's native working activity. Changing persisted context to remove hidden content, filtering provider context, patching installed harness code, or claiming coverage outside a supported renderer does not satisfy that boundary. -## Compatibility evidence - -[`calm.md`](calm.md#pi-compatibility) owns the current Pi compatibility contract. -Pi 0.81.1 was installed when Calm was first built, and Pi 0.82.0 was the later reverification target. -The inspected Pi CHANGELOG shows no relevant presentation API introduced at either version, so those versions remain verification evidence rather than compatibility bounds. -The exported classes used by the adapters (`AssistantMessageComponent` and `InteractiveMode`) are undocumented internals with no stated version guarantee. -`tests/fm-calm-pi-extension.test.sh` records the installed Pi version as evidence without gating on it and covers both newer synthetic versions and an unavailable adapter seam. - ## Pi 0.81.1 end-to-end reproduction -The Pi version installed at the time was verified on 2026-07-22. +The current installed and regression-supported Pi version was verified on 2026-07-22. ```text $ pi --version @@ -65,8 +57,7 @@ The single-thinking, tool-call-only, tool-result, Calm-off, and `clearOnShrink` PR 927 made Calm persistent and described controlled rows as gapless while retaining a documented unsupported boundary for collapsed-thinking spacing. PR 936 removed the unsafe operational-input reroute and preserved legacy zero-height entries but did not change assistant-message layout. -The fix installs one idempotent presentation adapter, verified on Pi 0.81.1 through 0.82.0, on the exported `AssistantMessageComponent.updateContent` method. -The adapter probes for that exact method and, per the [compatibility contract](calm.md#pi-compatibility), degrades independently with a diagnostic rather than gating on a version number. +The fix installs one idempotent Pi 0.81.1 through 0.82.0 presentation adapter on the exported `AssistantMessageComponent.updateContent` method. Only while Calm is active and Pi has collapsed thinking does the adapter pass a shallow thinking-free presentation copy into Pi's ordinary layout calculation, then retain the original message on the component for invalidation and thinking expansion. The persisted assistant message, provider context, tool execution, export data, and expansion history remain unchanged. Collapsed thinking-only assistant messages now render zero rows, thinking before visible assistant text adds no spacing beyond the text-only baseline, and expanding thinking still renders the original reasoning. @@ -123,8 +114,7 @@ The real Pi viewport moved the unchanged assistant text from row 7 to row 2, ren The leading cause would have been falsified if the row or height remained, the provider lost or duplicated the message, or the persisted role or bytes changed. None occurred. -The fix installs a separate idempotent presentation adapter, verified on Pi 0.81.1 through 0.82.0, on the exported `InteractiveMode.addMessageToChat` method. -The adapter probes for that exact method and, per the [compatibility contract](calm.md#pi-compatibility), degrades independently with a diagnostic rather than gating on a version number. +The fix installs a separate idempotent Pi 0.81.1 through 0.82.0 presentation adapter on the exported `InteractiveMode.addMessageToChat` method. It delegates current recognition to `bin/fm-operational-input.sh`, adds only the evidence-backed bare-U+2063 `Supervisor escalate (` presentation compatibility shape, mounts a `UserMessageComponent` subclass that preserves Pi's stock row plus leading spacer while Calm is off, and returns zero rendered lines while Calm is on. It never intercepts the input event, rewrites the message, changes its role, filters model context, or changes session data. Messages containing an image are left on Pi's ordinary path even when their text equals an operational envelope because Firstmate's authoritative producers are text-only. @@ -158,7 +148,7 @@ Serialized session data and Pi 0.81.1's sidebar tree also retain legacy hidden o The taxonomy was derived from Pi 0.81.1's installed public declarations, documentation, examples, `interactive-mode.js`, and its exported component implementations. The test fixture enumerates every class below through the centralized policy, and the interactive fixture exercises the screenshot classes, current user-role operational input, and legacy synthetic presentation entries. -| Policy class | Pi transcript path | Calm result (verified on Pi 0.81.1 through 0.82.0) | +| Policy class | Pi transcript path | Calm result on Pi 0.81.1 through 0.82.0 | | --- | --- | --- | | `genuine-user-prompt` | `UserMessageComponent` | Visible, including every tested operational near miss. | | `genuine-agent-response` | Assistant text in `AssistantMessageComponent` | Visible. | @@ -177,12 +167,12 @@ The test fixture enumerates every class below through the centralized policy, an | `system-notice` | `showStatus`, `showError`, compaction, retry, and startup warning rows | Unsupported boundary; remains visible. | | `cache-notice` | Non-persisted cache-miss `Text` row | Unsupported boundary; remains visible. | | `project-trust-warning` | Non-persisted startup `Text` row | Unsupported boundary; remains visible. | -| `synthetic-user` | Firstmate extension `sendUserMessage`, terminal-injected input, Firstmate-generated Pi positional brief, or the already non-displayed session-start nudge | Canonically classified text-only operational user messages stay ordinary semantic user messages but render through the zero-height adapter (verified on Pi 0.81.1 through 0.82.0) under Calm; legacy entries stay gaplessly controllable, and the session-start nudge retains its existing non-displayed custom-message path. | +| `synthetic-user` | Firstmate extension `sendUserMessage`, terminal-injected input, Firstmate-generated Pi positional brief, or the already non-displayed session-start nudge | Canonically classified text-only operational user messages stay ordinary semantic user messages but render through the zero-height Pi 0.81.1 through 0.82.0 adapter under Calm; legacy entries stay gaplessly controllable, and the session-start nudge retains its existing non-displayed custom-message path. | | `synthetic-assistant` | No authoritative Firstmate source found | Policy-hidden, but Pi exposes no generic assistant-role renderer. | | `unknown` | Future or unclassified transcript component | Policy-hidden, but no generic renderer exists; never claimed as covered. | The installed extension API has no supported global transcript filter, user-message renderer, assistant-message renderer, chat-container API, or generic custom-tool wrapper. -Pi 0.81.1 through 0.82.0 export `AssistantMessageComponent` and `InteractiveMode`, so Calm uses separate idempotent, API-probed adapters for assistant thinking layout and the complete operational-user transcript row while leaving all message data and non-Calm rendering unchanged; see the [compatibility contract](calm.md#pi-compatibility) for how a future Pi lacking one of those exports is handled. +Pi 0.81.1 through 0.82.0 export `AssistantMessageComponent` and `InteractiveMode`, so Calm uses separate version-bounded, idempotent adapters for assistant thinking layout and the complete operational-user transcript row while leaving all message data and non-Calm rendering unchanged. General component replacement, ANSI cursor erasure, provider-context mutation, and installed-file patching remain rejected as unsupported or preservation-breaking workarounds. ## Cross-harness verification record @@ -207,7 +197,7 @@ grok 0.2.106 (bde89716f679) | Claude Code 2.1.218 | Not feasible through the inspected supported project surface. | Project hooks can observe lifecycle and tool events, while the plugin CLI packages supported components; neither inspected surface exposes a transcript-row renderer or transcript-wide redraw API. | | Codex CLI 0.144.6 | Not feasible through the inspected supported project surface. | The tracked hooks expose session, pre-tool, and stop handling, while the plugin and feature inventories expose no TUI tool-row renderer or transcript redraw control. | | OpenCode 1.17.18 | Not feasible without violating the preservation boundary. | Plugins expose events and tool execution hooks, not a built-in transcript-row renderer; same-name tool replacement changes execution rather than presentation alone. | -| Pi (verified 0.81.1 through 0.82.0) | Partially feasible with two API-probed exported-class adapters. | Public APIs control working visibility, collapsed labels, known tool slots, custom entries, and expansion redraws; exported assistant and interactive-mode classes provide the collapsed-thinking and operational-user layout boundaries, gated on the exact method's presence rather than a version number, while generic user, tool, and status filtering remains unavailable. | +| Pi 0.81.1 through 0.82.0 | Partially feasible with two version-bounded exported-class adapters. | Public APIs control working visibility, collapsed labels, known tool slots, custom entries, and expansion redraws; exported assistant and interactive-mode classes provide the version-pinned collapsed-thinking and operational-user layout boundaries, while generic user, tool, and status filtering remains unavailable. | | Grok CLI 0.2.106 | Not feasible through the inspected supported project surface. | Project hooks expose lifecycle and tool interception, while the plugin CLI exposes no row-renderer contract; `--minimal` changes the whole screen mode rather than selected transcript rows. | These conclusions are deliberately limited to the named versions and supported surfaces. @@ -269,8 +259,8 @@ skip: set FM_PI_LIVE_E2E=1 to run the isolated interactive Pi regression ## 2026-07-26 Pi 0.82.0 compatibility verification -Pi 0.82.0 preserved both API-probed presentation seams and every deterministic Calm TUI guarantee. -The globally installed declaration package remained 0.81.1, so the strict typecheck continued to cover that earlier declaration-evidence version while the real CLI exercised 0.82.0. +Pi 0.82.0 preserved both version-bounded presentation seams and every deterministic Calm TUI guarantee. +The globally installed declaration package remained 0.81.1, so the strict typecheck continued to cover that lower supported boundary while the real CLI exercised 0.82.0. ```text $ pi --version diff --git a/docs/tmux-backend.md b/docs/tmux-backend.md index a33b385c142..ae24507c588 100644 --- a/docs/tmux-backend.md +++ b/docs/tmux-backend.md @@ -54,7 +54,10 @@ An existing Pi pane is therefore reported as ambiguous rather than auto-healed, This is the active tmux liveness limitation. Agent liveness and composer safety are separate checks. -The shared classifier in `bin/fm-composer-lib.sh` accepts a shell glyph as an empty agent composer only inside a verified bordered composer. +For a bordered composer, the tmux reader locates the complete box structurally and classifies every content row through the shared ANSI and ghost handling in `bin/fm-composer-lib.sh`. +Real text on any content row is pending, while only an unambiguous box with every row empty is proven empty. +Unreadable, incomplete, or structurally ambiguous boxes fail closed, and panes without a bordered composer retain the compatible cursor-row classification. +The shared classifier accepts a shell glyph as an empty agent composer only inside a verified bordered composer. A bare shell prompt is `unknown`, so away-mode escalation is never injected into a dead shell. Rendered busy detection is also harness-scoped. @@ -63,12 +66,15 @@ The exact selection contract and safety rationale live in [architecture](archite `bin/fm-tmux-lib.sh` owns exact type-and-submit mechanics. It types a message once and retries Enter only until the composer clears. -A cleared composer is the positive delivery acknowledgement; text left in the composer remains `pending`, and `fm-send.sh` reports the failure instead of retyping. +Only a proven empty composer is a positive delivery acknowledgement. +Text left in established structure remains `pending`, text in ambiguous structure remains unproven, and unreadable or unsafe state remains unknown. +`fm-send.sh` reports every unconfirmed verdict as a failure instead of retyping or assuming delivery. OpenCode 1.18.4 has one busy-queue exception. While OpenCode is mid-turn, Enter queues the message but leaves its text visible until the turn completes. -After the normal retry budget, a provably busy pane is accepted as queued, while an idle pane remains `pending` as a genuine swallowed Enter. -`tests/fm-tmux-submit-busy.test.sh` covers busy and idle panes with both pending and cleared composers. +After the normal retry budget, only structurally proven pending text in a provably busy pane is accepted as queued, while an idle pane remains `pending` as a genuine swallowed Enter. +Ambiguous pending text never receives the busy-queue conversion. +`tests/fm-tmux-submit-busy.test.sh` covers busy and idle panes with proven, ambiguous, and cleared composers. ## Limits and regression entry points @@ -78,6 +84,8 @@ After the normal retry budget, a provably busy pane is accepted as queued, while ```sh tests/fm-backend-tmux-smoke.test.sh +tests/fm-composer-ghost.test.sh +tests/fm-kimi-harness.test.sh tests/fm-tmux-submit-busy.test.sh tests/fm-bootstrap.test.sh ``` diff --git a/docs/verification/runtime-backends.md b/docs/verification/runtime-backends.md index b19135c3599..0c154c97603 100644 --- a/docs/verification/runtime-backends.md +++ b/docs/verification/runtime-backends.md @@ -32,13 +32,16 @@ Claude, Codex, OpenCode, and Grok were observed under their own process names. Kimi Code CLI 0.29.1 was observed under `kimi` on 2026-07-25. Pi remained a generic `node` process and is intentionally inconclusive. -The OpenCode 1.18.4 busy-queue behavior and the tmux fallback are pinned by: +The structural multi-row composer reader, Kimi pointer-delivery path, and OpenCode 1.18.4 busy-queue behavior are pinned by: ```sh +tests/fm-composer-ghost.test.sh +tests/fm-kimi-harness.test.sh tests/fm-tmux-submit-busy.test.sh ``` -Expected matrix: pending plus busy is accepted as queued; pending plus idle remains pending; a cleared composer succeeds in either state. +Expected structural matrix: real text on any content row is pending; all-empty complete boxes are empty; unreadable, incomplete, or unsafe boxes are unknown; and non-bordered panes retain cursor-row compatibility. +Expected submit matrix: proven pending plus busy is accepted as queued; proven pending plus idle remains pending; ambiguous pending is never converted by the busy exception; and only a proven empty composer succeeds directly. ## Herdr diff --git a/docs/zellij-backend.md b/docs/zellij-backend.md index b7b2ac08454..367da98ebbd 100644 --- a/docs/zellij-backend.md +++ b/docs/zellij-backend.md @@ -77,7 +77,7 @@ There is a narrow visible race between those calls that no current Zellij flag c Literal send uses bracketed paste followed by a separate explicit Enter. The adapter supports `Enter`, `Esc`, and the one-argument key expression `Ctrl c` through the shared key vocabulary. Zellij exposes no cursor-row, ANSI composer style, or native agent-state signal, so submit acknowledgement remains content-delta based. -This can distinguish no change from a changed screen but is less precise than tmux's cursor row or Herdr's native state plus structural classifier. +This can distinguish no change from a changed screen but is less precise than tmux's structural box reader or Herdr's native state plus structural classifier. Viewport capture has no line-bound option. Routine reads use `dump-screen` and larger peeks use `dump-screen --full`, followed by local trimming. diff --git a/tests/fm-calm-pi-extension.test.sh b/tests/fm-calm-pi-extension.test.sh index a8e09be1456..b63661ad7a4 100755 --- a/tests/fm-calm-pi-extension.test.sh +++ b/tests/fm-calm-pi-extension.test.sh @@ -16,15 +16,14 @@ PI_OPERATIONAL_INPUT="$ROOT/.pi/extensions/lib/fm-operational-input.ts" PI_PACKAGE_DIR=${FM_PI_PACKAGE_DIR:-"$(npm root -g 2>/dev/null)/@earendil-works/pi-coding-agent"} TMUX_SOCKET="fm-calm-$$" TMUX_SESSION="fm-calm-e2e" -# Verified against Pi 0.81.1 and 0.82.0 (docs/calm-mode-feasibility.md). This is -# known-good evidence, not a support ceiling: the fixtures below run against whatever -# Pi is actually installed, and record_pi_version_evidence never rejects a newer -# version. The tracked presentation adapters probe the exact API they patch (see -# .pi/extensions/fm-calm.ts) instead of relying on version inference, so a version -# string is evidence for the record, not a gate. -record_pi_version_evidence() { +PI_COMPAT_VERSIONS="0.81.1 0.82.0" + +require_pi_compat_version() { local version=$1 context=$2 - [ -n "$version" ] || fail "$context could not determine the installed Pi version" + case " $PI_COMPAT_VERSIONS " in + *" $version "*) return 0 ;; + *) fail "$context requires Pi $PI_COMPAT_VERSIONS, found $version" ;; + esac } cleanup() { @@ -67,6 +66,64 @@ find_chrome() { return 1 } +test_static_contract() { + local text assistant_layout operational_user_layout visibility watch operational + assert_present "$EXT" "tracked Pi calm extension is missing" + assert_present "$ASSISTANT_LAYOUT" "tracked Pi Calm assistant-layout adapter is missing" + assert_present "$OPERATIONAL_USER_LAYOUT" "tracked Pi Calm operational-user layout adapter is missing" + assert_present "$VISIBILITY" "tracked Pi calm visibility policy is missing" + text=$(cat "$EXT") + assistant_layout=$(cat "$ASSISTANT_LAYOUT") + operational_user_layout=$(cat "$OPERATIONAL_USER_LAYOUT") + visibility=$(cat "$VISIBILITY") + watch=$(cat "$WATCH_EXT") + operational=$(cat "$PI_OPERATIONAL_INPUT") + assert_contains "$text" 'pi.registerCommand("calm"' "Pi calm extension does not register /calm" + assert_contains "$text" 'pi.on("session_start"' "Pi calm extension does not restore presentation on every session start" + assert_contains "$text" 'loadCalmPreference()' "Pi calm extension does not restore the home-persistent toggle choice" + assert_contains "$text" 'persistCalmPreference(active)' "Pi calm extension does not persist the captain's toggle choice" + assert_not_contains "$text" 'setCalmPresentation(false)' "Pi calm extension still resets the toggle on session start" + assert_contains "$text" 'ctx.ui.setToolsExpanded(!expanded)' "Pi calm extension does not redraw existing custom entries" + assert_contains "$text" 'ctx.ui.setToolsExpanded(expanded)' "Pi calm extension does not restore Ctrl+O state after redraw" + assert_not_contains "$text" 'ctx.navigateTree' "Pi calm extension reconstructs the transcript and drops transient diagnostics" + assert_not_contains "$visibility" 'deliverFirstmateSyntheticInput' "Pi calm visibility policy can still replace operational input semantics" + assert_not_contains "$visibility" 'classifyFirstmateSyntheticInput' "Pi calm visibility policy still classifies operational input for interception" + assert_contains "$text" 'ctx.ui.setWorkingVisible(true)' "Pi calm extension does not preserve Pi's live working row" + assert_not_contains "$text" 'ctx.ui.setWorkingVisible(!active)' "Pi calm extension still hides Pi's live working row" + assert_contains "$text" 'ctx.ui.setHiddenThinkingLabel(active ? "" : undefined)' "Pi calm extension does not hide collapsed thinking labels" + assert_contains "$text" 'installCalmAssistantLayout()' "Pi Calm extension does not install its zero-height assistant layout" + assert_contains "$text" 'installCalmOperationalUserLayout()' "Pi Calm extension does not install its operational-user layout" + assert_contains "$assistant_layout" 'AssistantMessageComponent.prototype.updateContent' "Pi Calm assistant layout does not control the exported component presentation path" + assert_contains "$assistant_layout" 'block.type !== "thinking"' "Pi Calm assistant layout does not remove thinking from its presentation copy" + assert_contains "$operational_user_layout" 'InteractiveMode.prototype' "Pi Calm operational-user layout does not control the transcript owner" + assert_contains "$operational_user_layout" 'classifyFirstmateCurrentOperationalText(text)' "Pi Calm operational-user layout bypasses canonical current classification" + assert_contains "$operational_user_layout" 'text.includes("\u2063")' "Pi Calm operational-user layout spawns its classifier for ordinary captain rows" + assert_contains "$operational_user_layout" '"\u2063Supervisor escalate ("' "Pi Calm operational-user layout lost the narrow legacy marker" + assert_contains "$operational_user_layout" 'hidesOperationalInput()' "Pi Calm operational-user row does not use presentation-only hiding" + assert_not_contains "$operational_user_layout" 'FIRSTMATE_OP: ' "Pi Calm operational-user layout duplicates the canonical marker grammar" + assert_not_contains "$text" 'calm transcript' "Pi calm extension still adds a persistent Calm status row" + assert_not_contains "$text" 'pi.on("input"' "Pi calm extension still intercepts semantic input" + assert_not_contains "$text" 'sendMessage' "Pi calm extension still replaces user-role input with custom context" + assert_contains "$text" 'ctx.ui.onTerminalInput' "Pi calm extension does not scope export rendering to terminal submissions" + assert_contains "$text" 'getKeybindings().matches(data, "tui.input.submit")' "Pi calm export boundary ignores the active submit keybinding" + assert_contains "$text" 'input !== "/share"' "Pi calm export boundary does not cover /share" + assert_not_contains "$text" 'FIRSTMATE_PI_LAUNCH_BRIEF_ENV' "Pi calm presentation still depends on launch-input provenance" + assert_contains "$text" 'renderShell: "self"' "Pi calm extension cannot remove complete built-in tool shells" + assert_contains "$visibility" 'CALM_VISIBLE_CLASSES' "Pi calm policy does not centralize its visibility allowlist" + assert_contains "$operational" 'fm-operational-input.sh' "Pi adapter does not delegate to the canonical cross-language owner" + assert_not_contains "$visibility" 'FIRSTMATE WATCHER WAKE:' "current Calm classification still matches watcher payload prose" + assert_not_contains "$visibility" 'TURN WOULD END BLIND' "current Calm classification still matches turn-end payload prose" + # shellcheck disable=SC2016 # Backticks are literal prompt markup. + assert_not_contains "$visibility" 'Run `bin/fm-session-start.sh`' "current Calm classification still matches session-start payload prose" + assert_not_contains "$visibility" 'FIRSTMATE_OP: ' "current Calm classification duplicates the canonical marker grammar" + assert_contains "$watch" 'calmHides("assistant-tool-call")' "Firstmate watcher tool does not participate in Calm presentation" + assert_contains "$watch" 'renderShell: "self"' "Firstmate watcher tool cannot remove its complete shell" + for name in Read Bash Edit Write Grep Find Ls; do + assert_contains "$text" "create${name}ToolDefinition" "Pi calm extension does not wrap the $name built-in" + done + pass "Pi calm extension is presentation-only with one persisted visibility choice, no Calm status row, native working visibility, supported redraw controls, and the Firstmate watcher-tool integration" +} + test_home_resolution() { local fixture out status version if ! command -v node >/dev/null 2>&1 || ! command -v npm >/dev/null 2>&1; then @@ -78,7 +135,7 @@ test_home_resolution() { return 0 fi version=$(node -p "require('$PI_PACKAGE_DIR/package.json').version") - record_pi_version_evidence "$version" "Pi calm compatibility assumptions" + require_pi_compat_version "$version" "Pi calm compatibility assumptions" fixture="$TMP_ROOT/home-resolution" mkdir -p \ @@ -176,173 +233,6 @@ JS pass "Pi calm resolves its persistent home independently of Pi's launch directory" } -test_pi_compat_no_upper_bound() { - local version - for version in 0.83.0 0.90.0 1.0.0 2.3.4 0.82.1 10.20.30; do - record_pi_version_evidence "$version" "synthetic newer Pi" \ - || fail "record_pi_version_evidence rejected Pi $version solely for being newer than 0.82.0" - done - if (record_pi_version_evidence "" "malformed Pi version probe") 2>/dev/null; then - fail "record_pi_version_evidence accepted a missing/malformed Pi version" - fi - pass "Pi calm compatibility evidence never rejects a Pi version for being newer than 0.82.0, and still fails closed on a missing or malformed version" -} - -test_pi_compat_degraded_adapter() { - local fixture out status - if ! command -v node >/dev/null 2>&1 || ! command -v npm >/dev/null 2>&1; then - echo "skip: node or npm not found for Pi calm degraded-adapter test" - return 0 - fi - if [ ! -f "$PI_PACKAGE_DIR/package.json" ]; then - echo "skip: installed @earendil-works/pi-coding-agent package not found" - return 0 - fi - - fixture="$TMP_ROOT/degraded-adapter" - mkdir -p \ - "$fixture/project/.pi/extensions/lib" \ - "$fixture/project/node_modules/@earendil-works" - cp "$EXT" "$fixture/project/.pi/extensions/fm-calm.ts" - cp "$ASSISTANT_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-assistant-layout.ts" - cp "$OPERATIONAL_USER_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" - cp "$VISIBILITY" "$fixture/project/.pi/extensions/lib/fm-calm-visibility.ts" - cp "$PI_OPERATIONAL_INPUT" "$fixture/project/.pi/extensions/lib/fm-operational-input.ts" - ln -s "$PI_PACKAGE_DIR" "$fixture/project/node_modules/@earendil-works/pi-coding-agent" - ln -s "$PI_PACKAGE_DIR/node_modules/@earendil-works/pi-tui" "$fixture/project/node_modules/@earendil-works/pi-tui" - ln -s "$PI_PACKAGE_DIR/node_modules/typebox" "$fixture/project/node_modules/typebox" - printf '%s\n' '{"type":"module"}' >"$fixture/project/package.json" - - out=$(cd "$fixture/project" && \ - EXT="$fixture/project/.pi/extensions/fm-calm.ts" \ - PI_PACKAGE_DIR="$PI_PACKAGE_DIR" \ - node --input-type=module 2>&1 <<'JS' -import { pathToFileURL } from "node:url"; - -const packageRoot = process.env.PI_PACKAGE_DIR; -const { AssistantMessageComponent } = await import( - pathToFileURL(`${packageRoot}/dist/modes/interactive/components/assistant-message.js`).href -); -const originalUpdateContent = AssistantMessageComponent.prototype.updateContent; -if (typeof originalUpdateContent !== "function") { - throw new Error( - "fixture precondition failed: installed Pi lacks AssistantMessageComponent.prototype.updateContent", - ); -} -delete AssistantMessageComponent.prototype.updateContent; - -const diagnostics = []; -const originalConsoleError = console.error; -console.error = (...args) => diagnostics.push(args.join(" ")); - -let calmCommand; -const handlers = new Map(); -const pi = { - events: { emit() {}, on() {} }, - on(event, handler) { - handlers.set(event, handler); - }, - registerCommand(name, command) { - if (name === "calm") calmCommand = command; - }, - registerEntryRenderer() {}, - registerTool() {}, -}; - -let threw = false; -try { - const extension = await import(`${pathToFileURL(process.env.EXT).href}?degraded=${Date.now()}`); - extension.default(pi); -} catch { - threw = true; -} -console.error = originalConsoleError; - -if (threw) { - throw new Error( - "a missing presentation API crashed the whole Calm extension instead of degrading just that adapter", - ); -} -if (!calmCommand || !handlers.has("session_start")) { - throw new Error( - "Calm command/session lifecycle did not register when only one presentation adapter was unavailable", - ); -} -if (typeof AssistantMessageComponent.prototype.updateContent !== "undefined") { - throw new Error( - "the degraded adapter path patched updateContent anyway despite the missing API, which would claim false success", - ); -} -const sawClearSkipReason = diagnostics.some( - (line) => line.includes("collapsed-thinking") && /unavailable|skip/i.test(line), -); -if (!sawClearSkipReason) { - throw new Error( - `missing a clear skip reason for the degraded collapsed-thinking adapter; saw: ${JSON.stringify(diagnostics)}`, - ); -} - -AssistantMessageComponent.prototype.updateContent = originalUpdateContent; -JS -) - status=$? - [ "$status" -eq 0 ] || fail "Pi calm degraded-adapter path failed: $out" - [ -z "$out" ] || fail "Pi calm degraded-adapter test printed output: $out" - pass "a missing collapsed-thinking presentation API degrades only that Calm adapter with a clear skip reason, while the rest of Calm still registers" -} - -test_pi_compat_missing_adapter_exports() { - local fixture out status - if ! command -v node >/dev/null 2>&1; then - echo "skip: node not found for Pi calm missing-adapter-export test" - return 0 - fi - - fixture="$TMP_ROOT/missing-adapter-exports" - mkdir -p \ - "$fixture/project/.pi/extensions/lib" \ - "$fixture/project/node_modules/@earendil-works/pi-coding-agent" - cp "$ASSISTANT_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-assistant-layout.ts" - cp "$OPERATIONAL_USER_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" - cp "$VISIBILITY" "$fixture/project/.pi/extensions/lib/fm-calm-visibility.ts" - cp "$PI_OPERATIONAL_INPUT" "$fixture/project/.pi/extensions/lib/fm-operational-input.ts" - printf '%s\n' '{"type":"module"}' >"$fixture/project/package.json" - printf '%s\n' \ - '{"name":"@earendil-works/pi-coding-agent","type":"module","exports":"./index.js"}' \ - >"$fixture/project/node_modules/@earendil-works/pi-coding-agent/package.json" - printf '%s\n' \ - 'export function getMarkdownTheme() { return {}; }' \ - 'export class UserMessageComponent {}' \ - >"$fixture/project/node_modules/@earendil-works/pi-coding-agent/index.js" - - out=$(cd "$fixture/project" && node --input-type=module 2>&1 <<'JS' -const assistant = await import("./.pi/extensions/lib/fm-calm-assistant-layout.ts"); -const operational = await import("./.pi/extensions/lib/fm-calm-operational-user-layout.ts"); - -for (const [name, install, expected] of [ - ["collapsed-thinking", assistant.installCalmAssistantLayout, "AssistantMessageComponent"], - ["operational-user-row", operational.installCalmOperationalUserLayout, "InteractiveMode"], -]) { - let reason; - try { - install(); - } catch (error) { - reason = error instanceof Error ? error.message : String(error); - } - if (!reason?.includes(expected)) { - throw new Error( - `${name} adapter did not load and report its missing runtime export: ${String(reason)}`, - ); - } -} -JS -) - status=$? - [ "$status" -eq 0 ] || fail "Pi calm missing-adapter-export path failed: $out" - [ -z "$out" ] || fail "Pi calm missing-adapter-export test printed output: $out" - pass "missing Pi presentation class exports reach the independent adapter degradation path" -} - test_rendering_and_session_lifecycle() { local fixture out status version if ! command -v node >/dev/null 2>&1 || ! command -v npm >/dev/null 2>&1; then @@ -354,7 +244,7 @@ test_rendering_and_session_lifecycle() { return 0 fi version=$(node -p "require('$PI_PACKAGE_DIR/package.json').version") - record_pi_version_evidence "$version" "Pi calm compatibility assumptions" + require_pi_compat_version "$version" "Pi calm compatibility assumptions" fixture="$TMP_ROOT/renderer" mkdir -p "$fixture/home" "$fixture/lib" "$fixture/node_modules/@earendil-works" @@ -1005,7 +895,7 @@ test_operational_followup_turn_e2e() { return 0 fi version=$(pi --version 2>/dev/null || true) - record_pi_version_evidence "$version" "Pi operational follow-up E2E" + require_pi_compat_version "$version" "Pi operational follow-up E2E" project="$TMP_ROOT/followup-project" home="$TMP_ROOT/followup-home" @@ -1358,7 +1248,7 @@ test_hidden_block_geometry_e2e() { return 0 fi version=$(pi --version 2>/dev/null || true) - record_pi_version_evidence "$version" "Pi Calm hidden-block geometry E2E" + require_pi_compat_version "$version" "Pi Calm hidden-block geometry E2E" project="$TMP_ROOT/geometry-project" home="$TMP_ROOT/geometry-home" @@ -1598,7 +1488,7 @@ test_interactive_terminal_e2e() { return 0 fi version=$(pi --version 2>/dev/null || true) - record_pi_version_evidence "$version" "Pi calm interactive E2E" + require_pi_compat_version "$version" "Pi calm interactive E2E" project="$TMP_ROOT/e2e-project" config="$TMP_ROOT/e2e-config" @@ -2085,10 +1975,8 @@ JS pass "Pi calm native E2E keeps Working and captain turns visible, hides exact operational user rows without changing persistence, restores them Calm-off, survives restart, and preserves export plus Ctrl+O behavior" } +test_static_contract test_home_resolution -test_pi_compat_no_upper_bound -test_pi_compat_degraded_adapter -test_pi_compat_missing_adapter_exports test_rendering_and_session_lifecycle test_operational_followup_turn_e2e test_hidden_block_geometry_e2e diff --git a/tests/fm-composer-ghost.test.sh b/tests/fm-composer-ghost.test.sh index 7249574ea3e..d0285528894 100755 --- a/tests/fm-composer-ghost.test.sh +++ b/tests/fm-composer-ghost.test.sh @@ -473,12 +473,12 @@ test_all_tmux_harness_composers_share_classification() { dir="$TMP_ROOT/all-harness-composers"; mkdir -p "$dir" fb=$(make_fake_tmux "$dir") capture="$dir/styled.txt" - for harness in claude codex opencode pi pi-signed grok; do + for harness in claude codex opencode pi grok; do case "$harness" in claude) printf '╭────────────╮\n│ ❯ \033[2mtry\033[0m │\n╰────────────╯\n' > "$capture" ;; codex) printf '╭────────────╮\n│ › \033[2mtip\033[0m │\n╰────────────╯\n' > "$capture" ;; opencode) printf '╭────────────╮\n│ > │\n╰────────────╯\n' > "$capture" ;; - pi|pi-signed) printf '╭────────────╮\n│ │\n╰────────────╯\n' > "$capture" ;; + pi) printf '╭────────────╮\n│ │\n╰────────────╯\n' > "$capture" ;; grok) printf '╭────────────╮\n│ ❯ \033[38;2;50;47;70mType\033[0m │\n╰────────────╯\n' > "$capture" ;; esac out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=1 \ @@ -488,7 +488,7 @@ test_all_tmux_harness_composers_share_classification() { case "$harness" in claude|grok) printf '╭────────────╮\n│ ❯ fix │\n╰────────────╯\n' > "$capture" ;; codex) printf '╭────────────╮\n│ › fix │\n╰────────────╯\n' > "$capture" ;; - opencode|pi|pi-signed) printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$capture" ;; + opencode|pi) printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$capture" ;; esac out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=1 \ fm_tmux_composer_state "fakepane") diff --git a/tests/fm-kimi-harness.test.sh b/tests/fm-kimi-harness.test.sh index 6f3ba6745b2..9e0f450439b 100755 --- a/tests/fm-kimi-harness.test.sh +++ b/tests/fm-kimi-harness.test.sh @@ -45,9 +45,32 @@ make_spawn_fakebin() { set -u printf '%s\n' "$*" >> "$FM_FAKE_TMUX_CALL_LOG" state=$(cat "$FM_FAKE_KIMI_STATE" 2>/dev/null || true) +fake_screen() { + case "$state" in + ready) + printf 'Welcome to Kimi Code!\ncontext: 0%% (0/256k)\n╭────────────────────────────────╮\n│ > │\n╰────────────────────────────────╯\n' + ;; + pointer-typed) + printf 'context: 0%% (0/256k)\n╭────────────────────────────────╮\n│ > Read the brief and follow it │\n│ │\n╰────────────────────────────────╯\n' + ;; + delivered) + printf '✨ Read the brief at %s and follow it exactly.\ncontext: 1%% (2k/256k)\n╭────────────────────────────────╮\n│ > │\n╰────────────────────────────────╯\n' "$FM_FAKE_BRIEF_REAL" + ;; + *) + printf 'shell starting\n$ \n' + ;; + esac +} +fake_cursor_y() { + case "$state" in + pointer-typed) printf '3\n' ;; + ready|delivered) printf '3\n' ;; + *) printf '1\n' ;; + esac +} case "$*" in *"#{pane_current_path}"*) printf '%s\n' "$FM_FAKE_PANE_PATH"; exit 0 ;; - *"#{cursor_y}"*) printf '0\n'; exit 0 ;; + *"#{cursor_y}"*) fake_cursor_y; exit 0 ;; esac case "${1:-}" in display-message) printf 'firstmate\n'; exit 0 ;; @@ -99,19 +122,18 @@ case "${1:-}" in exit 0 ;; capture-pane) - case "$state" in - ready) - printf 'Welcome to Kimi Code!\ncontext: 0%% (0/256k)\n│ > │\n' - ;; - pointer-typed) - printf 'context: 0%% (0/256k)\n│ > pending │\n' - ;; - delivered) - printf '✨ Read the brief at %s and follow it exactly.\ncontext: 1%% (2k/256k)\n│ > │\n' "$FM_FAKE_BRIEF_REAL" - ;; - *) - printf 'shell starting\n$ \n' - ;; + start= end= prev= + for arg in "$@"; do + case "$prev" in + -S) start=$arg ;; + -E) end=$arg ;; + esac + case "$arg" in -S|-E) prev=$arg ;; *) prev= ;; esac + done + case "$start:$end" in + *[!0-9:]*|'':*|*:'') fake_screen ;; + *) fake_screen | awk -v start="$start" -v end="$end" \ + 'NR - 1 >= start && NR - 1 <= end' ;; esac exit 0 ;; diff --git a/tests/fm-pending-reply.test.sh b/tests/fm-pending-reply.test.sh index 254b03a6317..325125eeeb2 100755 --- a/tests/fm-pending-reply.test.sh +++ b/tests/fm-pending-reply.test.sh @@ -59,9 +59,9 @@ case "${1:-}" in fi exit 0 ;; display-message) - for a in "$@"; do case "$a" in *cursor_y*) printf '0\n'; exit 0 ;; esac; done + for a in "$@"; do case "$a" in *cursor_y*) printf '1\n'; exit 0 ;; esac; done printf 'fakepane\n'; exit 0 ;; - capture-pane) printf '\xe2\x94\x82 \xe2\x94\x82\n'; exit 0 ;; + capture-pane) printf '╭────╮\n│ │\n╰────╯\n'; exit 0 ;; list-windows) exit 0 ;; esac exit 0 @@ -725,14 +725,14 @@ test_kimi_capture_fallback_uses_recorded_harness() ( fm_write_secondmate_meta "$state/hibit.meta" "$sm_home" "session:fm-hibit" alpha kimi fm_backend_busy_state() { printf 'unknown'; } fm_backend_capture() { printf '%s' "$FM_PENDING_KIMI_CAPTURE"; } - export FM_PENDING_KIMI_CAPTURE=' 🌑 ·Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"' + export FM_PENDING_KIMI_CAPTURE=' 🌑 · Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"' [ "$(fm_pending_reply_backend_observation tmux session:fm-hibit fm-hibit codex)" = fallback-idle ] \ || fail "Kimi spinner leaked into another harness" export FM_PENDING_KIMI_CAPTURE='Ctrl+c:cancel' [ "$(fm_pending_reply_backend_observation tmux session:fm-hibit fm-hibit kimi)" = fallback-idle ] \ || fail "Grok's exact busy token leaked into Kimi pending-reply observation" - export FM_PENDING_KIMI_CAPTURE=' 🌑 ·Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"' + export FM_PENDING_KIMI_CAPTURE=' 🌑 · Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"' fm_pending_reply_tick "$state" rec=$(fm_pending_reply_path "$state" "$corr") [ "$(fm_pending_reply_get "$rec" turn_seen_busy)" = 1 ] \ diff --git a/tests/fm-tmux-submit-busy.test.sh b/tests/fm-tmux-submit-busy.test.sh index aafdbbae6ca..58932509d0e 100755 --- a/tests/fm-tmux-submit-busy.test.sh +++ b/tests/fm-tmux-submit-busy.test.sh @@ -27,7 +27,7 @@ COMPOSER="${FM_FAKE_COMPOSER:?}" case "${1:-}" in display-message) for a in "$@"; do - case "$a" in *cursor_y*) printf '0\n'; exit 0 ;; esac + case "$a" in *cursor_y*) printf '1\n'; exit 0 ;; esac done exit 0 ;; capture-pane) cat "$COMPOSER" 2>/dev/null; exit 0 ;; @@ -37,10 +37,11 @@ case "${1:-}" in case "$1" in -t) shift ;; -l) ;; Enter) is_enter=1 ;; esac; shift done if [ "$is_enter" = 1 ]; then + [ -z "${FM_FAKE_SENT:-}" ] || printf 'Enter\n' >> "$FM_FAKE_SENT" if [ -n "${FM_FAKE_SWALLOW:-}" ] && [ -f "$FM_FAKE_SWALLOW" ]; then [ "${FM_FAKE_PERSIST_SWALLOW:-0}" = 1 ] || rm -f "$FM_FAKE_SWALLOW" else - printf '│ > │\n' > "$COMPOSER" + printf '╭─────╮\n│ > │\n╰─────╯\n' > "$COMPOSER" fi fi exit 0 ;; @@ -59,7 +60,7 @@ test_busy_pane_pending_returns_empty() { composer="$dir/composer" sent="$dir/sent.log" vfile="$dir/verdict" - printf '│ > fix findings 1 and 3 │\n' > "$composer" + printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$composer" : > "$sent" touch "$dir/.swallow" # Pre-check: composer state should be pending (via function, not $()). @@ -70,8 +71,8 @@ test_busy_pane_pending_returns_empty() { FM_FAKE_SWALLOW="$dir/.swallow" FM_FAKE_PERSIST_SWALLOW=1 FM_FAKE_PANE_BUSY=1 \ fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null [ "$(cat "$vfile")" = empty ] || fail "busy-pane pending should return empty, got '$(cat "$vfile")'" - [ "$(grep -c 'fix findings' "$sent" 2>/dev/null || true)" -eq 0 ] \ - || fail "busy-pane should not retype text" + [ "$(grep -c '^Enter$' "$sent" 2>/dev/null || true)" -eq 3 ] \ + || fail "proven pending should consume the configured Enter retry budget" pass "fm_tmux_submit_enter_core: busy pane + pending composer returns empty (message queued)" } @@ -82,7 +83,7 @@ test_idle_pane_pending_returns_pending() { composer="$dir/composer" sent="$dir/sent.log" vfile="$dir/verdict" - printf '│ > fix findings 1 and 3 │\n' > "$composer" + printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$composer" : > "$sent" touch "$dir/.swallow" PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" FM_FAKE_SENT="$sent" \ @@ -99,7 +100,7 @@ test_busy_pane_composer_clears_first_try() { composer="$dir/composer" sent="$dir/sent.log" vfile="$dir/verdict" - printf '│ > fix findings 1 and 3 │\n' > "$composer" + printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$composer" : > "$sent" PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" FM_FAKE_SENT="$sent" FM_FAKE_PANE_BUSY=1 \ fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null @@ -114,7 +115,7 @@ test_idle_pane_composer_clears_first_try() { composer="$dir/composer" sent="$dir/sent.log" vfile="$dir/verdict" - printf '│ > fix findings 1 and 3 │\n' > "$composer" + printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$composer" : > "$sent" PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" FM_FAKE_SENT="$sent" FM_FAKE_PANE_BUSY=0 \ fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null @@ -122,6 +123,68 @@ test_idle_pane_composer_clears_first_try() { pass "fm_tmux_submit_enter_core: idle pane clears composer on first Enter - returns empty as before" } +test_busy_pane_unknown_stays_unknown() { + local dir fakebin composer vfile + dir="$TMP_ROOT/busy-unknown" + fakebin=$(make_submit_mock "$dir") + composer="$dir/composer" + vfile="$dir/verdict" + printf '│ > unbounded\n' > "$composer" + touch "$dir/.swallow" + PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" FM_FAKE_PANE_BUSY=1 \ + FM_FAKE_SWALLOW="$dir/.swallow" FM_FAKE_PERSIST_SWALLOW=1 \ + fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null + [ "$(cat "$vfile")" = unknown ] \ + || fail "a busy pane must not convert an unsafe composer to empty, got '$(cat "$vfile")'" + pass "fm_tmux_submit_enter_core: busy conversion is limited to proven pending input" +} + +test_busy_pane_ambiguous_pending_retries_without_conversion() { + local dir fakebin composer sent vfile + dir="$TMP_ROOT/busy-ambiguous-pending" + fakebin=$(make_submit_mock "$dir") + composer="$dir/composer" + sent="$dir/sent.log" + vfile="$dir/verdict" + : > "$sent" + printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$composer" + touch "$dir/.swallow" + PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" fm_tmux_composer_state "win" > "$vfile" 2>/dev/null + [ "$(cat "$vfile")" = pending-unproven ] \ + || fail "ambiguous composer text should be pending-unproven, got '$(cat "$vfile")'" + PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" FM_FAKE_SENT="$sent" FM_FAKE_PANE_BUSY=1 \ + FM_FAKE_SWALLOW="$dir/.swallow" FM_FAKE_PERSIST_SWALLOW=1 \ + fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null + [ "$(cat "$vfile")" = pending-unproven ] \ + || fail "a busy pane must not convert pending-unproven to empty, got '$(cat "$vfile")'" + [ "$(grep -c '^Enter$' "$sent" 2>/dev/null || true)" -eq 3 ] \ + || fail "pending-unproven should consume the configured Enter retry budget" + pass "fm_tmux_submit_enter_core: pending-unproven retries without busy conversion" +} + +test_unrecognized_state_skips_busy_conversion() { + local dir fakebin composer busy_called vfile + dir="$TMP_ROOT/unrecognized-state" + fakebin=$(make_submit_mock "$dir") + composer="$dir/composer" + busy_called="$dir/busy-called" + vfile="$dir/verdict" + printf '╭─────╮\n│ > │\n╰─────╯\n' > "$composer" + ( + # shellcheck disable=SC2329 + fm_tmux_composer_state() { printf 'future-state'; } + # shellcheck disable=SC2329 + fm_pane_is_busy() { touch "$busy_called"; return 0; } + PATH="$fakebin:$PATH" FM_FAKE_COMPOSER="$composer" \ + fm_tmux_submit_enter_core "win" 3 0.05 > "$vfile" 2>/dev/null + ) || fail "unrecognized-state submit check failed" + [ "$(cat "$vfile")" = future-state ] \ + || fail "unrecognized state should be preserved, got '$(cat "$vfile")'" + [ ! -e "$busy_called" ] \ + || fail "unrecognized state must not trigger busy conversion" + pass "fm_tmux_submit_enter_core: unrecognized states skip busy conversion" +} + test_claude_busy_signature_uses_real_capture_shapes() { local dir fakebin composer dir="$TMP_ROOT/claude-signature" @@ -197,4 +260,7 @@ test_busy_pane_pending_returns_empty test_idle_pane_pending_returns_pending test_busy_pane_composer_clears_first_try test_idle_pane_composer_clears_first_try +test_busy_pane_unknown_stays_unknown +test_busy_pane_ambiguous_pending_retries_without_conversion +test_unrecognized_state_skips_busy_conversion test_claude_busy_signature_uses_real_capture_shapes From 77bb8f5ea5c2ac312c2db37c0c91a84cffaab065 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Mon, 27 Jul 2026 12:28:14 -0700 Subject: [PATCH 18/52] feat(bin): add verified pi-signed runtime adapter (#1145) * feat: add verified pi-signed adapter * no-mistakes(review): Correct pi-signed maintainer verification date * no-mistakes(review): Correct remaining pi-signed verification dates * no-mistakes(review): Preserve authoritative pi-signed runtime identity * no-mistakes(document): Document pi-signed shared adapter semantics * no-mistakes: apply CI fixes --- .agents/skills/afk/SKILL.md | 2 +- .agents/skills/firstmate-orca/SKILL.md | 2 +- .agents/skills/harness-adapters/SKILL.md | 32 +-- AGENTS.md | 2 +- README.md | 8 +- bin/backends/tmux.sh | 2 +- bin/fm-bootstrap.sh | 6 +- bin/fm-harness.sh | 8 +- bin/fm-session-lock-lib.sh | 15 +- bin/fm-spawn.sh | 31 ++- bin/fm-tmux-lib.sh | 2 +- docs/architecture.md | 4 +- docs/arm-pretool-check.md | 6 +- docs/configuration.md | 8 +- docs/sessionstart-nudge.md | 4 +- docs/supervision-protocols/pi.md | 12 +- docs/tmux-backend.md | 8 +- docs/turnend-guard.md | 6 +- docs/verification/runtime-backends.md | 48 ++++- docs/verification/supervision.md | 2 + tests/fm-composer-ghost.test.sh | 6 +- tests/fm-instruction-owners.test.sh | 2 +- tests/fm-kimi-harness.test.sh | 4 +- tests/fm-secondmate-harness.test.sh | 252 +++-------------------- tests/fm-secondmate-liveness.test.sh | 24 ++- tests/fm-spawn-dispatch-profile.test.sh | 73 ++++++- tests/fm-tmux-submit-busy.test.sh | 1 + 27 files changed, 276 insertions(+), 294 deletions(-) diff --git a/.agents/skills/afk/SKILL.md b/.agents/skills/afk/SKILL.md index 21b7cdd3111..95f64b11e03 100644 --- a/.agents/skills/afk/SKILL.md +++ b/.agents/skills/afk/SKILL.md @@ -84,7 +84,7 @@ The daemon constructs every current injection as the `away-supervisor` kind owne The bare `FM_INJECT_MARK` form remains accepted for legacy daemon escalations during rollout. U+2063 has no normal keyboard keystroke and survives terminal transport as UTF-8 text. This is how firstmate tells a daemon escalation apart from a real message in the same pane. -The operational prefix travels with the message text; it does not rely on harness-level typed-vs-injected detection, which is not portable across claude, codex, opencode, pi, grok, and kimi. +The operational prefix travels with the message text; it does not rely on harness-level typed-vs-injected detection, which is not portable across claude, codex, opencode, pi, pi-signed, grok, and kimi. ## Busy-guard and composer guard diff --git a/.agents/skills/firstmate-orca/SKILL.md b/.agents/skills/firstmate-orca/SKILL.md index 3db6d22c98c..d8d50b07b47 100644 --- a/.agents/skills/firstmate-orca/SKILL.md +++ b/.agents/skills/firstmate-orca/SKILL.md @@ -13,7 +13,7 @@ It does not replace `AGENTS.md`, `docs/orca-backend.md`, or `harness-adapters`. Orca is a runtime backend, not an agent harness. The runtime backend owns the task endpoint and, for Orca, the task worktree. -The harness is the agent process launched inside that endpoint, such as `claude`, `codex`, `opencode`, `pi`, `grok`, or `kimi`. +The harness is the agent process launched inside that endpoint, such as `claude`, `codex`, `opencode`, `pi`, `pi-signed`, `grok`, or `kimi`. Load `harness-adapters` for harness-specific launch, interrupt, resume, trust-dialog, and skill-invocation facts. Implementation details, metadata fields, teardown guarantees, and limitations live in `docs/orca-backend.md`. diff --git a/.agents/skills/harness-adapters/SKILL.md b/.agents/skills/harness-adapters/SKILL.md index 85cd11c35ce..c5ba453f95b 100644 --- a/.agents/skills/harness-adapters/SKILL.md +++ b/.agents/skills/harness-adapters/SKILL.md @@ -1,6 +1,6 @@ --- name: harness-adapters -description: Agent-only reference for firstmate harness operations. Use before spawning or recovering a crewmate or secondmate, handling a trust dialog, sending a harness-specific skill invocation, interrupting or exiting an agent, resuming an exited agent, or verifying a new harness adapter. Contains verified facts for claude, codex, opencode, pi, grok, and kimi. +description: Agent-only reference for firstmate harness operations. Use before spawning or recovering a crewmate or secondmate, handling a trust dialog, sending a harness-specific skill invocation, interrupting or exiting an agent, resuming an exited agent, or verifying a new harness adapter. Contains verified facts for claude, codex, opencode, pi, pi-signed, grok, and kimi. user-invocable: false metadata: internal: true @@ -38,6 +38,7 @@ If the captain asks for a new harness, propose verifying it first: spawn a trivi ## Detection `bin/fm-harness.sh` prints firstmate's own harness, using verified env markers first and then process ancestry. +Within the Pi family, only the exact launch-boundary marker `FM_PI_HARNESS=pi-signed` alongside `PI_CODING_AGENT=true` selects the signed identity; unmarked shared launcher ancestry remains `pi`. `bin/fm-harness.sh crew` resolves the effective crewmate harness from `config/crew-harness` (absent or `default` -> own). `bin/fm-harness.sh secondmate` resolves the secondmate-launch harness through the chain `config/secondmate-harness` -> `config/crew-harness` -> own, so an unset `config/secondmate-harness` matches the crew harness. `bin/fm-spawn.sh` uses `crew` mode for a crewmate/scout launch and `secondmate` mode for a `--secondmate` launch, re-resolving on every spawn so the split is durable across respawns; an explicit per-spawn harness arg overrides either. @@ -50,9 +51,9 @@ Use that value for interrupt, exit, resume, and skill-invocation facts. ## Primary turn-end guard -The primary integrations for `claude`, `codex`, `opencode`, `pi`, and `grok` have empirically validated hook paths for the "no turn ends blind" guard. +The primary integrations for `claude`, `codex`, `opencode`, `pi`, `pi-signed`, and `grok` have empirically validated hook paths for the "no turn ends blind" guard. `claude` and `codex` block directly through Stop hooks that preserve exit status 2 and stderr from `bin/fm-turnend-guard.sh`. -`opencode`, `pi`, and `grok` expose passive lifecycle callbacks for this purpose, so their tracked primary adapters force one bounded follow-up or resume when the shared predicate blocks. +`opencode`, `pi`, `pi-signed`, and `grok` expose passive lifecycle callbacks for this purpose, so their tracked primary adapters force one bounded follow-up or resume when the shared predicate blocks. Kimi is outside the primary turn-end guard scope, while `docs/turnend-guard.md` owns its separate guarded global hook for crew wake signals. The exact hook files, commands, scoping rules, and fail-open tradeoffs are owned by `docs/turnend-guard.md`. `docs/verification/supervision.md` "Turn-end guard" owns active validation evidence. @@ -60,9 +61,9 @@ When changing any primary turn-end hook, validate the real harness behavior in a ## Primary pre-arm (PreToolUse) seatbelt -The primary integrations for `claude`, `codex`, `opencode`, `pi`, and `grok` also have wired PreToolUse-equivalent hooks that deny a watcher-arm anti-pattern (shell `&`, truncating pipe, bundling, broad `pkill -f fm-watch`) before it runs. +The primary integrations for `claude`, `codex`, `opencode`, `pi`, `pi-signed`, and `grok` also have wired PreToolUse-equivalent hooks that deny a watcher-arm anti-pattern (shell `&`, truncating pipe, bundling, broad `pkill -f fm-watch`) before it runs. `claude` and `codex` block directly through PreToolUse hooks; `grok` blocks the same way but requires every `$VAR` reference in its hook `command` string to carry an inline `:-default` or it fails to launch the hook entirely. -`opencode` and `pi` block by throwing from `tool.execute.before` / returning `{block: true}` from `tool_call`. +`opencode`, `pi`, and `pi-signed` block by throwing from `tool.execute.before` / returning `{block: true}` from `tool_call`. The exact hook files, commands, output-shaping quirks (Claude Code only honors the deny when stdout is empty), and validation transcripts are owned by `docs/arm-pretool-check.md`. When changing any watcher-arm PreToolUse hook, validate the real harness behavior in a scratch project before trusting it, then update that doc. ## Primary delegation-shape guard @@ -87,7 +88,7 @@ Full mechanics, scoping, and fail-open behavior live in `docs/sessionstart-nudge - `claude`: verified native `SessionStart` stdout injection; `.claude/settings.json` matches `startup`, `resume`, and `clear`, but not `compact`. - `codex`: verified on 0.144.4; `.codex/hooks.json` receives `source=startup`, and wrapper stdout reaches model context. - `opencode`: verified on 1.17.18; `session.created` plus `client.session.promptAsync` starts the nudge turn in the TUI, while `opencode run` remains fail-open headless. -- `pi`: verified native `session_start`; the existing primary extension handles `startup`, `new`, and `resume` and uses `pi.sendMessage` to inject context without racing a positional launch prompt. +- `pi` and `pi-signed`: verified native `session_start`; the existing primary extension handles `startup`, `new`, and `resume` and uses `pi.sendMessage` to inject context without racing a positional launch prompt. - `grok`: the 0.2.103 project `SessionStart` event fires with `source=new`, but stdout does not reach model context; the tracked project hook remains fail-open, and a global token-guarded fallback requires a captain decision. ## Primary watcher supervision @@ -97,7 +98,7 @@ Do not substitute another harness's wait shape when resuming supervision. Claude's Stop `asyncRewake` hook (`bin/fm-claude-stop-autoarm.sh`) owns tokenless re-arm around `bin/fm-watch-arm.sh`, and Grok uses tracked background-notify cycles around `bin/fm-watch-arm.sh`. Codex uses bounded foreground checkpoints through `bin/fm-watch-checkpoint.sh` because Codex cannot reason while a foreground tool call is running. OpenCode uses `.opencode/plugins/fm-primary-watch-arm.js`, which coordinates with the turn-end guard plugin and wakes the TUI with `client.session.promptAsync`. -Pi uses the tracked `.pi/extensions/fm-primary-turnend-guard.ts` plus the tracked `.pi/extensions/fm-primary-pi-watch.ts`, both project-local extensions Pi auto-discovers once trusted. +Pi and pi-signed use the tracked `.pi/extensions/fm-primary-turnend-guard.ts` plus the tracked `.pi/extensions/fm-primary-pi-watch.ts`, both project-local extensions the Pi engine auto-discovers once trusted. When changing any primary watcher adapter, update `docs/supervision-protocols/`, `docs/turnend-guard.md` if a shared idle or turn-end hook changed, and the relevant concise fact below. ## Launch profile axes @@ -120,7 +121,7 @@ The supported launch-profile flags below are verified locally; each row records | claude | `--model <model>` | `--effort <low\|medium\|high\|xhigh\|max>` | Verified on Claude Code 2.1.196. | | codex | `--model <model>` | `-c 'model_reasoning_effort="<low\|medium\|high\|xhigh>"'` | Verified on codex-cli 0.142.1. The installed binary schema contains `model_reasoning_effort`, the active config uses it, and the bundled model catalog advertises only low/medium/high/xhigh. `max` is omitted. | | grok | `--model <model>` | `--reasoning-effort <low\|medium\|high>` | Verified on grok 0.2.99 (2026-07-13). `--effort` is an alias, but firstmate's profile axis is reasoning effort. As of 0.2.99 the ceiling is `high`; both `xhigh` and `max` are rejected with `use one of: high, medium, low`, so firstmate omits them. | -| pi | `--model <model>` | `--thinking <low\|medium\|high\|xhigh\|max>` | Verified 2026-07-13 on Pi 0.80.6. `pi --help` advertises `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, and `max`; `pi --print --model openai-codex/gpt-5.6-sol --thinking max 'Reply with exactly OK.'` completed successfully. | +| pi / pi-signed | `--model <model>` | `--thinking <low\|medium\|high\|xhigh\|max>` | Verified 2026-07-27 on Pi and pi-signed 0.82.0. Both expose the same accepted thinking levels and completed the same model-qualified max-thinking smoke. | | opencode | `--model <provider/model>` | none for firstmate's interactive launch | Verified on opencode 1.17.6. `opencode run` has `--variant`, but firstmate launches the interactive `opencode --prompt` path, which has no verified effort flag. | | kimi | `--model <model>` | none | Verified 2026-07-25 on Kimi Code CLI 0.29.1. | @@ -134,7 +135,7 @@ Use the discovery surface in the current authenticated environment because suppo | claude | Open the current interactive session's `/model` picker; `claude --help` documents the accepted alias or full-model-name input shape. | | codex | Open the current interactive session's `/model` picker. | | opencode | Run `opencode models [provider]`, which lists available provider/model identifiers. | -| pi | Run `pi --list-models [search]`; Pi's installed `docs/models.md` owns how built-in, extension-registered, and custom provider/model entries reach that list. | +| pi / pi-signed | Run the selected executable as `<executable> --list-models [search]`; Pi's installed `docs/models.md` owns how built-in, extension-registered, and custom provider/model entries reach that list. | | grok | Run `grok models`, which lists the models available to the current Grok installation and account. | | kimi | Run `kimi provider list --json`, which lists the current provider and model configuration. | @@ -152,7 +153,7 @@ Natural language is acceptable if uncertain. - claude: `/<skill>`, for example `/no-mistakes`. - codex: `$<skill>`, for example `$no-mistakes`; `/<skill>` is claude-only and codex rejects it as "Unrecognized command". - opencode: no separate verified skill invocation beyond normal slash-command behavior; use natural language if the exact skill command is uncertain. -- pi: no separate verified skill invocation beyond normal command behavior; use natural language if the exact skill command is uncertain. +- pi and pi-signed: no separate verified skill invocation beyond normal command behavior; use natural language if the exact skill command is uncertain. - grok: `/<skill>`, for example `/no-mistakes` (same form as claude). Verified end to end: grok discovers the user-level `no-mistakes` skill, `/no-mistakes` invokes it, and grok drives a real `no-mistakes axi run`. Like codex's `$`/`/` popups, typing `/<skill>` opens grok's slash-autocomplete, so a too-fast Enter selects the popup entry instead of sending, and for an argument-taking command (like `/no-mistakes`'s optional task-first argument) that first Enter only expands the popup selection into an argument-hint placeholder rather than submitting - a genuine second Enter is required (see the grok section below for the 2026-07-03 incident and fix). `fm_tmux_submit_core`'s retried Enter (used by `fm-send` on the tmux backend) handles this through the structural composer reader; the herdr backend needed a dedicated fix (`fm_backend_herdr_composer_state`, docs/herdr-backend.md) because its prior delta-based verification false-positived on that same popup-close content change. - kimi: `/<skill>`, for example `/no-mistakes`. @@ -260,7 +261,7 @@ Throwing from `session.idle` does not block `opencode run`, so the primary adapt The companion `.opencode/plugins/fm-primary-watch-arm.js` owns normal TUI watcher wake supervision and coordinates with the guard plugin before the guard tries a blind-turn follow-up. The follow-up was verified in the interactive TUI; `opencode run` can exit before displaying a queued follow-up, so the adapter is fail-open in headless mode. -## pi (VERIFIED 2026-06-11) +## pi and pi-signed (VERIFIED 2026-07-27) | Fact | Value | |---|---| @@ -269,6 +270,11 @@ The follow-up was verified in the interactive TUI; `opencode run` can exit befor | Interrupt | single Escape | Pi has no permission system, so crewmates are always autonomous. +`pi-signed` is the signed wrapper identity verified on version 0.82.0 and exposes the same CLI and TUI behavior as Pi. +Firstmate launches the selected executable name from `PATH`, records `pi-signed` without normalization, and refuses rather than falling back to `pi` when that wrapper is unavailable. +The observed signed process tree is an exact `pi-signed` wrapper parent with the Pi application as its child, while tmux reports the foreground command as the exact `pi-launcher` name for both selected executables. +The installed plain `pi` command also execs that signed launcher, so `FM_PI_HARNESS=pi-signed` is the authoritative selection marker and shared unmarked ancestry remains `pi`. +Firstmate sets `FM_PI_HARNESS` explicitly for both worker launch identities, and a signed primary uses the README launch command to establish the same boundary. Keep the brief as one positional argument. Multiple positional args become separate queued messages; `fm-spawn`'s template already does this correctly. @@ -285,8 +291,8 @@ The firstmate PRIMARY's own `.pi/extensions/fm-primary-turnend-guard.ts` listens Without `deliverAs: "followUp"`, Pi rejects the send while the agent is still processing. Pi's primary watcher protocol also requires the tracked `.pi/extensions/fm-primary-pi-watch.ts` extension, same trust-once discovery as the turn-end guard. The model arms through `fm_watch_arm_pi`, never a foreground bash arm; the watcher tool result and clean-exit fallback are owned by `docs/supervision-protocols/pi.md`. -`bin/fm-session-start.sh` reports when the live Pi session has not loaded both the turn-end guard and watcher extensions, and points at plain `pi` after project trust as the fix, with `-e` as a trust-free fallback. -When a secondmate is launched on Pi, `fm-spawn.sh --secondmate` launches Pi with both `-e .pi/extensions/fm-primary-turnend-guard.ts` and `-e .pi/extensions/fm-primary-pi-watch.ts`, both already present in the secondmate home's git worktree. +`bin/fm-session-start.sh` reports when the live Pi-family session has not loaded both the turn-end guard and watcher extensions, and points at the selected executable after project trust as the fix, with `-e` as a trust-free fallback. +When a secondmate is launched on Pi or pi-signed, `fm-spawn.sh --secondmate` launches the selected executable with both `-e .pi/extensions/fm-primary-turnend-guard.ts` and `-e .pi/extensions/fm-primary-pi-watch.ts`, both already present in the secondmate home's git worktree. ## grok (VERIFIED 2026-06-29, grok 0.2.73; slash-submit re-verified 2026-07-03 on 0.2.82; reasoning-effort ceiling re-verified 2026-07-13 on 0.2.99; exit paths re-verified 2026-07-19 on grok 0.2.103) diff --git a/AGENTS.md b/AGENTS.md index 382413e8214..d85e90b8eaa 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -158,7 +158,7 @@ A silent bootstrap section needs no action; for any printed actionable diagnosti ## 4. Harness and runtime dispatch Load `harness-adapters` before every spawn or recovery and before trust handling, skill invocation, interrupt, exit, resume, or adapter verification. -The verified harnesses are `claude`, `codex`, `opencode`, `pi`, `grok`, and `kimi`; never dispatch on an unverified adapter. +The verified harnesses are `claude`, `codex`, `opencode`, `pi`, `pi-signed`, `grok`, and `kimi`; never dispatch on an unverified adapter. If static `config/crew-harness` or `config/secondmate-harness` names an unverified adapter, report it and fall back only to a verified adapter rather than launching it. `docs/configuration.md` owns dispatch-profile and runtime-backend schemas, `bin/fm-harness.sh` owns static resolution, and `bin/fm-spawn.sh` owns launch flags and fail-closed validation. diff --git a/README.md b/README.md index 05465762659..a7f69e39c23 100644 --- a/README.md +++ b/README.md @@ -58,7 +58,7 @@ Full detail on every feature lives in [docs/architecture.md](docs/architecture.m ### Requirements -- A verified primary agent harness: Claude Code, Grok, Pi, Codex, or OpenCode. +- A verified primary agent harness: Claude Code, Grok, Pi, `pi-signed`, Codex, or OpenCode. - Git and the GitHub CLI, authenticated through `gh auth login`. - The CLI and dependencies for your selected runtime backend; tmux is the reference default. @@ -67,7 +67,7 @@ Backend-specific setup is linked in [Documentation](#documentation). ### Recommended harnesses -**Claude Code, Grok, and Pi are equal co-primary recommendations** for running the primary firstmate session. +**Claude Code, Grok, and Pi are equal co-primary recommendations** for running the primary firstmate session, with `pi-signed` supported as Pi's distinct signed-wrapper identity. Claude Code uses a tracked Stop hook for tokenless watcher re-arm and rewake, Grok uses background-notify wake cycles, and Pi uses its tracked primary watcher extension. All three have verified turn-end guard paths when launched with their documented setup. Pick whichever one matches your subscription and workflow. @@ -100,6 +100,8 @@ grok --trust ```sh pi +# or, when the signed wrapper is installed +FM_PI_HARNESS=pi-signed pi-signed ``` For Grok, `--trust` is needed once per clone so project hooks and the turn-end guard load; `/hooks-trust` inside Grok works too. @@ -201,7 +203,7 @@ Firstmate's skills live in two separate places with different audiences: - [docs/gitlab-merge-watch.md](docs/gitlab-merge-watch.md) - maintainer verification for GitLab merge watching on arbitrary instances. - [docs/turnend-guard.md](docs/turnend-guard.md) - the primary session's current "no turn ends blind" backstop, scope, loop safety, and compatibility limits. - [docs/verification/supervision.md](docs/verification/supervision.md) - active maintainer verification for session-start, guard, continuity, and wedge integrations. -- [docs/supervision-protocols/](docs/supervision-protocols/) - rendered primary-harness watcher protocols for Claude, Codex, OpenCode, Pi, Grok, and unknown harness fallback. +- [docs/supervision-protocols/](docs/supervision-protocols/) - rendered primary-harness watcher protocols for Claude, Codex, OpenCode, Pi and `pi-signed`, Grok, and unknown harness fallback. - [docs/scripts.md](docs/scripts.md) - the `bin/` toolbelt reference. - [docs/documentation-audiences.md](docs/documentation-audiences.md) - documentation audiences and the machine-checked placement boundary. - [`AGENTS.md`](AGENTS.md) - the distro's always-loaded operating contract and routing index for conditional procedures. diff --git a/bin/backends/tmux.sh b/bin/backends/tmux.sh index b618e055bc4..fe0ed716a42 100644 --- a/bin/backends/tmux.sh +++ b/bin/backends/tmux.sh @@ -181,7 +181,7 @@ fm_backend_tmux_agent_state() { # <target> } comm=${comm#-} case "$comm" in - *claude*|*codex*|*opencode*|*grok*|*kimi*) printf 'alive' ;; + *claude*|*codex*|*opencode*|*grok*|*kimi*|pi|pi-signed|pi-launcher|Pi) printf 'alive' ;; zsh|bash|sh|dash|ash|ksh|mksh|tcsh|csh|fish) printf 'dead' ;; '') printf 'unreadable' ;; *) printf 'ambiguous' ;; diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index 12001223353..8685b2e2bbd 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -436,7 +436,7 @@ secondmate_liveness_sweep() { [ -n "$target" ] || target="$window" agent_state=$(fm_backend_agent_state "$backend" "$target" 2>/dev/null) || agent_state=unreadable case "$harness" in - claude|codex|opencode|pi|grok|kimi) ;; + claude|codex|opencode|pi|pi-signed|grok|kimi) ;; *) case "$agent_state" in dead|missing) agent_state=unverified-harness ;; esac ;; @@ -713,14 +713,14 @@ crew_dispatch_validate() { return 0 fi err=$(jq -r ' - def verified($h): ["claude","codex","opencode","pi","grok","kimi"] | index($h); + def verified($h): ["claude","codex","opencode","pi","pi-signed","grok","kimi"] | index($h); def effort_ok($h; $e): if $e == null then true elif ($e | type) != "string" then false elif $h == "claude" then (["low","medium","high","xhigh","max"] | index($e)) elif $h == "codex" then (["low","medium","high","xhigh"] | index($e)) elif $h == "grok" then (["low","medium","high"] | index($e)) - elif $h == "pi" then (["low","medium","high","xhigh","max"] | index($e)) + elif $h == "pi" or $h == "pi-signed" then (["low","medium","high","xhigh","max"] | index($e)) elif $h == "opencode" or $h == "kimi" then false else true end; diff --git a/bin/fm-harness.sh b/bin/fm-harness.sh index e9c1e1c24a3..f2ee8fe7e81 100755 --- a/bin/fm-harness.sh +++ b/bin/fm-harness.sh @@ -1,6 +1,6 @@ #!/usr/bin/env bash # Detect the agent harness this process tree runs on. -# Usage: fm-harness.sh print own harness: claude|codex|opencode|pi|grok|kimi|unknown +# Usage: fm-harness.sh print own harness: claude|codex|opencode|pi|pi-signed|grok|kimi|unknown # fm-harness.sh crew print the effective CREWMATE harness # (config/crew-harness; "default" resolves to own) # fm-harness.sh secondmate print the harness the PRIMARY uses to launch @@ -36,7 +36,10 @@ detect_own() { # ancestry is consulted. This is a precedence hazard, not evidence that # CLAUDECODE inheritance into a kimi child was observed; it was not observed. [ "${CLAUDECODE:-}" = "1" ] && { echo claude; return; } - [ "${PI_CODING_AGENT:-}" = "true" ] && { echo pi; return; } + if [ "${PI_CODING_AGENT:-}" = "true" ]; then + if [ "${FM_PI_HARNESS:-}" = pi-signed ]; then echo pi-signed; else echo pi; fi + return + fi # grok sets GROK_AGENT=1 for its child/tool processes (verified, grok 0.2.73). # It does NOT set CLAUDECODE despite being Claude-Code-compatible, so this marker # is unambiguous when firstmate runs natively on grok. @@ -51,6 +54,7 @@ detect_own() { *opencode*) echo opencode; return ;; *grok*) echo grok; return ;; kimi) echo kimi; return ;; + pi-signed) echo pi; return ;; pi) echo pi; return ;; node*|python*) # Bare interpreter: match the harness name in its script path. diff --git a/bin/fm-session-lock-lib.sh b/bin/fm-session-lock-lib.sh index 73aab2f2136..90303cda1c9 100644 --- a/bin/fm-session-lock-lib.sh +++ b/bin/fm-session-lock-lib.sh @@ -9,7 +9,7 @@ # This file is sourced by scripts and has no side effects on source. # Known harness command names; extend when a new adapter is verified. -FM_HARNESS_RE='claude|codex|opencode|grok|kimi|^pi$' +FM_HARNESS_RE='claude|codex|opencode|grok|kimi|^pi$|^pi-signed$' # Walk the current process ancestry (up to 8 hops) and print the first pid whose # command looks like a verified harness. The harness pid lives as long as the @@ -34,10 +34,19 @@ fm_harness_ancestry_pid() { # True if $1 is a live process that looks like a verified harness. fm_harness_pid_alive() { - local pid=$1 comm + local pid=$1 comm args kill -0 "$pid" 2>/dev/null || return 1 comm=$(ps -o comm= -p "$pid" 2>/dev/null) || return 1 - printf '%s' "$(basename "$comm") $(ps -o args= -p "$pid" 2>/dev/null)" | grep -qE "$FM_HARNESS_RE" + if printf '%s' "$(basename "$comm")" | grep -qE "$FM_HARNESS_RE"; then + return 0 + fi + case "$comm" in + *node*|*python*) + args=$(ps -o args= -p "$pid" 2>/dev/null) + printf '%s' "$args" | grep -qE "$FM_HARNESS_RE" + ;; + *) return 1 ;; + esac } # True when state dir $1 holds a session lock whose pid is the harness ancestor diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index 00a2262ef4d..fe659b54b6c 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -59,10 +59,11 @@ # profile consultation. A --secondmate spawn is exempt and resolves the SECONDMATE # harness (config/secondmate-harness -> config/crew-harness -> own), so the # secondmate-vs-crewmate split is DURABLE across every respawn (recovery, -# /updatefirstmate, restart). A bare adapter name (claude|codex|opencode|pi|grok|kimi) +# /updatefirstmate, restart). A bare adapter name (claude|codex|opencode|pi|pi-signed|grok|kimi) # overrides it for this spawn (either kind). A non-flag string containing # whitespace is treated as a RAW launch command - the escape hatch for verifying -# new adapters. +# new adapters. pi-signed launches that exact executable name from PATH and +# refuses before endpoint creation when it is unavailable; it never falls back to pi. # config/secondmate-harness may also carry an optional model and effort as extra # whitespace-separated tokens ("<harness> [<model>] [<effort>]"). For a # --secondmate spawn, those tokens apply only when this spawn also resolves its @@ -388,7 +389,7 @@ FIRSTMATE_HOME= if [ "$KIND" = secondmate ]; then case "${POS[1]:-}" in - ''|claude|codex|opencode|pi|grok|kimi) + ''|claude|codex|opencode|pi|pi-signed|grok|kimi) ARG3=${POS[1]:-} ;; *' '*) @@ -434,11 +435,11 @@ launch_template() { fi ;; opencode) printf '%s' 'OPENCODE_CONFIG_CONTENT='\''{"permission":{"*":"allow"}}'\'' opencode __MODELFLAG__--prompt "$(__OPINPUT__ encode launch-brief < __BRIEF__)"' ;; - pi) + pi|pi-signed) if [ "$kind" = secondmate ]; then - printf '%s' 'pi __MODELFLAG____EFFORTFLAG__-e __PITURNEND__ -e __PIWATCH__ "$(__OPINPUT__ encode launch-brief < __BRIEF__)"' + printf '%s%s' "$harness" ' __MODELFLAG____EFFORTFLAG__-e __PITURNEND__ -e __PIWATCH__ "$(__OPINPUT__ encode launch-brief < __BRIEF__)"' else - printf '%s' 'pi __MODELFLAG____EFFORTFLAG__-e __PIEXT__ "$(__OPINPUT__ encode launch-brief < __BRIEF__)"' + printf '%s%s' "$harness" ' __MODELFLAG____EFFORTFLAG__-e __PIEXT__ "$(__OPINPUT__ encode launch-brief < __BRIEF__)"' fi ;; # grok (Grok Build TUI): a positional prompt starts the supervised interactive @@ -494,6 +495,18 @@ case "$ARG3" in ;; esac +case "$HARNESS" in + pi|pi-signed) LAUNCH="FM_PI_HARNESS=$HARNESS $LAUNCH" ;; +esac + +# pi-signed is an explicitly selected executable identity, not an alias that may +# silently fall back to pi. Resolve it from PATH before creating an endpoint and +# retain the literal name in the launch command and task metadata. +if [ "$HARNESS" = pi-signed ] && ! command -v pi-signed >/dev/null 2>&1; then + echo "error: pi-signed executable not found on PATH; install the signed Pi wrapper or select a different verified harness" >&2 + exit 1 +fi + # config/secondmate-harness may carry optional model/effort tokens alongside the # harness ("<harness> [<model>] [<effort>]"). They apply only when this is a # --secondmate spawn and no explicit per-spawn harness/raw launch was supplied, so @@ -565,7 +578,7 @@ model_flag_for_harness() { local harness=$1 model=$2 [ -n "$model" ] && [ "$model" != default ] || return 0 case "$harness" in - claude|codex|opencode|pi|grok|kimi) + claude|codex|opencode|pi|pi-signed|grok|kimi) printf -- '--model %s ' "$(shell_quote "$model")" ;; esac @@ -597,7 +610,7 @@ effort_flag_for_harness() { low|medium|high) printf -- '--reasoning-effort %s ' "$(shell_quote "$effort")" ;; esac ;; - pi) + pi|pi-signed) # Pi 0.80.6 accepts the full shared effort vocabulary, including max, through # its --thinking flag. case "$effort" in @@ -1317,7 +1330,7 @@ export const FmTurnEnd = async ({ \$ }) => ({ EOF exclude_path '.opencode/plugins/fm-turn-end.js' ;; - pi*) + pi|pi-signed) # Written OUTSIDE the worktree: pi's project-trust gate fires on any extension # loaded from inside the project (verified live), but an explicit -e path # elsewhere loads without a dialog. Lives in state/, cleaned by teardown. diff --git a/bin/fm-tmux-lib.sh b/bin/fm-tmux-lib.sh index 7eddc323f33..cf8c3f7fa5e 100755 --- a/bin/fm-tmux-lib.sh +++ b/bin/fm-tmux-lib.sh @@ -94,7 +94,7 @@ fm_busy_lines_match() { # [harness] claude) regex=$FM_TMUX_CLAUDE_BUSY_REGEX_DEFAULT ;; codex) regex=$FM_TMUX_CODEX_BUSY_REGEX_DEFAULT ;; opencode) regex=$FM_TMUX_OPENCODE_BUSY_REGEX_DEFAULT ;; - pi) regex=$FM_TMUX_PI_BUSY_REGEX_DEFAULT ;; + pi|pi-signed) regex=$FM_TMUX_PI_BUSY_REGEX_DEFAULT ;; grok) regex=$FM_TMUX_GROK_BUSY_REGEX_DEFAULT ;; kimi) regex=$FM_TMUX_KIMI_BUSY_REGEX_DEFAULT ;; '') regex=$FM_TMUX_BUSY_REGEX_DEFAULT ;; diff --git a/docs/architecture.md b/docs/architecture.md index d1bbb1c1ff5..4840ef77a02 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -54,14 +54,14 @@ The default path remains local-only; live GitHub enrichment exists only behind t Optional X mode integrates with the watcher only after explicit opt-in; [configuration.md](configuration.md#x-mode-env) owns its generated-artifact and dispatch mechanics. At session start, `bin/fm-session-start.sh` emits exactly one primary-harness supervision block rendered by `bin/fm-supervision-instructions.sh` from `docs/supervision-protocols/`. -That block owns the live wait shape for the running primary harness: Claude's Stop `asyncRewake` hook owns tokenless re-arm cycles, Grok uses background-notify cycles, Codex uses bounded foreground checkpoints, Pi uses its two tracked primary extensions, and OpenCode uses its TUI plugin. +That block owns the live wait shape for the running primary harness: Claude's Stop `asyncRewake` hook owns tokenless re-arm cycles, Grok uses background-notify cycles, Codex uses bounded foreground checkpoints, Pi and pi-signed use the same two tracked primary extensions, and OpenCode uses its TUI plugin. `bin/fm-watch-arm.sh` remains the verified arm wrapper for protocols that call it; it forks the watcher as a tracked child, verifies it is genuinely alive with a fresh liveness beacon, and prints an honest `started`, `attached`, or nonzero `FAILED` status. On `attached` it stays live across identity-matched successors, and an unexplained clean child close either attaches to a verified healthy successor or becomes the typed nonzero `watcher: FAILED - cycle ended without an actionable reason` result. The arm layer records one bounded lifecycle row per observed cycle in `state/.watch-cycle-exits.log`; `state/.watch-triage.log` remains exclusively the absorbed-wake debug log. Pi and OpenCode verify session-lock ownership and launch one singleton successor from their child-close handlers before delivering an actionable wake prompt, with bounded exponential retry for failed restoration. Claude's `bin/fm-claude-stop-autoarm.sh` hook fires on every Stop and, when the home is eligible and still needs supervision, claims one home-scoped cycle, foregrounds the arm wrapper, and translates an actionable close or typed failure into one exit-2 rewake. [`watcher-continuity.md`](watcher-continuity.md) owns Claude's residual active-turn coverage and watcher-status command-gating boundary. -The existing turn-end guard remains the final backstop for all five harness protocols, cooperating with the auto-arm claim in its `--claude` mode. +The existing turn-end guard remains the final backstop for all five harness-engine protocols, with pi-signed sharing Pi's protocol and the `--claude` mode cooperating with the auto-arm claim. Its `--restart` mode signals only the watcher recorded in the current home's `state/.watch.lock`, so restarting one home cannot kill sibling secondmate watchers. A pull-based guard (`bin/fm-guard.sh`) warns through supervision tool output if the primary checkout is tangled, or if tasks are in flight and that watcher stops running or queued wakes are waiting to be drained. The drain script calls that guard after emptying the queue, which avoids repeating the queued-wakes warning for records it just consumed while still warning on stale watcher liveness. diff --git a/docs/arm-pretool-check.md b/docs/arm-pretool-check.md index c56f555f7e6..f4747e0abdc 100644 --- a/docs/arm-pretool-check.md +++ b/docs/arm-pretool-check.md @@ -24,7 +24,7 @@ It tokenizes the bytes and classifies lexical execution positions only. - Stdin JSON at `.tool_input.command` for Claude and Codex. - Stdin JSON at `.toolInput.command` for Grok. -- `--command <exact string>` for OpenCode and Pi. +- `--command <exact string>` for OpenCode, Pi, and pi-signed. - `--background` as a compatibility-only field that never changes the decision. - `--claude` to preserve Claude's stderr-only deny requirement. @@ -151,7 +151,7 @@ Prose may improve without changing adapter behavior. - `--claude` suppresses stdout completely because Claude ignores a PreToolUse deny when stdout is nonempty. - Codex blocks on exit 2 and displays stderr. - OpenCode throws only when the checker exits 2. -- Pi returns `{block: true}` only when the checker exits 2. +- Pi and pi-signed return `{block: true}` only when the checker exits 2. ## Harness wiring @@ -161,7 +161,7 @@ Prose may improve without changing adapter behavior. | Claude | `.tool_input.command` | `.claude/settings.json` forwards stdin with `--claude`, leaving stdout empty and returning the stderr deny object. | | Grok | `.toolInput.command` | `.grok/hooks/fm-primary-pretool-check.json` forwards stdin and Grok consumes the stdout `decision=deny` object. | | OpenCode | `output.args.command` | `.opencode/plugins/fm-primary-pretool-check.js` passes one `--command` argument and throws only for exit 2. | -| Pi | `event.input.command` | `.pi/extensions/fm-primary-turnend-guard.ts` passes one `--command` argument and returns `{block: true}` only for exit 2. | +| Pi / pi-signed | `event.input.command` | `.pi/extensions/fm-primary-turnend-guard.ts` passes one `--command` argument and returns `{block: true}` only for exit 2. | Grok project hooks require folder trust. Every shell variable reference in a Grok hook command must carry an inline default such as `${GROK_WORKSPACE_ROOT:-}` because Grok expands the raw hook command before `bash -lc` runs it. diff --git a/docs/configuration.md b/docs/configuration.md index 896b98ae31a..fed683343ea 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -174,15 +174,17 @@ The full cmux home label also includes a short hash of the resolved `FM_ROOT` pa ## Harness support -claude, codex, opencode, pi, grok, and kimi are empirically verified for crewmate and secondmate launches; [README requirements](../README.md#requirements) own the narrower set supported for the primary session. +claude, codex, opencode, pi, pi-signed, grok, and kimi are empirically verified for crewmate and secondmate launches; [README requirements](../README.md#requirements) own the set supported for the primary session. New harnesses get verified through a supervised trial task before joining the set. The verified adapter knowledge - busy signatures, interrupt and exit commands, skill-invocation syntax, and per-harness quirks - lives in [`.agents/skills/harness-adapters/SKILL.md`](../.agents/skills/harness-adapters/SKILL.md). Launch mechanics, including the verified command templates, live in [`bin/fm-spawn.sh`](../bin/fm-spawn.sh). Enabled primary-session turn-end guard integrations are tracked as repo-level hook files and documented in [`docs/turnend-guard.md`](turnend-guard.md). Kimi remains outside the primary turn-end guard integrations; [`docs/turnend-guard.md`](turnend-guard.md#compatibility-limits) owns its separate captain-approved crew wake hook. Primary-session watcher wake protocols are rendered at session start by [`bin/fm-supervision-instructions.sh`](../bin/fm-supervision-instructions.sh) from [`docs/supervision-protocols/`](supervision-protocols/). -Claude's Stop `asyncRewake` hook owns tokenless re-arm cycles, Grok uses background-notify cycles, Codex uses bounded foreground checkpoints, Pi uses its two tracked primary extensions, and OpenCode uses its TUI plugin. +Claude's Stop `asyncRewake` hook owns tokenless re-arm cycles, Grok uses background-notify cycles, Codex uses bounded foreground checkpoints, Pi and pi-signed use the same two tracked primary extensions, and OpenCode uses its TUI plugin. `config/crew-harness` is a local, gitignored file containing one adapter name for crewmate and scout launches. +When pi-signed is selected, Firstmate launches the executable named `pi-signed` from `PATH` with `FM_PI_HARNESS=pi-signed` and refuses the launch if it is unavailable rather than falling back to pi. +Plain Pi launches set `FM_PI_HARNESS=pi`, so a signed primary's environment cannot relabel a plain Pi worker. When it is absent or contains `default`, crewmates mirror the firstmate's own harness. `config/secondmate-harness` is a separate local, gitignored file containing the adapter the primary uses to launch secondmate agents, optionally followed by model and effort tokens on the same line. The first non-empty, non-comment line is parsed as `<harness> [<model>] [<effort>]`. @@ -200,7 +202,7 @@ For Kimi crews, `fm-spawn.sh` runs `fm-kimi-turnend-hook.sh install`, drops a pe Kimi continues to use the captain's normal Kimi home, including the existing config, skills, and memory; Firstmate does not create an isolated Kimi home. The Kimi installer requires an existing regular non-symlink `~/.kimi-code/config.toml`, `python3` with `tomllib`, and `jq`; it validates but never serializes the captain's TOML and refuses before writing when the config is missing, malformed, or surprising or when either tool requirement is unavailable. Its `remove` action excises only the marker-delimited Firstmate region and removes Firstmate's hook files. -For Pi secondmate launches, `fm-spawn.sh` starts Pi with `-e` pointed at the secondmate home's own tracked `.pi/extensions/fm-primary-pi-watch.ts` and `.pi/extensions/fm-primary-turnend-guard.ts`, both already present from the secondmate home's git worktree. +For Pi and pi-signed secondmate launches, `fm-spawn.sh` starts the selected executable with `-e` pointed at the secondmate home's own tracked `.pi/extensions/fm-primary-pi-watch.ts` and `.pi/extensions/fm-primary-turnend-guard.ts`, both already present from the secondmate home's git worktree. ## Crew dispatch profiles (config/crew-dispatch.json) diff --git a/docs/sessionstart-nudge.md b/docs/sessionstart-nudge.md index 1f0ee079f47..ef21cea1323 100644 --- a/docs/sessionstart-nudge.md +++ b/docs/sessionstart-nudge.md @@ -23,7 +23,7 @@ Every path exits 0, including malformed state and adapter errors, because a Clau | Claude | `.claude/settings.json` registers `SessionStart` for `startup`, `resume`, and `clear`, excludes `compact`, and invokes the wrapper through `CLAUDE_PROJECT_DIR`. | Native stdout context injection is supported. | | Codex | `.codex/hooks.json` anchors to the hook process working directory, verifies a Firstmate-shaped hook-bearing root, and executes the wrapper. | Native stdout context injection is supported. | | OpenCode | `.opencode/plugins/fm-primary-sessionstart-nudge.js` listens for `session.created`, runs once per session id, and calls `client.session.promptAsync` only when the wrapper prints a nudge. | Interactive TUI delivery is supported; headless `opencode run` is intentionally fail-open because the process can exit before the queued turn. | -| Pi | `.pi/extensions/fm-primary-turnend-guard.ts` handles `session_start` reasons `startup`, `new`, and `resume`, then injects the wrapper output with `pi.sendMessage`. | The custom message reaches model context without racing an initial positional prompt. | +| Pi / pi-signed | `.pi/extensions/fm-primary-turnend-guard.ts` handles `session_start` reasons `startup`, `new`, and `resume`, then injects the wrapper output with `pi.sendMessage`. | The custom message reaches model context without racing an initial positional prompt. | | Grok | `.grok/hooks/fm-primary-sessionstart-nudge.json` registers a project `SessionStart` hook and invokes the wrapper through inline-defaulted `${GROK_WORKSPACE_ROOT:-}`. | The project hook runs when the checkout is trusted, but Grok currently discards hook stdout from model context, so this path is intentionally fail-open. | The OpenCode nudge runs only on `session.created`. @@ -36,7 +36,7 @@ That alternative expands trust and writes outside this repository, so Firstmate `tests/fm-sessionstart-nudge.test.sh` proves wrapper silence for both gate signals, an unmarked linked worktree, a missing state directory, and an already-owned lock. It proves exact U+2063 `FIRSTMATE_OP:`-prefixed, `session-start`-typed one-line output for a plain primary and a marked linked secondmate primary. -It also verifies tracked wrapper registration for Claude, Codex, OpenCode, Pi, and Grok. +It also verifies every tracked transport registration listed above. `tests/fm-captain-translation-contract.test.sh` proves Ahoy's current marker rule, narrow legacy compatibility exclusions, genuine captain-message near misses, and the shared marker on supported user-role operational injections. `tests/fm-pi-primary-live-e2e.test.sh` and `tests/fm-opencode-primary-live-e2e.test.sh` exercise native startup paths with first-message and later-message Ahoy regressions. `tests/fm-turnend-guard.test.sh`, `tests/fm-pi-watch-extension.test.sh`, and `tests/fm-daemon.test.sh` cover marked guard, monitoring, and away-mode delivery. diff --git a/docs/supervision-protocols/pi.md b/docs/supervision-protocols/pi.md index 2316428a833..30c4aae35fc 100644 --- a/docs/supervision-protocols/pi.md +++ b/docs/supervision-protocols/pi.md @@ -8,14 +8,12 @@ When this session owns supervision and away mode is not active: Never run `bin/fm-watch-arm.sh` through Pi's bash tool because that foreground arm can wedge the agent and bypasses extension-owned cleanup. 4. If the extension says no live session holds the lock, run `bin/fm-session-start.sh` to reclaim the session lock, then call `fm_watch_arm_pi` again. 5. The extension starts `bin/fm-watch-arm.sh --restart`, keeps the child attached to the live Pi process, and owns every later successor launch. -6. Ordinary same-process session replacement (`/new`, `/resume`, `/fork`, reload) retires only the prior generation; call `fm_watch_arm_pi` once for the first cycle of the replacement session without restarting Pi. - The generation-owner contract lives in `.pi/extensions/fm-primary-pi-watch.ts`. -7. After an actionable child close, the extension rechecks session-lock ownership and verifies one successor before it delivers the follow-up wake; its bounded fallback is defined in `docs/watcher-continuity.md`. -8. Ordinary work, turn completion, and ordinary signal, stale, check, heartbeat, or other wake handling: do not call `fm_watch_arm_pi` again because continuity is extension-owned rather than model-memory-owned. -9. An unexpected child close enters bounded exponential retry, and an exhausted retry or lost session lock is surfaced as a watcher failure instead of disappearing. -10. Missing, failed, or unhealthy cycle only: if a later notification explicitly reports one of those repair conditions, drain queued wakes, inspect the failure text, call `fm_watch_arm_pi`, and restart the selected Pi-family executable with both extensions loaded if needed. +6. After an actionable child close, the extension rechecks session-lock ownership and verifies one successor before it delivers the follow-up wake; its bounded fallback is defined in `docs/watcher-continuity.md`. +7. Ordinary work, turn completion, and ordinary signal, stale, check, heartbeat, or other wake handling: do not call `fm_watch_arm_pi` again because continuity is extension-owned rather than model-memory-owned. +8. An unexpected child close enters bounded exponential retry, and an exhausted retry or lost session lock is surfaced as a watcher failure instead of disappearing. +9. Missing, failed, or unhealthy cycle only: if a later notification explicitly reports one of those repair conditions, drain queued wakes, inspect the failure text, call `fm_watch_arm_pi`, and restart the selected Pi-family executable with both extensions loaded if needed. A redundant call while the extension owns an arm child or scheduled retry is an ownership-based `watcher: unchanged` no-op, not an independent health claim. -11. Never use shell `&` for watcher supervision. +10. Never use shell `&` for watcher supervision. The arm mechanism above is extension-owned, not a model tool call, but a manual recovery probe that backgrounds, pipes, or bundles the arm is denied automatically by the PreToolUse seatbelt (`bin/fm-arm-pretool-check.sh`, wired into the turn-end guard extension at `__FM_PI_TURNEND_EXT__`). The turn-end guard extension lives at `__FM_PI_TURNEND_EXT__`. diff --git a/docs/tmux-backend.md b/docs/tmux-backend.md index ae24507c588..3bf20fe9d1a 100644 --- a/docs/tmux-backend.md +++ b/docs/tmux-backend.md @@ -46,12 +46,11 @@ Verify setup by spawning a small task and confirming its `fm-<id>` window appear A target-existence check proves only that the pane exists. The deeper tmux agent-liveness probe first verifies exact window membership, then reads `#{pane_current_command}` to distinguish a running harness process from a bare idle shell. -It classifies recognized Claude, Codex, OpenCode, Grok, and Kimi process names as `alive`, common shells as `dead`, an authoritatively absent window as `missing`, unreadable state as `unreadable`, and every other process as `ambiguous`. +It classifies recognized Claude, Codex, OpenCode, Pi, pi-signed, Grok, and Kimi process names as `alive`, common shells as `dead`, an authoritatively absent window as `missing`, unreadable state as `unreadable`, and every other process as `ambiguous`. Only `dead` and `missing` authorize recovery because a false dead result could launch a duplicate agent. -Pi runs through a generic `node` process name and cannot be attributed confidently from the tmux foreground-process field. -An existing Pi pane is therefore reported as ambiguous rather than auto-healed, while an authoritatively missing Pi window can be relaunched safely. -This is the active tmux liveness limitation. +The verified Pi Launcher path reports the exact foreground command `pi-launcher` for both pi and pi-signed, while direct executable identities `pi`, `pi-signed`, and `Pi` remain accepted exactly. +Similar or prefixed process names are not accepted through those exact Pi-family entries. Agent liveness and composer safety are separate checks. For a bordered composer, the tmux reader locates the complete box structurally and classifies every content row through the shared ANSI and ghost handling in `bin/fm-composer-lib.sh`. @@ -79,7 +78,6 @@ Ambiguous pending text never receives the busy-queue conversion. ## Limits and regression entry points - tmux is the reference path and supports secondmate homes. -- Existing Pi agent-process liveness is inconclusive, while an authoritatively missing Pi window can trigger recovery. - The OpenCode busy-queue exception is tmux-specific; Herdr retains its separately documented gap. ```sh diff --git a/docs/turnend-guard.md b/docs/turnend-guard.md index 5589ea23852..30690bb887e 100644 --- a/docs/turnend-guard.md +++ b/docs/turnend-guard.md @@ -55,7 +55,7 @@ The Claude mode waits up to `FM_CLAUDE_AUTOARM_SYNC_WAIT_MS` (default 800 millis When none of those proofs appears, it re-blocks up to `FM_CLAUDE_TURNEND_BLOCK_BUDGET` times (default 3, below Claude's 8-block override), then allows degraded with a visible `systemMessage`. Any allow resets the budget. -OpenCode, Pi, and Grok expose passive callbacks for this purpose. +OpenCode, Pi, pi-signed, and Grok expose passive callbacks for this purpose. Their adapters fail open at the hook boundary to protect the user session but schedule one bounded follow-up when the predicate blocks. The generated prompts use the canonical `turn-end-guard` kind after the U+2063 `FIRSTMATE_OP: ` prefix, so Ahoy does not treat them as captain messages. Each adapter owns a loop latch. @@ -70,7 +70,7 @@ That warning uses `bin/fm-supervision-instructions.sh --repair-line`, so it alwa - Child crewmate and scout worktrees are outside scope. - A valid secondmate home is in scope; an idle secondmate endpoint with no X-mode relay poll remains healthy because it has no supervision need. -- Claude and Codex block directly, while OpenCode, Pi, and Grok use bounded passive follow-ups. +- The direct-blocking and bounded passive-follow-up split is limited to the primary integrations listed above. - OpenCode headless mode and untrusted Grok project hooks remain fail-open at the host boundary. - Kimi Code CLI 0.29.1 exposes only global `[[hooks]]` configuration in `~/.kimi-code/config.toml`, including a `Stop` event with snake_case payload fields `hook_event_name`, `session_id`, `cwd`, and `stop_hook_active`. - Kimi has no project-level hook configuration and remains outside the primary guard integrations above. @@ -85,6 +85,6 @@ That warning uses `bin/fm-supervision-instructions.sh --repair-line`, so it alwa `tests/fm-turnend-guard.test.sh` covers the predicate, main and secondmate primary scope, child-worktree exclusion, `FM_HOME` and `FM_STATE_OVERRIDE` precedence, the cooperative `--claude` claim wait, epoch allow, re-block budget, Pi logical-run latching, missing-`jq` behavior, all five primary registrations, and Grok resume permission and recursion safety. `tests/fm-kimi-harness.test.sh` covers the separate Kimi crew hook's format preservation, idempotence, refusal cases, token guard, spawn registration, and teardown cleanup. -`tests/fm-supervision-instructions.test.sh` covers recovery-line ownership. +`tests/fm-supervision-instructions.test.sh` covers recovery-line ownership and pi-signed's identity-preserving reuse of Pi's protocol. `FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh` is the opt-in isolated Pi path. [`verification/supervision.md`](verification/supervision.md#turn-end-guard) records the active cross-harness empirical evidence, including the 2026-07-24 Claude `asyncRewake` revalidation. diff --git a/docs/verification/runtime-backends.md b/docs/verification/runtime-backends.md index 0c154c97603..a711d84ee5b 100644 --- a/docs/verification/runtime-backends.md +++ b/docs/verification/runtime-backends.md @@ -30,7 +30,53 @@ zsh A persistent parent shell waiting for a child remained reported as the parent process, while a shell that directly execed a simple command changed identity with the process itself. Claude, Codex, OpenCode, and Grok were observed under their own process names. Kimi Code CLI 0.29.1 was observed under `kimi` on 2026-07-25. -Pi remained a generic `node` process and is intentionally inconclusive. +Pi and pi-signed 0.82.0 were reverified on 2026-07-27 through real isolated `fm-spawn.sh` launches. + +Installed-wrapper checks: + +```sh +basename "$(command -v pi-signed)" +pi-signed --version +pi --version +``` + +Observed bounded output: + +```text +pi-signed +0.82.0 +0.82.0 +``` + +The isolated process and endpoint checks used: + +```sh +tmux display-message -p -t "$target" '#{pane_current_command}' +ps -o comm= -p "$wrapper_pid" +ps -o comm= -p "$engine_pid" +FM_HOME="$fixture_home" bin/fm-crew-state.sh "$task_id" +``` + +Observed bounded shapes: + +```text +pi-launcher +.../pi-signed +.../Pi Launcher.app/Contents/Resources/pi/pi +state: done ... +``` + +Both launches executed a submitted tool instruction and touched the generated `turn_end` marker. +The pi-signed launch retained `harness=pi-signed`, while the plain comparison retained `harness=pi`. +The exact wrapper ancestry was `pi-signed` parent to Pi engine child, and the plain Pi Launcher path also traversed the signed wrapper on this installation. +That shared plain-Pi path is retained as disconfirming evidence against using ancestry as runtime-selection authority. +Firstmate therefore sets the exact `FM_PI_HARNESS` selection marker on both worker launch paths, while an unmarked Pi-family process remains `pi`. +Both recorded runtime identities now classify the exact `pi-launcher` foreground command as `alive`. + +Backend applicability was reviewed across every spawn adapter. +Tmux needs the exact `pi-launcher`, `pi-signed`, `pi`, and `Pi` process identities for recovery-grade liveness. +Herdr uses native registered-agent state and needs no process-name branch. +Zellij has no verified recovery-grade agent process probe, while Orca and cmux do not support secondmate spawns, so those three retain their existing generic ordinary-launch semantics without a new liveness matcher. The structural multi-row composer reader, Kimi pointer-delivery path, and OpenCode 1.18.4 busy-queue behavior are pinned by: diff --git a/docs/verification/supervision.md b/docs/verification/supervision.md index 4063f5566dc..30a98311a40 100644 --- a/docs/verification/supervision.md +++ b/docs/verification/supervision.md @@ -45,6 +45,8 @@ pi -p -e .pi/extensions/fm-primary-turnend-guard.ts \ Observed result: `PI_SMOKE_DONE`, with one session-start execution. The earlier `sendUserMessage` counterfactual raced the positional prompt; the current non-triggering `pi.sendMessage` custom message did not. +The installed pi-signed 0.82.0 wrapper repeated the Pi primary extension and session-start path on 2026-07-27. +[`runtime-backends.md`](runtime-backends.md#tmux) owns the shared-ancestry evidence and authoritative selection-marker boundary. Current deterministic and live entry points: diff --git a/tests/fm-composer-ghost.test.sh b/tests/fm-composer-ghost.test.sh index d0285528894..7249574ea3e 100755 --- a/tests/fm-composer-ghost.test.sh +++ b/tests/fm-composer-ghost.test.sh @@ -473,12 +473,12 @@ test_all_tmux_harness_composers_share_classification() { dir="$TMP_ROOT/all-harness-composers"; mkdir -p "$dir" fb=$(make_fake_tmux "$dir") capture="$dir/styled.txt" - for harness in claude codex opencode pi grok; do + for harness in claude codex opencode pi pi-signed grok; do case "$harness" in claude) printf '╭────────────╮\n│ ❯ \033[2mtry\033[0m │\n╰────────────╯\n' > "$capture" ;; codex) printf '╭────────────╮\n│ › \033[2mtip\033[0m │\n╰────────────╯\n' > "$capture" ;; opencode) printf '╭────────────╮\n│ > │\n╰────────────╯\n' > "$capture" ;; - pi) printf '╭────────────╮\n│ │\n╰────────────╯\n' > "$capture" ;; + pi|pi-signed) printf '╭────────────╮\n│ │\n╰────────────╯\n' > "$capture" ;; grok) printf '╭────────────╮\n│ ❯ \033[38;2;50;47;70mType\033[0m │\n╰────────────╯\n' > "$capture" ;; esac out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=1 \ @@ -488,7 +488,7 @@ test_all_tmux_harness_composers_share_classification() { case "$harness" in claude|grok) printf '╭────────────╮\n│ ❯ fix │\n╰────────────╯\n' > "$capture" ;; codex) printf '╭────────────╮\n│ › fix │\n╰────────────╯\n' > "$capture" ;; - opencode|pi) printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$capture" ;; + opencode|pi|pi-signed) printf '╭────────────╮\n│ > fix │\n╰────────────╯\n' > "$capture" ;; esac out=$(PATH="$fb:$PATH" FM_FAKE_STYLED="$capture" FM_FAKE_CY=1 \ fm_tmux_composer_state "fakepane") diff --git a/tests/fm-instruction-owners.test.sh b/tests/fm-instruction-owners.test.sh index 5cb26268f7c..f55f3e905f9 100755 --- a/tests/fm-instruction-owners.test.sh +++ b/tests/fm-instruction-owners.test.sh @@ -123,7 +123,7 @@ test_agent_owned_quota_array_dispatch_contract() { '| claude | Open the current interactive session' \ '| codex | Open the current interactive session' \ '| opencode | Run `opencode models [provider]`' \ - '| pi | Run `pi --list-models [search]`' \ + '| pi / pi-signed | Run the selected executable as `<executable> --list-models [search]`' \ '| grok | Run `grok models`' \ "For an unfamiliar harness or model namespace, establish support and provider identity from that harness's authoritative CLI help, model listing, or current documentation rather than guessing" \ 'If those sources do not establish the relationship needed for dispatch, fail loudly and report the unresolved candidate.'; do diff --git a/tests/fm-kimi-harness.test.sh b/tests/fm-kimi-harness.test.sh index 9e0f450439b..8ac5922ec51 100755 --- a/tests/fm-kimi-harness.test.sh +++ b/tests/fm-kimi-harness.test.sh @@ -24,8 +24,8 @@ test_existing_launch_templates_are_byte_pinned() { assert_source_line " printf '%s' 'codex __MODELFLAG____EFFORTFLAG__--dangerously-bypass-approvals-and-sandbox \"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"'" assert_source_line " printf '%s' 'codex __MODELFLAG____EFFORTFLAG__--dangerously-bypass-approvals-and-sandbox -c \"notify=[\\\"bash\\\",\\\"-c\\\",\\\"touch __TURNEND__\\\"]\" \"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"'" assert_source_line " opencode) printf '%s' 'OPENCODE_CONFIG_CONTENT='\\''{\"permission\":{\"*\":\"allow\"}}'\\'' opencode __MODELFLAG__--prompt \"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"' ;;" - assert_source_line " printf '%s' 'pi __MODELFLAG____EFFORTFLAG__-e __PITURNEND__ -e __PIWATCH__ \"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"'" - assert_source_line " printf '%s' 'pi __MODELFLAG____EFFORTFLAG__-e __PIEXT__ \"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"'" + assert_source_line " printf '%s%s' \"\$harness\" ' __MODELFLAG____EFFORTFLAG__-e __PITURNEND__ -e __PIWATCH__ \"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"'" + assert_source_line " printf '%s%s' \"\$harness\" ' __MODELFLAG____EFFORTFLAG__-e __PIEXT__ \"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"'" assert_source_line " grok) printf '%s' 'grok --always-approve __MODELFLAG____EFFORTFLAG__\"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"' ;;" pass "fm-spawn: the five pre-existing adapters' launch templates stay byte-pinned" } diff --git a/tests/fm-secondmate-harness.test.sh b/tests/fm-secondmate-harness.test.sh index 446dcd93a17..39ca2021bf1 100755 --- a/tests/fm-secondmate-harness.test.sh +++ b/tests/fm-secondmate-harness.test.sh @@ -14,13 +14,11 @@ # explicit per-spawn harness arg still wins. # B) Inheritance. The primary pushes a declared, extensible set of LOCAL # (gitignored) config items - config/crew-dispatch.json, config/crew-harness, -# config/backlog-backend, config/backend, config/herdr-presentation-spaces, and -# config/startup-memory-budget - -# down into each secondmate home's config/, so the secondmate's OWN crewmates, -# dispatch profiles, backlog backend, runtime-backend default, and Herdr -# presentation opt-in inherit the primary's settings. It is primary-authoritative -# (re-pushed at secondmate spawn, on the bootstrap secondmate sweep, and by -# config push). +# config/backlog-backend, and config/herdr-presentation-spaces - down into +# each secondmate home's config/, so the secondmate's OWN crewmates, +# dispatch profiles, backlog backend, and Herdr presentation opt-in inherit +# the primary's settings. It is primary-authoritative (re-pushed at +# secondmate spawn, on the bootstrap secondmate sweep, and by config push). # config/secondmate-harness is deliberately NOT inherited (secondmates do # not spawn secondmates). After a successful push that changes allowlisted # config under an already-running home, a literal-content reread instruction @@ -168,7 +166,7 @@ SH [ "$got" = pi ] || fail "selected plain Pi resolved '$got', expected pi" got=$(PATH="$fakebin:$BASE_PATH" PI_CODING_AGENT=true FM_PI_HARNESS=pi-signed-helper "$ROOT/bin/fm-harness.sh") [ "$got" = pi ] || fail "inexact signed selection marker resolved '$got', expected pi" - got=$(env -u PI_CODING_AGENT PATH="$fakebin:$BASE_PATH" FM_PI_HARNESS=pi-signed "$ROOT/bin/fm-harness.sh") + got=$(PATH="$fakebin:$BASE_PATH" FM_PI_HARNESS=pi-signed "$ROOT/bin/fm-harness.sh") [ "$got" = pi ] || fail "signed selection marker without Pi's family marker resolved '$got', expected pi" got=$(PATH="$fakebin:$BASE_PATH" PI_CODING_AGENT=true FM_TEST_SIGNED_SHAPE=plain "$ROOT/bin/fm-harness.sh") [ "$got" = pi ] || fail "plain Pi marker resolved '$got', expected pi" @@ -189,73 +187,20 @@ SH pass "pi-signed identity: authoritative launch selection distinguishes shared wrapper ancestry" } -test_dash_leading_process_names_are_basename_operands() { - local dir fakebin got err status - dir="$TMP_ROOT/dash-leading-process-names" - fakebin=$(fm_fakebin "$dir") - cat > "$fakebin/ps" <<'SH' -#!/usr/bin/env bash -set -u -field= pid= -while [ "$#" -gt 0 ]; do - case "$1" in - -o) field=$2; shift 2 ;; - -p) pid=$2; shift 2 ;; - *) shift ;; - esac -done -case "$pid:$field" in - 4242:comm=) printf '%s\n' '/opt/test/bin/codex' ;; - 4242:args=) printf '%s\n' 'codex' ;; - 4242:ppid=) printf '%s\n' 1 ;; - 5252:comm=) printf '%s\n' '-codex' ;; - 5252:args=) printf '%s\n' '-codex' ;; - 5252:ppid=) printf '%s\n' 1 ;; - *:comm=) printf '%s\n' '-zsh' ;; - *:args=) printf '%s\n' '-zsh' ;; - *:ppid=) printf '%s\n' 4242 ;; -esac -SH - chmod +x "$fakebin/ps" - - err="$dir/fm-harness.err" - got=$(env -u CLAUDECODE -u PI_CODING_AGENT -u GROK_AGENT \ - PATH="$fakebin:$BASE_PATH" "$ROOT/bin/fm-harness.sh" 2>"$err") - [ "$got" = codex ] || fail "dash-leading shell ancestry resolved '$got', expected codex" - [ ! -s "$err" ] || fail "fm-harness wrote basename option noise for literal -zsh: $(cat "$err")" - - err="$dir/fm-session-lock-ancestry.err" - got=$(PATH="$fakebin:$BASE_PATH" bash -c \ - '. "$0/bin/fm-session-lock-lib.sh"; fm_harness_ancestry_pid' "$ROOT" 2>"$err") - [ "$got" = 4242 ] || fail "session-lock dash-leading ancestry selected '$got', expected pid 4242" - [ ! -s "$err" ] || fail "session-lock ancestry wrote basename option noise for literal -zsh: $(cat "$err")" - - err="$dir/fm-session-lock-alive.err" - PATH="$fakebin:$BASE_PATH" bash -c \ - '. "$0/bin/fm-session-lock-lib.sh"; kill() { return 0; }; fm_harness_pid_alive 5252' \ - "$ROOT" 2>"$err"; status=$? - expect_code 0 "$status" "session-lock liveness should accept literal -codex as a harness process name" - [ ! -s "$err" ] || fail "session-lock liveness wrote basename option noise for literal -codex: $(cat "$err")" - - pass "harness identity: dash-leading ps command names are basename operands, not options" -} - # =========================================================================== # B) propagate_inheritable_config unit behavior # =========================================================================== test_propagate_lib() { - local d src dest home m1 m2 outside stdout stderr guard_repo err_text + local d src dest m1 m2 outside stdout stderr guard_repo err_text d="$TMP_ROOT/prop-lib" src="$d/src" - home="$d/home1" - dest="$home/config" - mkdir -p "$src" "$dest" "$home/state" + dest="$d/dest" + mkdir -p "$src" "$dest" # 1. present source is copied printf '{"default":{"harness":"codex"}}\n' > "$src/crew-dispatch.json" printf 'codex\n' > "$src/crew-harness" printf 'manual\n' > "$src/backlog-backend" - printf 'tmux\n' > "$src/backend" : > "$src/herdr-presentation-spaces" stdout="$d/clean-copy.out" stderr="$d/clean-copy.err" @@ -265,11 +210,7 @@ test_propagate_lib() { [ "$(cat "$dest/crew-dispatch.json")" = '{"default":{"harness":"codex"}}' ] || fail "crew-dispatch.json not propagated" [ "$(cat "$dest/crew-harness")" = codex ] || fail "crew-harness not propagated" [ "$(cat "$dest/backlog-backend")" = manual ] || fail "backlog-backend not propagated" - [ "$(cat "$dest/backend")" = tmux ] || fail "backend not propagated" [ -f "$dest/herdr-presentation-spaces" ] || fail "herdr-presentation-spaces not propagated" - printf 'herdr\n' > "$dest/backend" - propagate_inheritable_config "$src" "$dest" - [ "$(cat "$dest/backend")" = tmux ] || fail "primary backend did not overwrite a divergent destination" # 2. idempotent: an unchanged re-run does not churn the mtime m1=$(date -r "$dest/crew-harness" +%s 2>/dev/null || stat -c %Y "$dest/crew-harness") @@ -286,12 +227,10 @@ test_propagate_lib() { printf '{"default":{"harness":"claude"}}\n' > "$src/crew-dispatch.json" printf 'claude\n' > "$src/crew-harness" printf 'tasks-axi\n' > "$src/backlog-backend" - printf 'zellij\n' > "$src/backend" propagate_inheritable_config "$src" "$dest" [ "$(cat "$dest/crew-dispatch.json")" = '{"default":{"harness":"claude"}}' ] || fail "changed dispatch profile did not converge" [ "$(cat "$dest/crew-harness")" = claude ] || fail "changed value did not converge" [ "$(cat "$dest/backlog-backend")" = tasks-axi ] || fail "changed backlog backend did not converge" - [ "$(cat "$dest/backend")" = zellij ] || fail "changed backend did not converge" outside="$d/outside-target" rm -f "$dest/crew-harness" "$outside" @@ -304,14 +243,11 @@ test_propagate_lib() { [ "$(cat "$outside")" = outside ] || fail "destination symlink target was overwritten" # 4. removing the source mirrors absence downstream (primary-authoritative) - printf 'herdr\n' > "$dest/backend" - rm -f "$src/crew-dispatch.json" "$src/crew-harness" "$src/backlog-backend" \ - "$src/backend" "$src/herdr-presentation-spaces" + rm -f "$src/crew-dispatch.json" "$src/crew-harness" "$src/backlog-backend" "$src/herdr-presentation-spaces" propagate_inheritable_config "$src" "$dest" [ -e "$dest/crew-dispatch.json" ] && fail "dispatch profile absence not mirrored downstream" [ -e "$dest/crew-harness" ] && fail "absence not mirrored downstream" [ -e "$dest/backlog-backend" ] && fail "backlog-backend absence not mirrored downstream" - [ -e "$dest/backend" ] && fail "backend absence not mirrored downstream" [ -e "$dest/herdr-presentation-spaces" ] && fail "herdr-presentation-spaces absence not mirrored downstream" rm -f "$dest/crew-harness" @@ -329,25 +265,22 @@ test_propagate_lib() { [ -d "$dest/crew-harness" ] || fail "failed absence mirror removed the wrong path" rm -rf "$dest/crew-harness" - # 5. secondmate-harness is never inherited; backend still is + # 5. secondmate-harness is never inherited printf 'grok\n' > "$src/secondmate-harness" printf '{"default":{"harness":"codex"}}\n' > "$src/crew-dispatch.json" printf 'codex\n' > "$src/crew-harness" printf 'manual\n' > "$src/backlog-backend" - printf 'herdr\n' > "$src/backend" - rm -rf "$d/home2" - mkdir -p "$d/home2/config" "$d/home2/state" - propagate_inheritable_config "$src" "$d/home2/config" - [ -e "$d/home2/config/secondmate-harness" ] && fail "secondmate-harness was inherited (must not be)" - [ "$(cat "$d/home2/config/crew-dispatch.json")" = '{"default":{"harness":"codex"}}' ] || fail "crew-dispatch.json not propagated alongside" - [ "$(cat "$d/home2/config/crew-harness")" = codex ] || fail "crew-harness not propagated alongside" - [ "$(cat "$d/home2/config/backlog-backend")" = manual ] || fail "backlog-backend not propagated alongside" - [ "$(cat "$d/home2/config/backend")" = herdr ] || fail "backend not propagated alongside" + rm -rf "$d/dest2" + mkdir -p "$d/dest2" + propagate_inheritable_config "$src" "$d/dest2" + [ -e "$d/dest2/secondmate-harness" ] && fail "secondmate-harness was inherited (must not be)" + [ "$(cat "$d/dest2/crew-dispatch.json")" = '{"default":{"harness":"codex"}}' ] || fail "crew-dispatch.json not propagated alongside" + [ "$(cat "$d/dest2/crew-harness")" = codex ] || fail "crew-harness not propagated alongside" + [ "$(cat "$d/dest2/backlog-backend")" = manual ] || fail "backlog-backend not propagated alongside" # 6. nothing to propagate -> destination dir is never created (a true no-op) rm -rf "$d/src3" "$d/dest3" mkdir -p "$d/src3" - # Keep backend out of the empty-source case by clearing it from src3 only. propagate_inheritable_config "$d/src3" "$d/dest3/config" [ -e "$d/dest3/config" ] && fail "empty-source propagation created a destination dir" @@ -438,7 +371,6 @@ test_spawn_split_and_inherit() { printf 'claude\n' > "$w/home/config/crew-harness" printf 'codex\n' > "$w/home/config/secondmate-harness" printf 'manual\n' > "$w/home/config/backlog-backend" - printf 'zellij\n' > "$w/home/config/backend" make_seeded_home "$sm" sm spawn_secondmate "$w" sm "$sm" @@ -453,8 +385,6 @@ test_spawn_split_and_inherit() { || fail "split: home crew-dispatch.json not inherited" [ "$(cat "$sm/config/backlog-backend" 2>/dev/null)" = manual ] \ || fail "split: home backlog-backend not inherited as manual" - [ "$(cat "$sm/config/backend" 2>/dev/null)" = zellij ] \ - || fail "split: home backend not inherited as zellij" [ -e "$sm/config/secondmate-harness" ] \ && fail "split: secondmate-harness leaked into the secondmate home" pass "B2 spawn: secondmate runs the secondmate harness; its home inherits declared config" @@ -607,50 +537,6 @@ spawn_secondmate_capture() { "$ROOT/bin/fm-spawn.sh" "$id" "$home" "$@" --secondmate } -test_spawn_backend_precedence_over_inherited_config() { - local w sm meta launchlog out status - w="$TMP_ROOT/spawn-backend-env-precedence" - sm="$w/sm" - launchlog="$w/launch.log" - mkdir -p "$w/home/config" - printf 'herdr\n' > "$w/home/config/backend" - make_seeded_home "$sm" sm - - out=$(FM_BACKEND=tmux spawn_secondmate_capture \ - "$w" sm "$sm" "$launchlog" 2>&1); status=$? - expect_code 0 "$status" \ - "FM_BACKEND=tmux should beat inherited config/backend=herdr"$'\n'"$out" - - meta="$w/home/state/sm.meta" - [ "$(cat "$sm/config/backend")" = herdr ] \ - || fail "backend precedence fixture did not inherit config/backend=herdr" - assert_no_grep '^backend=' "$meta" \ - "FM_BACKEND=tmux did not beat inherited config/backend=herdr" - pass "B5b spawn: FM_BACKEND wins over inherited config/backend" -} - -test_spawn_explicit_backend_precedence_over_env_and_inherited_config() { - local w sm meta launchlog out status - w="$TMP_ROOT/spawn-backend-flag-precedence" - sm="$w/sm" - launchlog="$w/launch.log" - mkdir -p "$w/home/config" - printf 'herdr\n' > "$w/home/config/backend" - make_seeded_home "$sm" sm - - out=$(FM_BACKEND=zellij spawn_secondmate_capture \ - "$w" sm "$sm" "$launchlog" --backend tmux 2>&1); status=$? - expect_code 0 "$status" \ - "explicit --backend tmux should beat FM_BACKEND=zellij and inherited config/backend=herdr"$'\n'"$out" - - meta="$w/home/state/sm.meta" - [ "$(cat "$sm/config/backend")" = herdr ] \ - || fail "explicit backend precedence fixture did not inherit config/backend=herdr" - assert_no_grep '^backend=' "$meta" \ - "explicit --backend tmux did not beat FM_BACKEND=zellij and inherited config/backend=herdr" - pass "B5c spawn: explicit --backend wins over FM_BACKEND and inherited config/backend" -} - # A bare "<harness>" secondmate-harness file (today's format) must launch with # NO --model/--effort flag at all, and meta must keep recording model=default, # effort=default - the core backward-compat requirement of the new format. @@ -884,7 +770,6 @@ new_world() { printf 'projects/\nstate/\ndata/\n.no-mistakes/\n' [ "$dispatch_ignore" = no ] || printf 'config/crew-dispatch.json\n' printf 'config/crew-harness\nconfig/secondmate-harness\nconfig/backlog-backend\n' - printf 'config/backend\nconfig/herdr-presentation-spaces\nconfig/startup-memory-budget\n' } > "$w/main/.gitignore" printf 'v1\n' > "$w/main/AGENTS.md" printf 'r1\n' > "$w/main/README.md" @@ -1080,7 +965,6 @@ test_bootstrap_sweep_propagates_and_reconverges() { printf '{"default":{"harness":"codex"}}\n' > "$w/home/config/crew-dispatch.json" printf 'codex\n' > "$w/home/config/crew-harness" printf 'manual\n' > "$w/home/config/backlog-backend" - printf 'tmux\n' > "$w/home/config/backend" printf 'grok\n' > "$w/home/config/secondmate-harness" run_bootstrap "$w" >/dev/null [ "$(cat "$w/sm/config/crew-harness" 2>/dev/null)" = codex ] \ @@ -1089,8 +973,6 @@ test_bootstrap_sweep_propagates_and_reconverges() { || fail "sweep: crew-dispatch.json not pushed into the live home" [ "$(cat "$w/sm/config/backlog-backend" 2>/dev/null)" = manual ] \ || fail "sweep: backlog-backend not pushed into the live home" - [ "$(cat "$w/sm/config/backend" 2>/dev/null)" = tmux ] \ - || fail "sweep: backend not pushed into the live home" [ -e "$w/sm/config/secondmate-harness" ] \ && fail "sweep: secondmate-harness was inherited (must not be)" @@ -1098,7 +980,6 @@ test_bootstrap_sweep_propagates_and_reconverges() { printf '{"default":{"harness":"claude"}}\n' > "$w/home/config/crew-dispatch.json" printf 'claude\n' > "$w/home/config/crew-harness" printf 'tasks-axi\n' > "$w/home/config/backlog-backend" - printf 'zellij\n' > "$w/home/config/backend" run_bootstrap "$w" >/dev/null [ "$(cat "$w/sm/config/crew-harness" 2>/dev/null)" = claude ] \ || fail "sweep: home did not re-converge to the primary's new crew-harness" @@ -1106,12 +987,9 @@ test_bootstrap_sweep_propagates_and_reconverges() { || fail "sweep: home did not re-converge to the primary's new crew-dispatch.json" [ "$(cat "$w/sm/config/backlog-backend" 2>/dev/null)" = tasks-axi ] \ || fail "sweep: home did not re-converge to the primary's new backlog-backend" - [ "$(cat "$w/sm/config/backend" 2>/dev/null)" = zellij ] \ - || fail "sweep: home did not re-converge to the primary's new backend" # Mirror absence: primary clears inherited config; the home's copies are removed. - rm -f "$w/home/config/crew-dispatch.json" "$w/home/config/crew-harness" \ - "$w/home/config/backlog-backend" "$w/home/config/backend" + rm -f "$w/home/config/crew-dispatch.json" "$w/home/config/crew-harness" "$w/home/config/backlog-backend" run_bootstrap "$w" >/dev/null [ -e "$w/sm/config/crew-dispatch.json" ] \ && fail "sweep: home crew-dispatch.json not removed after the primary cleared it" @@ -1119,8 +997,6 @@ test_bootstrap_sweep_propagates_and_reconverges() { && fail "sweep: home crew-harness not removed after the primary cleared it" [ -e "$w/sm/config/backlog-backend" ] \ && fail "sweep: home backlog-backend not removed after the primary cleared it" - [ -e "$w/sm/config/backend" ] \ - && fail "sweep: home backend not removed after the primary cleared it" pass "B7 bootstrap sweep pushes, re-converges, and mirrors absence; never inherits secondmate-harness" } @@ -1135,7 +1011,6 @@ test_bootstrap_sweep_propagates_when_tracked_current() { printf '{"default":{"harness":"codex"}}\n' > "$w/home/config/crew-dispatch.json" printf 'codex\n' > "$w/home/config/crew-harness" printf 'manual\n' > "$w/home/config/backlog-backend" - printf 'tmux\n' > "$w/home/config/backend" run_bootstrap "$w" >/dev/null [ "$(cat "$w/sm/config/crew-dispatch.json" 2>/dev/null)" = '{"default":{"harness":"codex"}}' ] \ || fail "crew-dispatch.json did not propagate to a tracked-current home" @@ -1143,8 +1018,6 @@ test_bootstrap_sweep_propagates_when_tracked_current() { || fail "config did not propagate to a tracked-current home" [ "$(cat "$w/sm/config/backlog-backend" 2>/dev/null)" = manual ] \ || fail "backlog-backend did not propagate to a tracked-current home" - [ "$(cat "$w/sm/config/backend" 2>/dev/null)" = tmux ] \ - || fail "backend did not propagate to a tracked-current home" pass "B8 bootstrap sweep propagates config even when the home's tracked files are already current" } @@ -1177,10 +1050,10 @@ test_bootstrap_sweep_defers_dispatch_on_stale_unignored_home() { pass "B9 bootstrap sweep defers new inherited config until the home ignores it" } -# The primary bootstrap always materializes the startup-memory default, so an -# otherwise empty inherited surface converges that one visible value while -# ordinary tracked-file fast-forward behavior remains unchanged. -test_bootstrap_sweep_materializes_and_inherits_memory_default() { +# Backward-compat: with no inherited config set, the sweep is a no-op for the +# home's config/ - exactly as before this feature - and ordinary sweep behavior +# (fast-forward) is unaffected. +test_bootstrap_sweep_no_inheritance_is_noop() { local w c1 w=$(new_world boot-noop) c1=$(git -C "$w/main" rev-parse HEAD) @@ -1194,52 +1067,12 @@ test_bootstrap_sweep_materializes_and_inherits_memory_default() { run_bootstrap "$w" >/dev/null - [ -e "$w/sm/config/crew-dispatch.json" ] && fail "default-only sweep created a home crew-dispatch.json" - [ -e "$w/sm/config/crew-harness" ] && fail "default-only sweep created a home crew-harness" - [ -e "$w/sm/config/backend" ] && fail "default-only sweep created a home backend" - [ "$(cat "$w/home/config/startup-memory-budget")" = 7500 ] \ - || fail "primary bootstrap did not materialize the startup-memory default" - [ "$(cat "$w/sm/config/startup-memory-budget")" = 7500 ] \ - || fail "default-only sweep did not converge startup-memory-budget" + [ -e "$w/sm/config/crew-dispatch.json" ] && fail "no-inheritance sweep created a home crew-dispatch.json" + [ -e "$w/sm/config/crew-harness" ] && fail "no-inheritance sweep created a home crew-harness" + [ -e "$w/sm/config" ] && fail "no-inheritance sweep created a home config/ dir" [ "$(git -C "$w/sm" rev-parse HEAD)" = "$head" ] \ - || fail "default-only sweep did not still fast-forward the tracked files" - pass "B10 bootstrap sweep materializes and inherits the startup-memory default while fast-forwarding" -} - -# config/backend: present and absent primary state converges exactly. -test_backend_inheritance_present_and_absent() { - local w head out err status instruction - w=$(new_world backend-inherit) - head=$(git -C "$w/main" rev-parse HEAD) - add_sm_worktree "$w" sm "$head" - - printf 'tmux\n' > "$w/home/config/backend" - err="$w/backend-inherit.err" - out=$(run_config_push "$w" 2>"$err"); status=$? - expect_code 0 "$status" "backend present push should succeed" - assert_contains "$out" "backend: pushed" "backend present value should report pushed" - [ "$(cat "$w/sm/config/backend")" = tmux ] || fail "backend present value not pushed" - instruction=$(reread_instruction_path "$w/sm") || fail "backend present reread instruction missing" - assert_contains "$(cat "$instruction")" $'-----BEGIN config/backend-----\ntmux\n-----END config/backend-----' \ - "backend present reread must include exact bytes" - - printf 'herdr\n' > "$w/sm/config/backend" - printf 'zellij\n' > "$w/home/config/backend" - out=$(run_config_push "$w" 2>"$err"); status=$? - expect_code 0 "$status" "backend changed push should succeed" - assert_contains "$out" "backend: pushed" "backend changed value should report pushed" - [ "$(cat "$w/sm/config/backend")" = zellij ] \ - || fail "primary backend did not overwrite the divergent destination" - - rm -f "$w/home/config/backend" - out=$(run_config_push "$w" 2>"$err"); status=$? - expect_code 0 "$status" "backend absence push should succeed" - assert_contains "$out" "backend: pushed - mirrored primary absence" "backend should mirror primary absence" - [ -e "$w/sm/config/backend" ] && fail "backend not removed on primary absence" - instruction=$(reread_instruction_path "$w/sm") || fail "backend absence reread instruction missing" - assert_contains "$(cat "$instruction")" $'-----BEGIN config/backend-----\nABSENT\n-----END config/backend-----' \ - "backend absence reread must use ABSENT token" - pass "B12b backend inheritance: present values and primary absence converge exactly" + || fail "no-inheritance sweep did not still fast-forward the tracked files" + pass "B10 bootstrap sweep with no inherited config is a config no-op and still fast-forwards" } test_bootstrap_sweep_surfaces_config_propagation_failure() { @@ -1280,7 +1113,7 @@ test_bootstrap_rereads_after_partial_propagation() { } test_config_push_propagates_reports_without_ff_or_nudge() { - local w c1 sm_real old_head out err status out2 tmp log instruction + local w c1 sm_real old_head out err status out2 tmp log w=$(new_world config-push-basic) c1=$(git -C "$w/main" rev-parse HEAD) add_sm_worktree "$w" sm "$c1" @@ -1298,7 +1131,6 @@ test_config_push_propagates_reports_without_ff_or_nudge() { printf '{"default":{"harness":"codex"}}\n' > "$w/home/config/crew-dispatch.json" printf 'codex\n' > "$w/home/config/crew-harness" printf 'manual\n' > "$w/home/config/backlog-backend" - printf 'tmux\n' > "$w/home/config/backend" err="$w/config-push-basic.err" log="$w/config-push-basic.tmux.log" out=$(run_config_push "$w" "$log" 2>"$err"); status=$? @@ -1314,18 +1146,12 @@ test_config_push_propagates_reports_without_ff_or_nudge() { "config push did not report crew-harness as pushed" assert_contains "$out" "backlog-backend: pushed" \ "config push did not report backlog-backend as pushed" - assert_contains "$out" "backend: pushed" \ - "config push did not report backend as pushed" assert_contains "$out" "config-reread: sent" \ "config push with changed config must send a literal reread instruction" assert_not_contains "$out" "NUDGE_SECONDMATES" \ "config push must not use the AGENTS.md instruction-surface nudge channel" [ "$(git -C "$w/sm" rev-parse HEAD)" = "$old_head" ] \ || fail "config push fast-forwarded tracked files" - [ "$(cat "$w/sm/config/backend")" = tmux ] || fail "config push did not write backend" - instruction=$(reread_instruction_path "$w/sm") || fail "config-push reread instruction missing" - assert_contains "$(cat "$instruction")" $'-----BEGIN config/backend-----\ntmux\n-----END config/backend-----' \ - "config-push reread must include exact backend bytes" [ ! -s "$err" ] || fail "clean config push wrote unexpected stderr: $(cat "$err")" assert_contains "$(cat "$log")" "[fm-from-firstmate]" \ "config reread must use the marked routed secondmate path" @@ -1339,8 +1165,6 @@ test_config_push_propagates_reports_without_ff_or_nudge() { "idempotent config push did not report crew-harness as unchanged" assert_contains "$out2" "backlog-backend: unchanged" \ "idempotent config push did not report backlog-backend as unchanged" - assert_contains "$out2" "backend: unchanged" \ - "idempotent config push did not report backend as unchanged" assert_not_contains "$out2" "config-reread: sent" \ "unchanged config must not send a reread message" [ ! -s "$log" ] || fail "unchanged config push still invoked tmux send: $(cat "$log")" @@ -1477,7 +1301,6 @@ test_config_reread_per_home_changed_sets_and_exact_bytes() { printf '%s' "$multiline_json" > "$w/home/config/crew-dispatch.json" printf 'codex\n' > "$w/home/config/crew-harness" printf 'manual\n' > "$w/home/config/backlog-backend" - printf 'tmux\n' > "$w/home/config/backend" { shared_captain_header_for_tests printf '%s\n' "shared secret preference body that must never appear in a config reread" @@ -1496,7 +1319,6 @@ test_config_reread_per_home_changed_sets_and_exact_bytes() { || fail "beta did not receive multiline dispatch" [ "$(cat "$w/alpha/config/crew-harness")" = codex ] || fail "alpha harness not updated" [ "$(cat "$w/alpha/config/backlog-backend")" = manual ] || fail "alpha backlog-backend not updated" - [ "$(cat "$w/alpha/config/backend")" = tmux ] || fail "alpha backend not updated" instr_a=$(reread_instruction_path "$w/alpha") || fail "alpha instruction missing after config push" instr_b=$(reread_instruction_path "$w/beta") || fail "beta instruction missing after config push" @@ -1506,21 +1328,19 @@ test_config_reread_per_home_changed_sets_and_exact_bytes() { [ "$(reread_mode "$instr_b")" = 600 ] || fail "beta instruction is not private" # Deterministic allowlist path order and exact destination bytes for alpha - # (allowlisted config items were missing/stale and therefore pushed). + # (all three config items were missing/stale and therefore pushed). assert_grep "These inherited config files changed" "$instr_a" "alpha framing missing" assert_grep "defaults/rules" "$instr_a" "alpha must preserve agent judgment framing" assert_contains "$(cat "$instr_a")" "config/crew-dispatch.json" "alpha missing dispatch path" assert_contains "$(cat "$instr_a")" "config/crew-harness" "alpha missing harness path" assert_contains "$(cat "$instr_a")" "config/backlog-backend" "alpha missing backlog path" - assert_contains "$(cat "$instr_a")" "config/backend" "alpha missing backend path" # Path order follows FM_INHERITABLE_CONFIG. awk ' /config\/crew-dispatch\.json/ { d=NR } /config\/crew-harness/ { h=NR } /config\/backlog-backend/ { b=NR } - /config\/backend/ && !/backlog-backend/ { k=NR } END { - if (!(d && h && b && k && d < h && h < b && b < k)) exit 1 + if (!(d && h && b && d < h && h < b)) exit 1 } ' "$instr_a" || fail "alpha instruction path order is not deterministic allowlist order" @@ -1531,8 +1351,6 @@ test_config_reread_per_home_changed_sets_and_exact_bytes() { "alpha instruction must include exact harness scalar bytes" assert_contains "$(cat "$instr_a")" $'-----BEGIN config/backlog-backend-----\nmanual\n-----END config/backlog-backend-----' \ "alpha instruction must include exact backlog-backend scalar bytes" - assert_contains "$(cat "$instr_a")" $'-----BEGIN config/backend-----\ntmux\n-----END config/backend-----' \ - "alpha instruction must include exact backend scalar bytes" # No parsed/effective summary, no SHA, no captain-shared dump. assert_not_contains "$(cat "$instr_a")" "Default worker" "must not emit parsed worker summary" @@ -1616,7 +1434,6 @@ test_config_reread_isolation_and_absent_and_send_failure() { printf '%s\n' $'crew-dispatch.json\tpushed\tmirrored primary absence' printf '%s\n' $'crew-harness\tunchanged\t' printf '%s\n' $'backlog-backend\tunchanged\t' - printf '%s\n' $'backend\tunchanged\t' printf '%s\n' $'data/captain-shared.md\tpushed\t' } > "$report" rm -f "$w/beta/config/crew-dispatch.json" @@ -2197,7 +2014,6 @@ cat > "$w/main/bin/fm-spawn.sh" <<SH . '$w/main/bin/fm-config-inherit-lib.sh' printf '%s' spawn >> '$log' printf '%s' codex > '$w/sm/config/crew-harness' -printf '%s\n' 7500 > '$w/sm/config/startup-memory-budget' SH chmod +x "$w/main/bin/fm-spawn.sh" fakebin=$(make_fake_toolchain "$w") @@ -2302,15 +2118,12 @@ SH test_harness_resolution test_secondmate_model_effort_tokens test_pi_signed_detection_and_session_lock_identity -test_dash_leading_process_names_are_basename_operands test_propagate_lib test_spawn_split_and_inherit test_spawn_backward_compat_crew_fallback test_spawn_bare_backward_compat test_spawn_explicit_harness_wins test_spawn_unverified_secondmate_harness_refused -test_spawn_backend_precedence_over_inherited_config -test_spawn_explicit_backend_precedence_over_env_and_inherited_config test_spawn_bare_harness_no_model_effort_flag test_spawn_secondmate_harness_model_token test_spawn_secondmate_harness_model_and_effort_tokens @@ -2322,8 +2135,7 @@ test_spawn_fallback_chain_and_crew_scout_unaffected test_bootstrap_sweep_propagates_and_reconverges test_bootstrap_sweep_propagates_when_tracked_current test_bootstrap_sweep_defers_dispatch_on_stale_unignored_home -test_bootstrap_sweep_materializes_and_inherits_memory_default -test_backend_inheritance_present_and_absent +test_bootstrap_sweep_no_inheritance_is_noop test_bootstrap_sweep_surfaces_config_propagation_failure test_bootstrap_rereads_after_partial_propagation test_config_push_propagates_reports_without_ff_or_nudge diff --git a/tests/fm-secondmate-liveness.test.sh b/tests/fm-secondmate-liveness.test.sh index 2b572681572..ff5c07a948c 100755 --- a/tests/fm-secondmate-liveness.test.sh +++ b/tests/fm-secondmate-liveness.test.sh @@ -97,7 +97,7 @@ SH test_tmux_agent_state_classifies() { local fb out - for harness in claude codex opencode grok kimi; do + for harness in claude codex opencode grok kimi pi pi-signed pi-launcher Pi; do fb=$(make_probe_tmux "$TMP_ROOT/tmux-$harness" "$harness") out=$(PATH="$fb:$BASE_PATH" bash -c '. "$0/bin/fm-backend.sh"; fm_backend_agent_state tmux sess:win' "$ROOT") [ "$out" = alive ] || fail "a live $harness foreground process should classify as alive, got '$out'" @@ -206,7 +206,7 @@ test_agent_state_dispatcher_and_compatibility() { make_toolchain() { local dir=$1 fakebin fakebin=$(fm_fakebin "$dir") - fm_fake_exit0 "$fakebin" node gh-axi chrome-devtools-axi lavish-axi + fm_fake_exit0 "$fakebin" node gh-axi chrome-devtools-axi lavish-axi pi-signed cat > "$fakebin/gh" <<'SH' #!/usr/bin/env bash exit 0 @@ -391,6 +391,25 @@ test_sweep_respawns_authoritatively_missing_pi_secondmate() { pass "sweep: an authoritatively missing Pi secondmate window is relaunched" } +test_sweep_respawns_authoritatively_missing_pi_signed_secondmate() { + local w fb tmuxfb log out + w=$(new_world sweep-missing-pi-signed) + printf '%s\n' pi-signed > "$w/home/config/secondmate-harness" + add_sm_home "$w" sm1 firstmate:fm-sm1 pi-signed + fb=$(make_toolchain "$w"); tmuxfb=$(make_liveness_tmux "$w") + log="$w/calls.log"; : > "$log" + + out=$(run_bootstrap "$tmuxfb:$fb" "$w/home" missing "$log") + + assert_not_contains "$out" "unverified for recovery" \ + "a recorded pi-signed secondmate should be verified for recovery" + assert_contains "$(cat "$log")" "new-window" \ + "an authoritatively missing pi-signed secondmate should be relaunched" + assert_not_contains "$(cat "$log")" "kill-window" \ + "an absent pi-signed window should not need a destructive pre-kill" + pass "sweep: an authoritatively missing pi-signed secondmate window is relaunched" +} + test_sweep_never_acts_on_ambiguous_existing_process() { local w fb tmuxfb log out w=$(new_world sweep-ambiguous) @@ -514,6 +533,7 @@ test_agent_state_dispatcher_and_compatibility test_sweep_respawns_confirmed_dead_secondmate test_sweep_leaves_alive_secondmate_untouched test_sweep_respawns_authoritatively_missing_pi_secondmate +test_sweep_respawns_authoritatively_missing_pi_signed_secondmate test_sweep_never_acts_on_ambiguous_existing_process test_sweep_never_acts_on_transient_unreadability test_sweep_reports_missing_endpoint_relaunch_failure diff --git a/tests/fm-spawn-dispatch-profile.test.sh b/tests/fm-spawn-dispatch-profile.test.sh index 4f587a9a398..3a8dfb3a4e7 100755 --- a/tests/fm-spawn-dispatch-profile.test.sh +++ b/tests/fm-spawn-dispatch-profile.test.sh @@ -42,7 +42,7 @@ esac exit 0 SH chmod +x "$fakebin/tmux" - fm_fake_exit0 "$fakebin" treehouse + fm_fake_exit0 "$fakebin" treehouse pi-signed printf '%s\n' "$fakebin" } @@ -342,7 +342,7 @@ test_pi_threads_model_and_max_effort() { expect_code 0 "$status" "pi spawn with max effort should succeed" assert_meta_profile "$HOME_DIR/state/$id.meta" pi openai-codex/gpt-5.6-sol max launch=$(cat "$LAUNCH_LOG") - assert_contains "$launch" "pi --model 'openai-codex/gpt-5.6-sol' --thinking 'max' -e" \ + assert_contains "$launch" "FM_PI_HARNESS=pi pi --model 'openai-codex/gpt-5.6-sol' --thinking 'max' -e" \ "pi launch did not thread the requested model and max thinking level" assert_not_contains "$launch" "FM_FIRSTMATE_PI_LAUNCH_BRIEF=" \ "pi launch still exports the removed Calm input-reroute binding" @@ -351,6 +351,72 @@ test_pi_threads_model_and_max_effort() { pass "pi receives --model and --thinking max profile flags" } +test_pi_signed_threads_shared_pi_profile_and_preserves_identity() { + local rec id out status launch + id=profile-pi-signed-z8b + rec=$(make_spawn_case profile-pi-signed pi-signed "$id") + read_case_record "$rec" + + out=$(run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR" \ + --model openai-codex/gpt-5.6-sol --effort max) + status=$? + expect_code 0 "$status" "pi-signed spawn with max effort should succeed" + assert_contains "$out" "spawned $id harness=pi-signed" "pi-signed spawn did not preserve its visible identity" + assert_meta_profile "$HOME_DIR/state/$id.meta" pi-signed openai-codex/gpt-5.6-sol max + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "FM_PI_HARNESS=pi-signed pi-signed --model 'openai-codex/gpt-5.6-sol' --thinking 'max' -e" \ + "pi-signed launch did not share Pi's model, thinking, and extension semantics" + assert_contains "$launch" "fm-operational-input.sh' encode launch-brief" \ + "pi-signed launch lost the canonical typed launch-brief envelope" + assert_present "$HOME_DIR/state/$id.pi-ext.ts" "pi-signed launch did not install Pi's turn-end extension" + pass "pi-signed shares Pi launch semantics while preserving its configured and recorded identity" +} + +test_pi_signed_missing_binary_refuses_before_endpoint_or_metadata() { + local rec id out status + id=profile-pi-signed-missing-z8c + rec=$(make_spawn_case profile-pi-signed-missing pi-signed "$id") + read_case_record "$rec" + rm -f "$FAKEBIN_DIR/pi-signed" + : > "$LAUNCH_LOG" + + out=$(FM_ROOT_OVERRIDE='' FM_HOME="$HOME_DIR" \ + FM_STATE_OVERRIDE="$HOME_DIR/state" FM_DATA_OVERRIDE="$HOME_DIR/data" \ + FM_PROJECTS_OVERRIDE="$HOME_DIR/projects" FM_CONFIG_OVERRIDE="$HOME_DIR/config" \ + FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$WT_DIR" TMUX="fake,1,0" \ + FM_FAKE_LAUNCH_LOG="$LAUNCH_LOG" PATH="$FAKEBIN_DIR:/usr/bin:/bin:/usr/sbin:/sbin" \ + "$SPAWN" "$id" "$PROJ_DIR" 2>&1) + status=$? + expect_code 1 "$status" "a missing pi-signed executable should refuse the spawn" + assert_contains "$out" "pi-signed executable not found on PATH" \ + "missing pi-signed refusal did not name the actionable requirement" + assert_absent "$HOME_DIR/state/$id.meta" "missing pi-signed refusal wrote task metadata" + [ ! -s "$LAUNCH_LOG" ] || fail "missing pi-signed refusal typed a launch command" + pass "pi-signed refuses safely and actionably when the selected executable is unavailable" +} + +test_pi_signed_persistent_secondmate_uses_pi_extensions_and_identity() { + local rec id sm out status launch + id=profile-pi-signed-secondmate-z8d + rec=$(make_spawn_case profile-pi-signed-secondmate codex "$id") + read_case_record "$rec" + printf '%s\n' pi-signed > "$HOME_DIR/config/secondmate-harness" + sm="$CASE_DIR/secondmate-home" + make_seeded_secondmate_home "$sm" "$id" + sm=$(cd "$sm" && pwd -P) + + out=$(run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$sm" --secondmate) + status=$? + expect_code 0 "$status" "pi-signed persistent secondmate spawn should succeed" + assert_contains "$out" "spawned $id harness=pi-signed kind=secondmate" \ + "pi-signed secondmate spawn did not preserve its runtime identity" + assert_meta_profile "$HOME_DIR/state/$id.meta" pi-signed default default + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "FM_PI_HARNESS=pi-signed pi-signed -e '$sm/.pi/extensions/fm-primary-turnend-guard.ts' -e '$sm/.pi/extensions/fm-primary-pi-watch.ts'" \ + "pi-signed secondmate did not share Pi's primary extension launch shape" + pass "pi-signed is a distinct persistent secondmate runtime with shared Pi supervision semantics" +} + test_batch_forwards_shared_profile_flags() { local rec id1 id2 out status id1=profile-batch-a-z9 @@ -402,6 +468,9 @@ test_grok_omits_invalid_max_reasoning_effort test_grok_omits_invalid_xhigh_reasoning_effort test_opencode_threads_model_and_ignores_effort_axis test_pi_threads_model_and_max_effort +test_pi_signed_threads_shared_pi_profile_and_preserves_identity +test_pi_signed_missing_binary_refuses_before_endpoint_or_metadata +test_pi_signed_persistent_secondmate_uses_pi_extensions_and_identity test_batch_forwards_shared_profile_flags test_active_dispatch_profile_does_not_block_secondmate_launch diff --git a/tests/fm-tmux-submit-busy.test.sh b/tests/fm-tmux-submit-busy.test.sh index 58932509d0e..f3eb49a7eb7 100755 --- a/tests/fm-tmux-submit-busy.test.sh +++ b/tests/fm-tmux-submit-busy.test.sh @@ -251,6 +251,7 @@ test_claude_busy_signature_uses_real_capture_shapes() { pane_busy old-claude claude || fail "older Claude escape footer should be busy" printf 'Working...\n' > "$composer" pane_busy pi pi || fail "Pi Working footer should be busy" + pane_busy pi-signed pi-signed || fail "pi-signed should share Pi's exact Working footer" printf 'Ctrl+c:cancel\n' > "$composer" pane_busy grok grok || fail "Grok cancel footer should be busy" pass "fm_pane_is_busy: Claude spinner is scoped, multi-frame, and backward-compatible" From 82b785c98c264af7ec8f2b154a40b0049a9b1bbc Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Mon, 27 Jul 2026 22:30:47 -0700 Subject: [PATCH 19/52] fix(pi): rearm watcher across session transitions (#1166) * fix(pi): rearm watcher across same-process session transitions Pi emits session_shutdown for ordinary /new, /resume, and /fork replacement as well as terminal quit. The primary watcher extension latched a module-level stopping flag on every shutdown, so a replacement session in the same process could not arm monitoring until Pi restarted. Own arm authority per session generation so only the active live generation may start, stop, or rearm the child. Replacement sessions can arm again without restarting Pi, stale prior-generation callbacks cannot mutate the active cycle, and real quit still blocks late rearm. * no-mistakes(review): Preserve Pi generation isolation and exit cleanup * no-mistakes(document): Correct Pi watcher transition documentation --- docs/supervision-protocols/pi.md | 12 +++-- docs/verification/supervision.md | 13 +++++ docs/watcher-continuity.md | 2 + tests/fm-pi-watch-extension.test.sh | 81 +++++++++++++++++++++++++++++ 4 files changed, 103 insertions(+), 5 deletions(-) diff --git a/docs/supervision-protocols/pi.md b/docs/supervision-protocols/pi.md index 30c4aae35fc..2316428a833 100644 --- a/docs/supervision-protocols/pi.md +++ b/docs/supervision-protocols/pi.md @@ -8,12 +8,14 @@ When this session owns supervision and away mode is not active: Never run `bin/fm-watch-arm.sh` through Pi's bash tool because that foreground arm can wedge the agent and bypasses extension-owned cleanup. 4. If the extension says no live session holds the lock, run `bin/fm-session-start.sh` to reclaim the session lock, then call `fm_watch_arm_pi` again. 5. The extension starts `bin/fm-watch-arm.sh --restart`, keeps the child attached to the live Pi process, and owns every later successor launch. -6. After an actionable child close, the extension rechecks session-lock ownership and verifies one successor before it delivers the follow-up wake; its bounded fallback is defined in `docs/watcher-continuity.md`. -7. Ordinary work, turn completion, and ordinary signal, stale, check, heartbeat, or other wake handling: do not call `fm_watch_arm_pi` again because continuity is extension-owned rather than model-memory-owned. -8. An unexpected child close enters bounded exponential retry, and an exhausted retry or lost session lock is surfaced as a watcher failure instead of disappearing. -9. Missing, failed, or unhealthy cycle only: if a later notification explicitly reports one of those repair conditions, drain queued wakes, inspect the failure text, call `fm_watch_arm_pi`, and restart the selected Pi-family executable with both extensions loaded if needed. +6. Ordinary same-process session replacement (`/new`, `/resume`, `/fork`, reload) retires only the prior generation; call `fm_watch_arm_pi` once for the first cycle of the replacement session without restarting Pi. + The generation-owner contract lives in `.pi/extensions/fm-primary-pi-watch.ts`. +7. After an actionable child close, the extension rechecks session-lock ownership and verifies one successor before it delivers the follow-up wake; its bounded fallback is defined in `docs/watcher-continuity.md`. +8. Ordinary work, turn completion, and ordinary signal, stale, check, heartbeat, or other wake handling: do not call `fm_watch_arm_pi` again because continuity is extension-owned rather than model-memory-owned. +9. An unexpected child close enters bounded exponential retry, and an exhausted retry or lost session lock is surfaced as a watcher failure instead of disappearing. +10. Missing, failed, or unhealthy cycle only: if a later notification explicitly reports one of those repair conditions, drain queued wakes, inspect the failure text, call `fm_watch_arm_pi`, and restart the selected Pi-family executable with both extensions loaded if needed. A redundant call while the extension owns an arm child or scheduled retry is an ownership-based `watcher: unchanged` no-op, not an independent health claim. -10. Never use shell `&` for watcher supervision. +11. Never use shell `&` for watcher supervision. The arm mechanism above is extension-owned, not a model tool call, but a manual recovery probe that backgrounds, pipes, or bundles the arm is denied automatically by the PreToolUse seatbelt (`bin/fm-arm-pretool-check.sh`, wired into the turn-end guard extension at `__FM_PI_TURNEND_EXT__`). The turn-end guard extension lives at `__FM_PI_TURNEND_EXT__`. diff --git a/docs/verification/supervision.md b/docs/verification/supervision.md index 30a98311a40..6945b3491dc 100644 --- a/docs/verification/supervision.md +++ b/docs/verification/supervision.md @@ -121,10 +121,23 @@ grok 0.2.103 (89c3d36fb6f1) [stable] Pi 0.81.1 repeated the continuity and clean-exit lifecycle on 2026-07-23 after the Calm presentation changes. +Pi same-process session-transition ownership was verified on 2026-07-27 against the tracked extension with a faithful in-process factory rebind (module cache retained, real arm children): + +```sh +pi --version +tests/fm-pi-watch-extension.test.sh +tests/fm-pi-primary-types.test.sh +``` + +Observed guarantee: after ordinary `session_shutdown` for `/new`, `/resume`, and `/fork`, plus same-instance shutdown-plus-start, the replacement generation armed again without a Pi restart and without the `watcher: not armed - Pi session is shutting down` refusal. +Stale prior-generation tool callbacks could not mutate the active child, repeated transitions kept exactly one live arm cycle, and terminal `quit` still refused late rearm. +Plain Pi and pi-signed share the same tracked `.pi/extensions/fm-primary-pi-watch.ts` path, so both inherit the generation owner; other primary harnesses are not applicable because they do not use this Pi extension lifecycle. + Deterministic entry points: ```sh tests/fm-pi-watch-extension.test.sh +tests/fm-pi-primary-types.test.sh tests/fm-watcher-lock.test.sh tests/fm-subagent-pretool-check.test.sh tests/fm-claude-stop-autoarm.test.sh diff --git a/docs/watcher-continuity.md b/docs/watcher-continuity.md index 457c035268d..52b3a9eec3a 100644 --- a/docs/watcher-continuity.md +++ b/docs/watcher-continuity.md @@ -8,6 +8,7 @@ Must-work continuity now lives above that process boundary instead of depending Pi's `.pi/extensions/fm-primary-pi-watch.ts` and OpenCode's `.opencode/plugins/fm-primary-watch-arm.js` own continuous re-arm after an actionable child close. Each adapter starts the next arm before delivering the wake prompt, checks current session-lock ownership at launch, preserves one child or scheduled retry at a time, and applies bounded exponential retry after an unexpected or failed close. A failed follow-up never cancels continuity restoration. +Pi same-process session replacement follows the generation-owner contract in `.pi/extensions/fm-primary-pi-watch.ts`. Claude's `.claude/settings.json` Stop `asyncRewake` hook (`bin/fm-claude-stop-autoarm.sh`) owns routine tokenless re-arm. The hook fires on every Stop, and an eligible primary with supervision need admits one home-scoped owner that foregrounds `bin/fm-watch-arm.sh` inside the hook-owned process tree. A numeric session-lock owner that fails the shared `fm_harness_pid_alive` predicate is reclaimed through `bin/fm-lock.sh` before auto-arm state changes, while a live owner, absent lock, or malformed lock keeps the competing hook inert. @@ -52,6 +53,7 @@ Only the watcher process touches `state/.last-watcher-beat`; no helper process c ## Regression coverage `tests/fm-pi-watch-extension.test.sh` checks Pi's first-cycle-or-explicit-repair tool metadata and ownership-based redundant-call no-ops, then simulates actionable and empty child closes against the actual Pi and OpenCode close handlers, blocks prompt delivery to prove the successor launches first, verifies single-flight behavior, changes the session lock before close to prove ownership is rechecked, and hangs each successor arm to prove bounded fallback delivery includes the typed restoration failure. +The same suite covers ordinary same-process session replacement for `/new`, `/resume`, and `/fork`, same-instance shutdown-plus-start, stale prior-generation callbacks, repeated transitions with exactly one live cycle, disappearance of the shutting-down refusal after a valid replacement activates, and terminal quit still refusing late rearm. `tests/fm-watcher-lock.test.sh` covers verified-successor attach, the typed self-eviction failure, bounded and successor-linked lifecycle rows, and a SIGSTOP counterfactual that distinguishes a live PID from a stale beacon before classifying termination. `tests/fm-subagent-pretool-check.test.sh` proves Claude retains only the non-status Bash seatbelts. `tests/fm-claude-stop-autoarm.test.sh` covers the auto-arm's scope, stale and live session owners, unchanged AFK and need boundaries, single-flight, and exit-2 translation. diff --git a/tests/fm-pi-watch-extension.test.sh b/tests/fm-pi-watch-extension.test.sh index f8883194898..518df0e874e 100755 --- a/tests/fm-pi-watch-extension.test.sh +++ b/tests/fm-pi-watch-extension.test.sh @@ -59,6 +59,64 @@ export const Type = { JS } +test_tracked_extension_present_and_self_hashing() { + local text expected_config_source + expected_config_source="config_dir=\\\"\${FM_CONFIG_OVERRIDE:-\$FM_HOME/config}\\\"" + assert_present "$EXT" "tracked Pi primary watcher extension is missing" + text=$(cat "$EXT") + assert_contains "$text" "fm_watch_arm_pi" "tracked extension missing tool name" + assert_contains "$text" "fm-watch-arm-pi" "tracked extension missing command name" + assert_contains "$text" "fm-watch-arm.sh" "tracked extension missing watcher arm" + assert_contains "$text" "sendUserMessage" "tracked extension missing Pi wake API" + assert_contains "$text" 'encodeFirstmateOperationalInput' "tracked extension does not construct typed synthetic user-role wakes" + assert_contains "$text" "deliverAs: \"followUp\"" "tracked extension missing followUp delivery" + assert_contains "$text" ".pi-watch-extension-loaded" "tracked extension missing loaded marker" + assert_contains "$text" 'createHash("sha256").update(readFileSync(extensionFile)).digest("hex")' "tracked extension does not self-hash its own content for extensionVersion" + assert_contains "$text" 'fileURLToPath(import.meta.url)' "tracked extension does not self-locate via import.meta.url" + assert_contains "$text" 'type LockOwnership = "owned" | "missing" | "other"' "tracked extension does not distinguish missing lock from another owner" + assert_contains "$text" "readFileSync(\`\${state}/.lock\`" "tracked extension does not read the effective session lock" + assert_contains "$text" 'return pidAlive(lockPid) ? "other" : "missing"' "tracked extension does not allow a pre-lock load marker" + assert_contains "$text" 'if (lockOwnership() === "other") return' "tracked extension overwrites another live session marker" + assert_contains "$text" 'const ownership = lockOwnership()' "tracked extension arm does not inspect the distinct lock ownership state" + assert_contains "$text" 'if (ownership === "other") return { ok: false' "tracked extension arm does not preserve the live-other read-only refusal" + assert_contains "$text" 'if (ownership === "missing")' "tracked extension arm collapses a stale or absent lock into the live-other refusal" + assert_contains "$text" "no live session holds the lock" "tracked extension arm missing stale-lock recovery guidance" + assert_contains "$text" "run bin/fm-session-start.sh to reclaim it" "tracked extension arm does not direct stale-lock reclamation" + assert_contains "$text" "call fm_watch_arm_pi to re-arm" "tracked extension arm does not direct supervision re-arm" + assert_contains "$text" "writeFileSync(marker, \`\${extensionVersion}\\n\${process.pid}\\n\`)" "tracked extension does not write the content version and process marker" + assert_contains "$text" "const config = process.env.FM_CONFIG_OVERRIDE" "tracked extension missing effective config resolution" + assert_contains "$text" "FM_CONFIG_OVERRIDE: config" "tracked extension does not pass the effective config to the watcher arm" + assert_contains "$text" "FM_WATCH_ARM_SCRIPT: armScript" "tracked extension does not pass the effective watcher arm script" + assert_contains "$text" "$expected_config_source" "tracked extension does not source the effective x-mode config" + assert_contains "$text" "exec \\\"\$FM_WATCH_ARM_SCRIPT\\\" --restart" "tracked extension does not restart into a Pi-owned watcher child" + assert_contains "$text" 'label: "Arm firstmate watcher"' "tracked extension tool is missing its human-readable label" + assert_not_contains "$text" "Always use this tool" "tracked extension kept broad tool-selection guidance" + assert_contains "$text" "only for the first required cycle or after a notification says the cycle is missing, failed, or unhealthy" "tracked extension tool metadata is missing the Pi first-cycle or explicit-repair rule" + assert_contains "$text" "Do not call it after ordinary work, turn completion, or ordinary signal, stale, check, or heartbeat handling" "tracked extension prompt guidance does not prevent redundant ordinary-notification calls" + assert_contains "$text" 'parameters: Type.Object({})' "tracked extension tool is not using Pi's canonical TypeBox schema" + assert_contains "$text" 'content: [{ type: "text", text: result.message }]' "tracked extension tool is missing Pi text content" + assert_contains "$text" 'details: result' "tracked extension tool is missing structured result details" + assert_contains "$text" 'ctx.ui.notify' "tracked extension command does not notify through Pi's UI" + assert_contains "$text" 'process.once("exit", cleanupOnProcessExit)' "tracked extension lacks clean-process-exit cleanup" + assert_contains "$text" "type SessionGeneration" "tracked extension lacks an explicit session-generation owner" + assert_contains "$text" "function activateGeneration" "tracked extension does not activate a live generation for replacement sessions" + assert_contains "$text" "function generationIsLive" "tracked extension does not gate arm mutations on the live generation" + assert_contains "$text" "watcher: not armed - Pi session is shutting down" "tracked extension missing the terminal shutdown refusal" + assert_not_contains "$text" "[ -f config/x-mode.env ]" "tracked extension kept a repo-relative x-mode config path" + pass "Pi primary watcher extension is tracked, self-hashing, and self-locating" +} + +test_spawn_template_mentions_pi_watch_placeholder() { + local text + text=$(cat "$ROOT/bin/fm-spawn.sh") + assert_contains "$text" "-e __PITURNEND__ -e __PIWATCH__" "Pi secondmate launch template does not include both primary extensions" + assert_contains "$text" "\$PROJ_ABS/.pi/extensions/fm-primary-pi-watch.ts" "fm-spawn does not point the Pi secondmate watch placeholder at the tracked extension" + assert_not_contains "$text" "fm-pi-watch-extension.sh" "fm-spawn should no longer generate the Pi watch extension before launch" + assert_contains "$text" "__PITURNEND__" "fm-spawn does not replace the Pi turn-end guard extension placeholder" + assert_contains "$text" "__PIWATCH__" "fm-spawn does not replace the Pi watch extension placeholder" + pass "Pi secondmate launch wiring includes both tracked primary extensions" +} + test_pi_extension_reports_external_healthy_watcher() { local repo home plugin out status repo="$TMP_ROOT/pi-external-healthy-root" @@ -1176,6 +1234,26 @@ EOF pass "Pi process-exit cleanup stops the attached arm child" } +test_opencode_primary_watch_plugin_static_wiring() { + local plugin module_boundary text + plugin="$ROOT/.opencode/plugins/fm-primary-watch-arm.js" + module_boundary="$ROOT/.opencode/plugins/package.json" + assert_present "$plugin" "OpenCode primary watch plugin missing" + assert_present "$module_boundary" "OpenCode plugin ESM package boundary missing" + assert_contains "$(cat "$module_boundary")" '"type": "module"' "OpenCode plugin package boundary is not explicitly ESM" + text=$(cat "$plugin") + assert_contains "$text" "session.idle" "OpenCode plugin does not listen for session.idle" + assert_contains "$text" "fm-watch-arm.sh" "OpenCode plugin does not spawn the watcher arm" + assert_contains "$text" "promptAsync" "OpenCode plugin does not wake with promptAsync" + assert_contains "$text" 'encodeFirstmateOperationalInput' "OpenCode plugin does not construct typed synthetic user-role wakes" + assert_contains "$text" ".fm-secondmate-home" "OpenCode plugin does not scope out secondmate homes" + assert_contains "$text" "rev-parse\", \"--git-dir" "OpenCode plugin does not check linked worktree scope" + assert_contains "$text" "sessionOwnsLock" "OpenCode plugin does not gate arm attempts on the session lock" + assert_contains "$text" 'fm-watch-arm.sh" --restart' "OpenCode plugin does not restart into its own watcher child" + assert_contains "$text" 'setArmStatus("external")' "OpenCode plugin still treats an external healthy watcher as armed" + pass "OpenCode primary watcher plugin has the verified TUI wake wiring" +} + test_opencode_plugin_package_boundary_is_explicit_esm() { local fixture plugin out status fixture="$TMP_ROOT/opencode-esm-boundary/.opencode" @@ -2124,6 +2202,8 @@ EOF pass "OpenCode healthy arm output does not suppress the turn-end guard" } +test_tracked_extension_present_and_self_hashing +test_spawn_template_mentions_pi_watch_placeholder test_pi_extension_reports_external_healthy_watcher test_pi_tool_returns_agent_tool_result test_pi_redundant_tool_call_is_owned_noop @@ -2139,6 +2219,7 @@ test_pi_arm_distinguishes_session_lock_ownership test_pi_session_transition_generation_owner test_pi_process_exit_cleanup_listener_lifecycle test_pi_process_exit_cleanup_stops_arm_child +test_opencode_primary_watch_plugin_static_wiring test_opencode_plugin_package_boundary_is_explicit_esm test_opencode_primary_watch_plugin_uses_effective_state_home test_opencode_primary_watch_plugin_sources_effective_config From 717f87b6fac670ea710aba4f6dc854c409c35609 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Tue, 28 Jul 2026 00:37:11 -0700 Subject: [PATCH 20/52] feat: route crew dispatch using quota-window pace (#1172) * Consume quota-axi pace signals in dispatch profile array selection. Add quota-array-dispatch as the single owner of the pace-aware candidate choice, keep AGENTS.md to the intake boundary and load trigger, and cover the acceptance cases with sanitized schemaVersion 3 fixtures. * no-mistakes(review): Stop and report genuine quota dispatch ties * no-mistakes(document): Document quota pace freshness and uncertainty --- .agents/skills/harness-adapters/SKILL.md | 1 + .agents/skills/quota-array-dispatch/SKILL.md | 175 ++++++++--- AGENTS.md | 6 +- bin/fm-bootstrap.sh | 3 +- docs/architecture.md | 2 +- docs/configuration.md | 7 +- docs/documentation-audiences.json | 4 + docs/examples/crew-dispatch.json | 2 +- .../fixtures/quota-array-dispatch/cases.json | 42 --- tests/fm-instruction-owners.test.sh | 16 +- tests/fm-quota-array-dispatch.test.sh | 278 ++++++++++++++++++ 11 files changed, 446 insertions(+), 90 deletions(-) create mode 100755 tests/fm-quota-array-dispatch.test.sh diff --git a/.agents/skills/harness-adapters/SKILL.md b/.agents/skills/harness-adapters/SKILL.md index c5ba453f95b..9ea4112153c 100644 --- a/.agents/skills/harness-adapters/SKILL.md +++ b/.agents/skills/harness-adapters/SKILL.md @@ -12,6 +12,7 @@ Use this reference before any harness-specific firstmate operation: spawn, recov Crewmates default to the same harness firstmate is running on unless `config/crew-harness` records an adapter name. Optional dispatch profiles in `config/crew-dispatch.json` can override that static default for one crewmate or scout dispatch by selecting concrete harness, model, and effort axes at intake. +When a matched rule or default is a profile array, load `quota-array-dispatch` for the pace-aware candidate choice after this skill establishes harness and model/provider facts. The captain may override that file at session start or later; a per-task instruction such as "run this one on codex" overrides it for that dispatch only. `default` means mirror firstmate's own harness. diff --git a/.agents/skills/quota-array-dispatch/SKILL.md b/.agents/skills/quota-array-dispatch/SKILL.md index c384553a859..a5fe06d6ec6 100644 --- a/.agents/skills/quota-array-dispatch/SKILL.md +++ b/.agents/skills/quota-array-dispatch/SKILL.md @@ -12,54 +12,159 @@ metadata: # quota-array-dispatch This skill is the single owner of the pace-aware profile-array selection procedure. -`AGENTS.md` section 4 owns the always-loaded intake boundary, load trigger, malformed-config refusal, every-candidate accounting, and strongest-reasoning/tie safety rules. -`harness-adapters` owns harness verification, model/provider discovery, and effort fallback. +The concise always-loaded intake boundary remains in `AGENTS.md` section 4. +`docs/configuration.md` owns the `config/crew-dispatch.json` schema only. `quota-axi` remains data-only and never recommends a route. +Firstmate owns the judgment. Do not add a daemon, opaque composite score, routing wrapper, hard-coded model-specific policy, or producer-side route recommendation. -## Collect facts +## When to load -Run `quota-axi --json` once per intake and reuse that snapshot for every candidate. -For each candidate, preserve explicit `harness`, `model`, and `provider`; `harness-adapters` owns identity, and model/provider never infer harness: +Load this skill whenever a matched dispatch rule or the configured default resolves to a profile array (more than one candidate), before choosing the concrete `--harness`, `--model`, and `--effort` passed to `fm-spawn`. +Keep using `harness-adapters` for harness verification, model/provider discovery, and effort fallback. -- task/profile fit and required reasoning class -- raw applicable headroom (`effectivePercentRemaining` or tightest applicable percentage) -- effective pace, signed reserve per window, and worst reserve (`worstReservePercentPoints` or minimum signed reserve) -- whether applicable windows/summary are ahead, or pace is `unknown` -- schema note when pace fields are absent +## Intake boundary this skill does not relax -Stale raw windows are diagnostic, never headroom. -Read all windows named by `boundedBy`, `limitingWindowIds`, `aheadWindowIds`, `behindWindowIds`, `onPaceWindowIds`, and `unknownWindowIds`. +1. Explicit per-task captain overrides still win over configured profiles. +2. Configured profile matching precedence is unchanged: best-fit rule, then configured default, then static crewmate harness. +3. Malformed `config/crew-dispatch.json` remains an actionable error; never select around it. +4. Every configured candidate in the matched array must be accounted for. +5. If any harness/model/provider relationship, applicable quota data, or interpretation cannot be established, stop and report that candidate instead of omitting it, guessing, falling back, or calling the result quota-informed. +6. When every candidate is tight, preserve the captain's strongest-reasoning class rather than silently downgrading it solely to conserve quota; stop and report the tight choice if that class cannot proceed. +7. Genuine ties must remain free of array-order or harness bias. -## Pace semantics +## Collect inspectable facts for every candidate -`reservePercentPoints = percentRemaining - timeRemainingPercent`. -Negative reserve means usage is ahead of reset pace and creates conservation pressure. -Positive reserve means usage is behind reset pace. -`on_pace` is neutral. -Conservation pressure is present for effective pace status `ahead`, effective pace status is `mixed` and any `aheadWindowIds` remain, or a bounding window is `ahead`. -`unknown` is valid explicit uncertainty from quota-axi, not parser failure or permission to assume health. +For each candidate profile: -## Selection order +1. Establish the harness/model/provider relationship from current authoritative discovery owned by `harness-adapters`. + Fail loudly on an unresolved relationship. +2. Run `quota-axi --json` once per intake and reuse that snapshot for every candidate. +3. Require a current provider report with known quota semantics and a known applicable effective-availability record for that candidate's provider and model scope. + Stale raw windows remain diagnostic evidence only and are never current headroom. +4. Read every bounding window relevant to that candidate, including windows named by `boundedBy`, `limitingWindowIds`, `aheadWindowIds`, `behindWindowIds`, `onPaceWindowIds`, and `unknownWindowIds` on the effective record. +5. Record these inspectable facts, never a hidden score: + - task/profile fit + - reasoning class required by the captain request or task ambiguity + - raw applicable headroom (`effectivePercentRemaining` or the tightest applicable remaining percentage) + - effective pace status when present + - signed reserve for each applicable window and the effective worst reserve when present + - whether any applicable window or effective summary is ahead of reset + - whether any applicable pace is `unknown` + - schema compatibility note when pace fields are absent -Apply only among candidates satisfying required fit and strongest reasoning class. -Never use pace or raw headroom to silently replace that reasoning class. +## Pace signals -1. Unresolved relationship or quota: stop and report the tuple and concrete evidence. -2. All-tight: keep strongest reasoning; dispatch inside it or report if blocked. -3. Comparable fit/reasoning: prefer no ahead pressure over pressure, even with higher raw headroom. -4. Among pressured candidates, prefer the least-negative worst applicable reserve. -5. Sustainable candidates: use known pace plus raw headroom. - Prefer known sustainable evidence over `unknown` when comparable. +quota-axi `schemaVersion` 3 window pace uses: + +- `reservePercentPoints = percentRemaining - timeRemainingPercent` +- Negative reserve means usage is ahead of reset pace and creates conservation pressure. +- Positive reserve means usage is behind reset pace. +- `on_pace` is neutral. + +Effective-availability pace summaries may report `ahead`, `behind`, `on_pace`, `mixed`, or `unknown`. + +Treat conservation pressure as present when: + +- effective pace status is `ahead`, or +- effective pace status is `mixed` and any `aheadWindowIds` remain, or +- any applicable bounding window itself has pace status `ahead`. + +An effective `mixed` result is never healthy merely because one window is behind. +Any remaining `aheadWindowIds` keep conservation pressure. + +Signed reserve comparison uses the worst applicable reserve, preferring the producer field `worstReservePercentPoints` when present and otherwise the minimum signed reserve across applicable bounding windows. + +## Selection procedure + +Apply these steps only among candidates that already satisfy required task/profile fit and the strongest reasoning class the request genuinely needs. +Never use pace or raw headroom to silently replace that reasoning class with a weaker one. + +1. **Unresolved relationship or quota data** + Stop and report the blocked candidate. +2. **Strongest-reasoning / all-tight** + If every remaining candidate is tight, keep the strongest-reasoning class and either dispatch inside that class or stop and report that the tight choice cannot proceed. + Do not conserve quota through an unapproved downgrade. +3. **Conservation pressure vs sustainable pace** + When fit and reasoning class are comparable, prefer a candidate without ahead-of-reset conservation pressure over one with conservation pressure, even when the pressured candidate has somewhat higher raw remaining percentage. +4. **Among pressured candidates** + Prefer the least-negative worst applicable reserve. + Example: worst reserve `-4` is safer than `-18` when other inspectable facts are comparable. +5. **Among sustainable candidates** + Use known behind/on-pace evidence plus raw headroom transparently. Do not collapse those facts into an opaque composite score. -6. If unresolved pace changes the choice, report uncertainty. -7. Absent pace or older schema: do not crash, fabricate pace, or treat absence as healthy/`on_pace`. - Compare raw headroom only, state pace is unavailable, and keep safety rules. -8. Genuine ties: stop and report every tied candidate for captain choice. + Prefer known sustainable evidence over `unknown` pace when otherwise comparable. + Between known sustainable candidates, prefer the clearly better inspectable pair of pace reserve and raw headroom; state both facts in the choice rationale. +6. **Unknown pace** + `unknown` is valid explicit uncertainty from quota-axi, not a parser failure and not permission to assume the window is healthy or exhausted. + Inspect `unknownWindowIds` and each window's pace `reason` so the rationale preserves the producer's stated uncertainty. + Prefer known sustainable evidence when otherwise comparable. + If the dispatch choice materially hinges on unresolved pace, report the uncertainty rather than inventing a conclusion. +7. **Absent pace / older schema** + `schemaVersion` 2 payloads or missing pace fields must degrade explicitly and safely. + Do not crash, fabricate pace, or silently reinterpret absence as healthy/`on_pace`. + Compare raw applicable headroom only, using known effective availability rather than stale or isolated window percentages, state that pace is unavailable, and keep every other safety rule above. +8. **Genuine ties** + If every inspectable selection fact is equal, stop and report every tied candidate for captain choice. Do not select by array order, harness name, or another arbitrary identity ordering. Report duplicate concrete profiles as a configuration error. -Name the inspectable facts used for every candidate. -After selecting, check auth only through that tuple's surface; another harness CLI cannot block it. -A blocked credential report must name `harness`, `model`, authentication surface, and concrete failure evidence; never emit a bare `Grok unauthenticated` statement. +The intake rationale must name the inspectable facts used for every candidate. Never conclude with an unexplained "best quota" label. + +## Acceptance scenarios + +These scenarios are normative examples of the procedure above. + +### Higher raw quota but materially ahead vs lower raw quota on/behind pace + +Candidate A has higher `effectivePercentRemaining` but conservation pressure from an ahead bounding window. +Candidate B has lower raw headroom, no conservation pressure, and known behind or on-pace evidence. +Choose B when fit and reasoning class are comparable. + +### Mixed effective pace with an ahead bound + +Effective pace status is `mixed` and `aheadWindowIds` is non-empty. +Treat the candidate as conservation-pressured even if another window is behind or on pace. + +### Both candidates ahead with different worst reserves + +Both candidates have conservation pressure. +Choose the least-negative worst applicable reserve when fit and reasoning class are comparable. + +### Known sustainable versus unknown + +Candidate A has known behind or on-pace evidence. +Candidate B has comparable fit, reasoning class, and raw headroom but `unknown` pace. +Prefer A. +If the only way to prefer one side depends on unresolved pace and no known sustainable candidate remains, report the uncertainty. + +### Every candidate tight while strongest-reasoning applies + +All candidates are tight on real headroom. +Keep the strongest reasoning class required by the request. +Do not pick a weaker class only to save quota. +Dispatch inside that class or stop and report that the tight strongest-class choice cannot proceed. + +### Genuine tie without array-order or harness bias + +Two candidates match on fit, reasoning class, conservation pressure, worst reserve, pace class, raw headroom, and unknown flags. +Choosing either array order or a standing harness preference is forbidden. +Stop and report both tied candidates for captain choice. + +### schemaVersion 2 or absent-pace compatibility + +Older quota-axi output or missing pace fields still allow array resolution. +Compare raw headroom only, state that pace is unavailable, and do not invent ahead/behind/on_pace. + +## Sanitized producer shape + +Validate consumers against a sanitized `schemaVersion` 3 shape derived from quota-axi 0.1.15: + +- top level: `schemaVersion`, `generatedAt`, `providers[]` +- each provider: `provider`, `state`, `windows[]`, and optional `quotaSemantics` with `status` and `effectiveAvailability[]` +- each window: `id`, `label`, `kind`, and optional `percentRemaining` and `pace`; pace has `status` plus optional `reason`, `timeRemainingPercent`, and `reservePercentPoints` +- each effective-availability entry: `scope`, `status`, `boundedBy`, optional `effectivePercentRemaining`, optional `limitingWindowIds`, and optional pace summary +- each effective pace summary: `status` plus optional `aheadWindowIds`, `behindWindowIds`, `onPaceWindowIds`, `unknownWindowIds`, `worstReservePercentPoints`, and `worstReserveWindowId` + +Never persist live provider balances, reset timestamps, account identifiers, or other private account details in tracked fixtures. diff --git a/AGENTS.md b/AGENTS.md index d85e90b8eaa..f838dfb27ca 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -164,12 +164,13 @@ If static `config/crew-harness` or `config/secondmate-harness` names an unverifi `docs/configuration.md` owns dispatch-profile and runtime-backend schemas, `bin/fm-harness.sh` owns static resolution, and `bin/fm-spawn.sh` owns launch flags and fail-closed validation. When dispatch profiles exist, consult them at every crewmate or scout intake and pass the resolved concrete profile required by `fm-spawn`. Routing precedence is an explicit per-task captain override, then the best-fit configured rule, then the configured default, then the static crewmate harness. -Firstmate alone resolves a matched profile array: run `quota-axi --json` at that intake, evaluate every configured candidate against that current output, and choose the candidate with the most real headroom. +Firstmate alone resolves a matched profile array: run `quota-axi --json` at that intake, evaluate every configured candidate against that current output, and choose with inspectable real headroom including quota-window pace. Account for every candidate; if any harness/model/provider relationship, applicable quota data, or interpretation cannot be established, stop and report that candidate instead of omitting it, guessing, falling back, or calling the result quota-informed. Preserve malformed profile configuration as an actionable error rather than selecting around it. When every candidate is tight, preserve the captain's strongest-reasoning class rather than silently downgrading it solely to conserve quota; stop and report the tight choice if that class cannot proceed. Break genuine headroom ties without array-order or harness bias. -`quota-axi` owns how model or product windows relate to bounding account windows. +`quota-axi` owns how model or product windows relate to bounding account windows and remains data-only. +Load `quota-array-dispatch` before choosing among a matched profile array; that skill is the single owner of the pace-aware selection procedure. The generic effort fallback and its precedence are owned by `harness-adapters`: explicit captain and standing configured effort win; otherwise use low for well-understood explicit work, xhigh for ambiguous investigation or design, intermediate levels proportionally, and never max without explicit captain preference. Do not add model-specific versions of that policy. @@ -472,6 +473,7 @@ These skills are not captain-invocable; load them only at their precise triggers - `bootstrap-diagnostics` - load whenever the session-start digest's bootstrap section prints an actionable diagnostic line (`MISSING:`, `MISSING_MANUAL:`, `BACKEND_INVALID:`, `NEEDS_GH_AUTH`, `TANGLE:`, `CREW_DISPATCH: invalid`, `FLEET_SYNC:`, `PR_CHECK_MIGRATION:`, `SECONDMATE_SYNC:`, `SECONDMATE_LIVENESS:`, `NUDGE_SECONDMATES:`, or `FMX:`); silence and `BOOTSTRAP_INFO:` need no load. - `diagnostic-reasoning` - load before scoping a reported bug and before acting on a diagnostic report. - `ask-user-authority` - load before deciding any ask-user finding, regardless of the project's `yolo` posture. +- `quota-array-dispatch` - load before choosing among a matched crew-dispatch profile array from current quota-axi output. - `harness-adapters` - load before spawning or recovering a crewmate or secondmate, handling a trust dialog, sending a harness-specific skill invocation, interrupting or exiting an agent, resuming an exited agent, or verifying a new harness adapter. - `firstmate-orca` - load before switching to Orca, spawning or supervising Orca-backed work, smoke-testing Orca backend behavior, debugging Orca task state, or reconciling Orca-backed task metadata. - `project-management` - load before adding, creating, removing, or initializing a project. diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index 8685b2e2bbd..c86b7e839ab 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -53,7 +53,8 @@ # with update --archive-body and mv [<id>...]); an installed but # incompatible build reports MISSING like no-mistakes. A compatible # tasks-axi default backend is silent. quota-axi is required for the -# agent-owned dispatch-profile array procedure in AGENTS.md section 4. +# agent-owned dispatch-profile array procedure in AGENTS.md section 4 +# and .agents/skills/quota-array-dispatch/SKILL.md. # X mode is OPTIONAL and inert unless FM_HOME/.env has a non-empty # FMX_PAIRING_TOKEN. When opted in, bootstrap requires curl+jq, writes # the relay poll shim and 30s cadence config, and prints an FMX line. diff --git a/docs/architecture.md b/docs/architecture.md index 4840ef77a02..da1519a0d44 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -142,7 +142,7 @@ The intake and authority contract in `AGENTS.md` owns when separate scout resear ## Dispatch profiles Crewmate and scout dispatch can stay on the static crewmate harness resolved by `config/crew-harness`, or it can use local dispatch profiles in `config/crew-dispatch.json`. -The dispatch file is intentionally judgment-based: firstmate reads the natural-language rules at intake, chooses the best matching rule, resolves profile arrays itself from current quota output under `AGENTS.md` section 4, and passes only concrete `--harness`, `--model`, and `--effort` axes to `fm-spawn.sh`. +The dispatch file is intentionally judgment-based: firstmate reads the natural-language rules at intake, chooses the best matching rule, resolves profile arrays itself from current quota output under the `AGENTS.md` section 4 intake boundary and the `quota-array-dispatch` selection procedure, and passes only concrete `--harness`, `--model`, and `--effort` axes to `fm-spawn.sh`. The shell scripts validate the JSON shape and verified harness/effort combinations, but they do not parse task intent, match natural-language rules, or own array selection. The session-start bootstrap step keeps valid dispatch configuration silent unless verbose facts are enabled and surfaces a concise invalid-config line when validation fails. When the file exists, `fm-spawn.sh` refuses crewmate and scout launches without an explicit harness, so `config/crew-harness` is only automatic when no dispatch profile file is active. diff --git a/docs/configuration.md b/docs/configuration.md index fed683343ea..7f23866c334 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -207,11 +207,12 @@ For Pi and pi-signed secondmate launches, `fm-spawn.sh` starts the selected exec ## Crew dispatch profiles (config/crew-dispatch.json) `config/crew-dispatch.json` is an optional local, gitignored file containing natural-language rules that firstmate reads before dispatching a crewmate or scout. -The shell scripts do not match those rules; firstmate chooses the best matching rule with judgment, resolves its profile object or array under the operating contract in `AGENTS.md` section 4, and passes only concrete `--harness`, `--model`, and `--effort` flags to `fm-spawn.sh`. +The shell scripts do not match those rules; firstmate chooses the best matching rule with judgment, resolves its profile object or array under the operating contract in `AGENTS.md` section 4 and `quota-array-dispatch`, and passes only concrete `--harness`, `--model`, and `--effort` flags to `fm-spawn.sh`. When the file exists, `fm-spawn.sh` enforces that contract by refusing crewmate and scout spawns that lack an explicit harness (`--harness`, a positional adapter, or a raw launch command). Batch spawns satisfy the same requirement with a shared `--harness`. Secondmate spawns are exempt and still resolve through `config/secondmate-harness` and its optional model and effort tokens. -This section is the single owner of the canonical schema and its per-field semantics; `AGENTS.md` section 4 owns the dispatch and array-selection procedure. +This section is the single owner of the canonical schema and its per-field semantics. +`AGENTS.md` section 4 owns the always-loaded dispatch intake boundary, and `quota-array-dispatch` owns the pace-aware profile-array selection procedure. ```json { @@ -235,7 +236,7 @@ Both `use` and the optional top-level `default` accept either one profile object The single-object form stays fully backward-compatible, and every profile needs `harness`. Profile `model` and `effort` fields and rule `why` are optional. An omitted model or effort means the selected harness uses its own default for that axis. -Every profile array is an implicit quota-aware choice. +Every profile array is an implicit quota-aware choice resolved through `quota-array-dispatch`. If no dispatch rule fits, firstmate resolves `default` through the same object-or-array path before falling back to `config/crew-harness`. If a selected profile carries an effort value the chosen harness does not accept, `fm-spawn.sh` records the requested `effort=` in task meta for traceability but omits the launch flag, and bootstrap reports the invalid harness/effort pair as a `CREW_DISPATCH` diagnostic when it is visible in the file. See [`docs/examples/crew-dispatch.json`](examples/crew-dispatch.json) for a starting point to copy into local `config/crew-dispatch.json`. diff --git a/docs/documentation-audiences.json b/docs/documentation-audiences.json index 60773b0d451..54b2190f6c8 100644 --- a/docs/documentation-audiences.json +++ b/docs/documentation-audiences.json @@ -155,6 +155,10 @@ "path": ".agents/skills/project-management/SKILL.md", "audience": "agent-runtime" }, + { + "path": ".agents/skills/quota-array-dispatch/SKILL.md", + "audience": "agent-runtime" + }, { "path": ".agents/skills/secondmate-provisioning/SKILL.md", "audience": "agent-runtime" diff --git a/docs/examples/crew-dispatch.json b/docs/examples/crew-dispatch.json index 4c8fc36993c..23a5391d20a 100644 --- a/docs/examples/crew-dispatch.json +++ b/docs/examples/crew-dispatch.json @@ -16,7 +16,7 @@ { "harness": "claude", "model": "claude-sonnet-5", "effort": "high" }, { "harness": "codex", "model": "gpt-5.5", "effort": "high" } ], - "why": "Firstmate compares every candidate with current relevant quota before dispatch, so use a strong coding profile." + "why": "Firstmate compares every candidate with current relevant quota and pace before dispatch, so use a strong coding profile." } ], "default": [ diff --git a/tests/fixtures/quota-array-dispatch/cases.json b/tests/fixtures/quota-array-dispatch/cases.json index c6fc3c3a867..23d097be463 100644 --- a/tests/fixtures/quota-array-dispatch/cases.json +++ b/tests/fixtures/quota-array-dispatch/cases.json @@ -199,48 +199,6 @@ } ] }, - { - "id": "select-pi-xai-before-authentication", - "expect": "pi-xai", - "reason": "an unauthenticated standalone Grok candidate cannot block selected authenticated Pi/xAI", - "candidates": [ - { - "id": "pi-xai", - "harness": "pi", - "model": "xai/grok-4.5", - "provider": "xai", - "authenticationSurface": "Pi xAI OAuth", - "authAvailable": true, - "fit": "comparable", - "reasoningClass": "strong", - "tight": false, - "rawHeadroom": 55, - "paceStatus": "behind", - "aheadWindowIds": [], - "worstReserve": 15.0, - "unknownPace": false, - "paceAvailable": true - }, - { - "id": "standalone-grok", - "harness": "grok", - "model": "grok-4.5", - "provider": "grok", - "authenticationSurface": "Grok Build CLI", - "authAvailable": false, - "authFailure": "Grok Build CLI login missing", - "fit": "comparable", - "reasoningClass": "strong", - "tight": false, - "rawHeadroom": 80, - "paceStatus": "ahead", - "aheadWindowIds": ["weekly"], - "worstReserve": -12.0, - "unknownPace": false, - "paceAvailable": true - } - ] - }, { "id": "all-tight-strongest-reasoning", "expect": "A", diff --git a/tests/fm-instruction-owners.test.sh b/tests/fm-instruction-owners.test.sh index f55f3e905f9..754e00ddc83 100755 --- a/tests/fm-instruction-owners.test.sh +++ b/tests/fm-instruction-owners.test.sh @@ -109,13 +109,15 @@ test_agent_owned_quota_array_dispatch_contract() { 'Firstmate alone resolves a matched profile array' \ 'run `quota-axi --json` at that intake' \ 'evaluate every configured candidate against that current output' \ - 'choose the candidate with the most real headroom' \ + 'inspectable real headroom including quota-window pace' \ 'if any harness/model/provider relationship, applicable quota data, or interpretation cannot be established, stop and report that candidate' \ 'instead of omitting it, guessing, falling back, or calling the result quota-informed' \ 'Preserve malformed profile configuration as an actionable error' \ "preserve the captain's strongest-reasoning class rather than silently downgrading it" \ 'Break genuine headroom ties without array-order or harness bias' \ - '`quota-axi` owns how model or product windows relate to bounding account windows'; do + '`quota-axi` owns how model or product windows relate to bounding account windows' \ + 'remains data-only' \ + 'Load `quota-array-dispatch` before choosing among a matched profile array'; do assert_grep "$phrase" "$AGENTS" "array-dispatch contract lost '$phrase'" done @@ -131,12 +133,16 @@ test_agent_owned_quota_array_dispatch_contract() { done assert_grep 'not as a permanent namespace or provider mapping' "$HARNESS" \ "model discovery guidance permits a fixed provider table" - assert_grep '`AGENTS.md` section 4 owns the dispatch and array-selection procedure.' "$CONFIG" \ - "configuration docs do not point to the agent-owned array procedure" + assert_grep 'load `quota-array-dispatch` for the pace-aware candidate choice' "$HARNESS" \ + "harness-adapters lost the quota-array-dispatch handoff" + assert_grep '`quota-array-dispatch` owns the pace-aware profile-array selection procedure' "$CONFIG" \ + "configuration docs do not point to quota-array-dispatch" assert_grep 'quota-axi is required for the' "$BOOTSTRAP" \ "bootstrap docs lost the quota-axi dependency pointer" - assert_grep 'agent-owned dispatch-profile array procedure in AGENTS.md section 4.' "$BOOTSTRAP" \ + assert_grep 'agent-owned dispatch-profile array procedure in AGENTS.md section 4' "$BOOTSTRAP" \ "bootstrap docs do not point to the agent-owned array procedure" + assert_grep 'quota-array-dispatch/SKILL.md' "$BOOTSTRAP" \ + "bootstrap docs do not point to quota-array-dispatch" pass "firstmate directly compares every quota candidate with authoritative model discovery" } diff --git a/tests/fm-quota-array-dispatch.test.sh b/tests/fm-quota-array-dispatch.test.sh new file mode 100755 index 00000000000..a958e56c306 --- /dev/null +++ b/tests/fm-quota-array-dispatch.test.sh @@ -0,0 +1,278 @@ +#!/usr/bin/env bash +# Contract and deterministic fixture tests for quota-array-dispatch. +# +# The skill owns the agent-facing decision procedure. +# This test encodes the same inspectable comparison rules against sanitized +# fixtures so acceptance cases stay deterministic without introducing a +# production routing wrapper. +# shellcheck disable=SC2016 +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +AGENTS="$ROOT/AGENTS.md" +OWNER="$ROOT/.agents/skills/quota-array-dispatch/SKILL.md" +HARNESS="$ROOT/.agents/skills/harness-adapters/SKILL.md" +CONFIG="$ROOT/docs/configuration.md" +ARCHITECTURE="$ROOT/docs/architecture.md" +BOOTSTRAP="$ROOT/bin/fm-bootstrap.sh" +AUDIENCES="$ROOT/docs/documentation-audiences.json" +CASES="$ROOT/tests/fixtures/quota-array-dispatch/cases.json" +SHAPE="$ROOT/tests/fixtures/quota-array-dispatch/schema-v3-shape.json" + +intake_boundary() { + awk ' + /^## 4\. Harness and runtime dispatch$/ { found = 1; next } + found && /^## 5\. Recovery$/ { exit } + found { print } + ' "$AGENTS" +} + +select_candidate_py() { + python3 - "$@" <<'PY' +import json, sys + +def conservation_pressure(c): + if not c.get("paceAvailable", True): + return False + status = c.get("paceStatus") + ahead_ids = c.get("aheadWindowIds") or [] + bounding_windows = c.get("boundingWindows") or [] + if status == "ahead": + return True + if status == "mixed" and ahead_ids: + return True + if any(window.get("paceStatus") == "ahead" for window in bounding_windows): + return True + return False + +def select(case): + required = case.get("requiredReasoningClass") + cands = list(case["candidates"]) + if required: + matching = [c for c in cands if c.get("reasoningClass") == required] + if not matching: + return {"error": "required reasoning class unavailable"} + # Strongest-reasoning rule: never drop to a weaker class for quota. + cands = matching + + # Fit filter: fixtures mark comparable; keep only comparable for these cases. + cands = [c for c in cands if c.get("fit") == "comparable"] + if not cands: + return {"error": "no comparable candidates"} + + def sort_key(c): + pressured = conservation_pressure(c) + unknown = bool(c.get("unknownPace")) or c.get("paceStatus") == "unknown" + pace_available = bool(c.get("paceAvailable", True)) + reserve = c.get("worstReserve") + if reserve is None: + reserve_key = float("-inf") + else: + reserve_key = float(reserve) + raw = float(c.get("rawHeadroom") or 0) + # Sort ascending by preference rank components that python min understands + # via a tuple where lower is better only for pressure/unknown flags. + return ( + 1 if pressured else 0, + 1 if (unknown and pace_available) else 0, + 0 if pace_available else 1, # when pace absent, still comparable via raw only + # Among pressured: least-negative reserve => higher reserve first => negate + (-reserve_key if pressured else 0), + # Among sustainable with pace: prefer higher reserve then higher raw + (-reserve_key if (not pressured and pace_available and not unknown) else 0), + -raw, + ) + + # Special-case all-tight already constrained to required class above. + best_key = min(sort_key(c) for c in cands) + winners = [c for c in cands if sort_key(c) == best_key] + if len(winners) > 1: + return { + "error": "genuine tie requires captain choice", + "candidates": sorted(c["id"] for c in winners), + } + winner = winners[0] + return { + "id": winner["id"], + "pressured": conservation_pressure(winner), + } + +case = json.loads(sys.argv[1]) +print(json.dumps(select(case))) +PY +} + +test_owner_and_always_loaded_boundary() { + local boundary trigger_count + boundary=$(intake_boundary) + + assert_present "$OWNER" "quota-array-dispatch owner is missing" + assert_grep 'name: quota-array-dispatch' "$OWNER" "quota-array-dispatch skill has the wrong name" + assert_grep 'user-invocable: false' "$OWNER" "quota-array-dispatch skill must be agent-only" + assert_grep 'single owner of the pace-aware profile-array selection procedure' "$OWNER" \ + "quota-array-dispatch skill does not declare ownership" + + assert_contains "$boundary" 'Firstmate alone resolves a matched profile array' \ + "intake boundary lost agent-owned array resolution" + assert_contains "$boundary" 'run `quota-axi --json` at that intake' \ + "intake boundary lost quota-axi intake read" + assert_contains "$boundary" 'evaluate every configured candidate against that current output' \ + "intake boundary lost full-candidate accounting" + assert_contains "$boundary" 'inspectable real headroom including quota-window pace' \ + "intake boundary lost pace-aware headroom wording" + assert_contains "$boundary" 'if any harness/model/provider relationship, applicable quota data, or interpretation cannot be established, stop and report that candidate' \ + "intake boundary lost unresolved-candidate refusal" + assert_contains "$boundary" 'instead of omitting it, guessing, falling back, or calling the result quota-informed' \ + "intake boundary lost no-guess wording" + assert_contains "$boundary" 'Preserve malformed profile configuration as an actionable error' \ + "intake boundary lost malformed-config refusal" + assert_contains "$boundary" "preserve the captain's strongest-reasoning class rather than silently downgrading it" \ + "intake boundary lost strongest-reasoning rule" + assert_contains "$boundary" 'Break genuine headroom ties without array-order or harness bias' \ + "intake boundary lost genuine-tie rule" + assert_contains "$boundary" '`quota-axi` owns how model or product windows relate to bounding account windows' \ + "intake boundary lost quota-axi window ownership" + assert_contains "$boundary" 'remains data-only' \ + "intake boundary lost data-only producer boundary" + assert_contains "$boundary" 'Load `quota-array-dispatch` before choosing among a matched profile array' \ + "intake boundary lost quota-array-dispatch load trigger" + + trigger_count=$(grep -Fc -- '- `quota-array-dispatch` -' "$AGENTS") + [ "$trigger_count" -eq 1 ] || fail "quota-array-dispatch must have exactly one section 13 trigger, found $trigger_count" + + # Full pace procedure stays out of AGENTS.md. + if printf '%s\n' "$boundary" | grep -q 'reservePercentPoints'; then + fail "AGENTS.md intake boundary duplicated pace formula detail" + fi + if printf '%s\n' "$boundary" | grep -q 'aheadWindowIds'; then + fail "AGENTS.md intake boundary duplicated aheadWindowIds detail" + fi + + pass "quota-array-dispatch has one conditional owner and a concise always-loaded boundary" +} + +test_owner_contains_acceptance_procedure() { + local phrase + for phrase in \ + 'reservePercentPoints = percentRemaining - timeRemainingPercent' \ + 'Negative reserve means usage is ahead of reset pace and creates conservation pressure' \ + 'Positive reserve means usage is behind reset pace' \ + '`on_pace` is neutral' \ + 'effective pace status is `mixed` and any `aheadWindowIds` remain' \ + 'prefer a candidate without ahead-of-reset conservation pressure over one with conservation pressure' \ + 'even when the pressured candidate has somewhat higher raw remaining percentage' \ + 'Prefer the least-negative worst applicable reserve' \ + 'Use known behind/on-pace evidence plus raw headroom transparently' \ + 'Do not collapse those facts into an opaque composite score' \ + '`unknown` is valid explicit uncertainty from quota-axi' \ + 'Prefer known sustainable evidence over `unknown` pace when otherwise comparable' \ + 'If the dispatch choice materially hinges on unresolved pace, report the uncertainty' \ + 'Do not crash, fabricate pace, or silently reinterpret absence as healthy' \ + 'stop and report every tied candidate for captain choice' \ + 'Do not select by array order, harness name, or another arbitrary identity ordering' \ + 'Do not add a daemon, opaque composite score, routing wrapper, hard-coded model-specific policy'; do + assert_grep "$phrase" "$OWNER" "quota-array-dispatch procedure lost '$phrase'" + done + + for phrase in \ + 'Higher raw quota but materially ahead vs lower raw quota on/behind pace' \ + 'Mixed effective pace with an ahead bound' \ + 'Both candidates ahead with different worst reserves' \ + 'Known sustainable versus unknown' \ + 'Every candidate tight while strongest-reasoning applies' \ + 'Genuine tie without array-order or harness bias' \ + 'schemaVersion 2 or absent-pace compatibility'; do + assert_grep "$phrase" "$OWNER" "acceptance scenario missing: $phrase" + done + pass "quota-array-dispatch owns the full pace procedure and acceptance scenarios" +} + +test_cross_references_stay_pointers() { + assert_grep '`quota-array-dispatch` owns the pace-aware profile-array selection procedure' "$CONFIG" \ + "configuration docs do not point to quota-array-dispatch" + assert_no_grep '`AGENTS.md` section 4 owns the dispatch and array-selection procedure.' "$CONFIG" \ + "configuration docs still claim AGENTS.md owns the full array-selection procedure" + assert_grep 'quota-array-dispatch' "$ARCHITECTURE" \ + "architecture docs lost the quota-array-dispatch pointer" + assert_grep 'quota-array-dispatch' "$BOOTSTRAP" \ + "bootstrap header lost the quota-array-dispatch pointer" + assert_grep 'load `quota-array-dispatch` for the pace-aware candidate choice' "$HARNESS" \ + "harness-adapters lost the array-selection handoff" + assert_grep '.agents/skills/quota-array-dispatch/SKILL.md' "$AUDIENCES" \ + "documentation audience inventory missing quota-array-dispatch" + pass "cross-references point at the single procedure owner" +} + +test_schema_v3_shape_fixture() { + python3 - "$SHAPE" <<'PY' || fail "schema v3 shape fixture is invalid" +import json, sys +path = sys.argv[1] +data = json.load(open(path)) +assert data.get("schemaVersion") == 3, data.get("schemaVersion") +assert isinstance(data.get("providers"), list) and data["providers"], "providers" +provider = data["providers"][0] +assert "windows" in provider and provider["windows"], "windows" +window = provider["windows"][0] +assert "pace" in window and "status" in window["pace"], window +eff = provider["quotaSemantics"]["effectiveAvailability"][0] +assert "pace" in eff and "status" in eff["pace"], eff +assert "effectivePercentRemaining" in eff +# Privacy: no live account residue markers. +blob = json.dumps(data) +for bad in ("sk-", "@", "Bearer ", "accountId", "organizationId"): + assert bad not in blob, bad +PY + pass "sanitized schemaVersion 3 fixture preserves producer pace shape without private details" +} + +test_deterministic_acceptance_cases() { + local raw case_json case_id expect expect_error got reason + raw=$(cat "$CASES") + while IFS= read -r case_json; do + case_id=$(python3 -c 'import json,sys; print(json.loads(sys.argv[1])["id"])' "$case_json") + expect=$(python3 -c 'import json,sys; print(json.loads(sys.argv[1]).get("expect", ""))' "$case_json") + expect_error=$(python3 -c 'import json,sys; print(json.loads(sys.argv[1]).get("expectError", ""))' "$case_json") + reason=$(python3 -c 'import json,sys; print(json.loads(sys.argv[1])["reason"])' "$case_json") + got=$(select_candidate_py "$case_json") + python3 -c ' +import json,sys +got=json.loads(sys.argv[1]) +expect=sys.argv[2] +expect_error=sys.argv[3] +case_id=sys.argv[4] +err=got.get("error") +if expect_error: + if err != expect_error: + raise SystemExit("%s: expected error %s, got %s" % (case_id, expect_error, got)) +elif err: + raise SystemExit("%s: selector error: %s" % (case_id, err)) +elif got.get("id") != expect: + raise SystemExit("%s: expected %s, got %s" % (case_id, expect, got)) +' "$got" "$expect" "$expect_error" "$case_id" \ + || fail "case $case_id failed ($reason); selector returned $got" + if [ -n "$expect_error" ]; then + pass "case $case_id -> $expect_error ($reason)" + else + pass "case $case_id -> $expect ($reason)" + fi + done < <(python3 -c 'import json,sys; data=json.load(sys.stdin); [print(json.dumps(c, separators=(",", ":"))) for c in data["cases"]]' <<<"$raw") +} + +test_no_duplicate_procedure_in_agents() { + # Guard against re-expanding the full procedure into AGENTS.md. + local count + count=$(grep -c 'conservation pressure' "$AGENTS" || true) + [ "$count" -eq 0 ] || fail "AGENTS.md should not restate conservation-pressure procedure detail" + count=$(grep -c 'worst applicable reserve' "$AGENTS" || true) + [ "$count" -eq 0 ] || fail "AGENTS.md should not restate worst-reserve procedure detail" + pass "AGENTS.md does not duplicate the pace procedure body" +} + +test_owner_and_always_loaded_boundary +test_owner_contains_acceptance_procedure +test_cross_references_stay_pointers +test_schema_v3_shape_fixture +test_deterministic_acceptance_cases +test_no_duplicate_procedure_in_agents From f36b94951fbe8cc9e7c9dbfca22cf600e8937bc7 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Tue, 28 Jul 2026 10:39:46 -0700 Subject: [PATCH 21/52] fix: adapt Grok Stop continuation and harden endpoint cleanup (#1171) * fix(grok): adapt Stop continuation to runtime capability * no-mistakes(review): Reject ambiguous Grok Stop payloads * no-mistakes(review): Reject duplicate Grok fields and accept spaced tmux sessions * no-mistakes(review): Enforce exact tmux cleanup selectors * no-mistakes(test): Fix historical tmux fixture and validate Grok Stop * no-mistakes: apply CI fixes --- .agents/skills/harness-adapters/SKILL.md | 13 +-- .claude/settings.json | 4 +- AGENTS.md | 2 +- bin/backends/tmux.sh | 18 +++- bin/fm-spawn.sh | 1 + bin/fm-test-run.sh | 5 +- bin/fm-turnend-guard.sh | 24 +++-- docs/architecture.md | 2 +- docs/configuration.md | 5 +- docs/turnend-guard.md | 21 +++-- docs/verification/runtime-backends.md | 28 ++++++ docs/verification/supervision.md | 22 ++++- tests/fm-backend.test.sh | 56 +----------- tests/fm-secondmate-liveness.test.sh | 2 +- tests/fm-turnend-guard.test.sh | 111 ++++++++++++++++++++++- tests/lib.sh | 11 ++- 16 files changed, 235 insertions(+), 90 deletions(-) diff --git a/.agents/skills/harness-adapters/SKILL.md b/.agents/skills/harness-adapters/SKILL.md index 9ea4112153c..429907041a3 100644 --- a/.agents/skills/harness-adapters/SKILL.md +++ b/.agents/skills/harness-adapters/SKILL.md @@ -54,7 +54,8 @@ Use that value for interrupt, exit, resume, and skill-invocation facts. The primary integrations for `claude`, `codex`, `opencode`, `pi`, `pi-signed`, and `grok` have empirically validated hook paths for the "no turn ends blind" guard. `claude` and `codex` block directly through Stop hooks that preserve exit status 2 and stderr from `bin/fm-turnend-guard.sh`. -`opencode`, `pi`, `pi-signed`, and `grok` expose passive lifecycle callbacks for this purpose, so their tracked primary adapters force one bounded follow-up or resume when the shared predicate blocks. +`opencode`, `pi`, and `pi-signed` expose passive lifecycle callbacks and force one bounded follow-up when the shared predicate blocks. +Grok selects native blocking or its pre-native bounded resume fallback from the exact running Stop payload; [`docs/turnend-guard.md`](../../../docs/turnend-guard.md) owns that contract. Kimi is outside the primary turn-end guard scope, while `docs/turnend-guard.md` owns its separate guarded global hook for crew wake signals. The exact hook files, commands, scoping rules, and fail-open tradeoffs are owned by `docs/turnend-guard.md`. `docs/verification/supervision.md` "Turn-end guard" owns active validation evidence. @@ -343,13 +344,13 @@ This keeps the hook outside the worktree, needs no trust grant, and writes only `fm-teardown` removes the worktree pointer before returning a pooled worktree. Secondmate spawns skip the pointer (idle panes are healthy, no stale-pane detection for them). -**Primary-session guard fact (verified 2026-07-08, Grok 0.2.91).** +**Primary-session guard fact (verified 2026-07-28, Grok 0.2.112 and 0.2.73).** The firstmate PRIMARY's own `.grok/hooks/fm-primary-turnend-guard.json` invokes `bin/fm-turnend-guard-grok.sh`. -Grok Stop hooks are passive for this purpose: exit 2 does not make the model continue. -The adapter therefore runs the shared predicate and, when it returns 2, forces one same-session follow-up with `grok --resume <sessionId> -p <guard-reason>` while setting `GROK_TURNEND_GUARD_ACTIVE=1` so the nested Stop hook does not recurse. -It does not pass `--permission-mode`, so the passive hook cannot escalate the primary session's tool permissions. +Grok 0.2.112 exposes native same-process Stop continuation in its running payload, while the genuine pre-native 0.2.73 payload omits that capability and still needs one guarded `grok --resume`. +The exact adaptive and malformed-input contract is owned by `docs/turnend-guard.md`. +The tracked Claude Stop hooks skip themselves under `GROK_AGENT`, because Grok also loads Claude-compatible project settings and otherwise creates a second blocking path. Project-local Grok hooks require folder trust, verified with launch-time `--trust`; if the primary firstmate checkout is not trusted for Grok hooks, this primary guard fails open and `fm-guard.sh` remains the next-command alarm. -Grok's primary watcher protocol is Claude-shaped background-notify around `bin/fm-watch-arm.sh`; the passive Stop hook is only a backstop for blind turn ends. +Grok's primary watcher protocol remains background-notify around `bin/fm-watch-arm.sh`; native Stop continuation does not provide Pi-like extension ownership. ## kimi (VERIFIED 2026-07-25, kimi 0.29.1) diff --git a/.claude/settings.json b/.claude/settings.json index e77613c98a4..0be379c46b7 100644 --- a/.claude/settings.json +++ b/.claude/settings.json @@ -40,11 +40,11 @@ "hooks": [ { "type": "command", - "command": "\"$CLAUDE_PROJECT_DIR\"/bin/fm-turnend-guard.sh --claude" + "command": "[ -z \"${GROK_AGENT:-}\" ] || exit 0; exec \"$CLAUDE_PROJECT_DIR\"/bin/fm-turnend-guard.sh --claude" }, { "type": "command", - "command": "\"$CLAUDE_PROJECT_DIR\"/bin/fm-claude-stop-autoarm.sh", + "command": "[ -z \"${GROK_AGENT:-}\" ] || exit 0; exec \"$CLAUDE_PROJECT_DIR\"/bin/fm-claude-stop-autoarm.sh", "asyncRewake": true, "timeout": 28800 } diff --git a/AGENTS.md b/AGENTS.md index f838dfb27ca..4991118cb4b 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -88,7 +88,7 @@ state/ volatile runtime signals; gitignored <id>.turn-ended touched by turn-end hooks <id>.grok-turnend-token firstmate-owned grok hook registry token for the task; removed by teardown <id>.kimi-turnend-token firstmate-owned Kimi hook registry token for the task; removed by teardown - <id>.meta written by fm-spawn: window=, worktree=, project=, harness=, model=, effort=, kind=, mode=, yolo=, tasktmp=; kind=secondmate also records home= and projects=; a non-default runtime backend records further backend-specific fields (docs/configuration.md "Runtime backend"; bin/fm-backend.sh, section 8); fm-pr-check, including through fm-pr-merge, records one canonical pr= and the forge's pr_head= when available (GitHub pull requests and GitLab merge requests; docs/gitlab-merge-watch.md); fm-x-link appends x_request=, x_request_ts=, x_followups=, and optional x_platform=/x_reply_max_chars= for an X-mode-originated task (section 14) + <id>.meta written by fm-spawn: window=, endpoint_task_id=, worktree=, project=, harness=, model=, effort=, kind=, mode=, yolo=, tasktmp=; kind=secondmate also records home= and projects=; a non-default runtime backend records further backend-specific fields (docs/configuration.md "Runtime backend"; bin/fm-backend.sh, section 8); fm-pr-check, including through fm-pr-merge, records one canonical pr= and the forge's pr_head= when available (GitHub pull requests and GitLab merge requests; docs/gitlab-merge-watch.md); fm-x-link appends x_request=, x_request_ts=, x_followups=, and optional x_platform=/x_reply_max_chars= for an X-mode-originated task (section 14) <id>.herdr-presentation quarantinable attempt and restart-binding journal for Herdr's optional visual projection; never task or endpoint authority; see docs/herdr-backend.md "Optional presentation spaces" <id>.check.sh authenticated slow poll; the watcher dispatches validated PR data and the byte-identified X shim through trusted repository scripts, runs registered custom checks from hash-validated private snapshots, and rejects every other state check without execution <id>.check-trust private content binding created by fm-check-register.sh for an intentional custom check diff --git a/bin/backends/tmux.sh b/bin/backends/tmux.sh index fe0ed716a42..f8da21bf0de 100644 --- a/bin/backends/tmux.sh +++ b/bin/backends/tmux.sh @@ -117,10 +117,22 @@ fm_backend_tmux_send_literal() { # <target> <text> tmux send-keys -t "$1" -l "$2" } -# fm_backend_tmux_kill: remove the task's window, best-effort. Mirrors -# fm-teardown.sh's `tmux kill-window -t "$T" 2>/dev/null || true`. +# fm_backend_tmux_kill: remove one explicitly named task window, best-effort. +# Empty, omitted, and malformed targets return nonzero before invoking tmux so +# tmux can never interpret an empty target as the caller's current window. fm_backend_tmux_kill() { # <target> - tmux kill-window -t "$1" 2>/dev/null || true + local target=${1:-} session window + case "$target" in + *:*) + session=${target%%:*} + window=${target#*:} + ;; + *) return 1 ;; + esac + case "$session:$window" in + :*|*:|*:*:*) return 1 ;; + esac + tmux kill-window -t "=$session:=$window" 2>/dev/null || true } # fm_backend_tmux_current_command: <target>'s live foreground process name - diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index fe659b54b6c..98273f704ef 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -1435,6 +1435,7 @@ META_WINDOW=$T [ "$BACKEND" = orca ] && META_WINDOW=$W { echo "window=$META_WINDOW" + echo "endpoint_task_id=$ID" echo "worktree=$WT" echo "project=$PROJ_ABS" echo "harness=$HARNESS" diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index c90d759c0db..255c1cdc31d 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -156,13 +156,14 @@ family_for_basename() { ;; fm-afk-pi-herdr-return-e2e.test.sh|\ fm-codex-continuity-live-e2e.test.sh|fm-grok-continuity-live-e2e.test.sh|\ - fm-opencode-primary-live-e2e.test.sh|fm-pi-primary-live-e2e.test.sh|\ + fm-grok-stop-live-e2e.test.sh|fm-opencode-primary-live-e2e.test.sh|fm-pi-primary-live-e2e.test.sh|\ fm-send-secondmate-marker-herdr-e2e.test.sh) printf '%s\n' live-harness-optin ;; fm-backend-herdr.test.sh|fm-backend-tmux-smoke.test.sh|fm-backend.test.sh|\ fm-herdr-session-cleanup.test.sh|fm-send-strict.test.sh|fm-spawn-batch.test.sh|\ - fm-spawn-dispatch-profile.test.sh|fm-spawn-worktree-settle.test.sh) + fm-spawn-dispatch-profile.test.sh|fm-spawn-worktree-settle.test.sh|\ + fm-teardown-endpoint-safety.test.sh) printf '%s\n' backend-dispatch ;; fm-pr-check-security.test.sh|fm-pr-merge.test.sh|fm-review-diff.test.sh|\ diff --git a/bin/fm-turnend-guard.sh b/bin/fm-turnend-guard.sh index 515a859cd2b..2e96fb33e48 100755 --- a/bin/fm-turnend-guard.sh +++ b/bin/fm-turnend-guard.sh @@ -11,8 +11,10 @@ # This script is push-based: verified harness turn-end hooks invoke it every time # the primary is about to end a turn. # Claude and codex can block directly by preserving exit status 2 and stderr. -# OpenCode, pi, and grok adapters use the same predicate and force one bounded -# follow-up because their turn-end events are passive. +# OpenCode and pi adapters use the same predicate and force one bounded +# follow-up because their turn-end events are passive. Grok delegates native +# blocking when its running Stop payload advertises that capability, with one +# bounded resume fallback for payloads from pre-native processes. # See docs/turnend-guard.md for the per-harness mechanics, validation evidence, # and fail-open tradeoffs. # @@ -26,10 +28,10 @@ # primary checkout - the main home or a genuinely marked secondmate home - and # stay a silent, fast no-op inside child task worktrees. # -# Loop-guard, codex (default) mode: never block twice in the same turn. Codex -# Stop payloads carry stop_hook_active=true when the CURRENT stop attempt was -# itself already forced by an earlier block this turn; on that signal we always -# allow the stop, whether or not watcher supervision actually got resumed. +# Loop-guard, codex/Grok (default) mode: never block twice in the same turn. +# Codex uses stop_hook_active and Grok uses stopHookActive; typed camel-case +# takes precedence when both spellings are present. A true value means the +# current stop attempt already follows a block, so this guard always allows it. # Passive harness adapters provide their own one-follow-up guard before calling # this script. # That bounds those harnesses to at most one forced continuation per turn - @@ -94,7 +96,15 @@ PAYLOAD=$(cat 2>/dev/null || true) # loop-guard field, so we must never block - fail open, not noisy. command -v jq >/dev/null 2>&1 || exit 0 -STOP_HOOK_ACTIVE=$(printf '%s' "$PAYLOAD" | jq -r '.stop_hook_active // false' 2>/dev/null) || exit 0 +STOP_HOOK_ACTIVE=$(printf '%s' "$PAYLOAD" | jq -r ' + if type != "object" then error("payload") + elif has("stopHookActive") then + if ((.stopHookActive | type) == "boolean") then .stopHookActive else error("stopHookActive") end + elif has("stop_hook_active") then + if ((.stop_hook_active | type) == "boolean") then .stop_hook_active else error("stop_hook_active") end + else false + end +' 2>/dev/null) || exit 0 if [ "$CLAUDE_MODE" -eq 0 ] && [ "$STOP_HOOK_ACTIVE" = "true" ]; then exit 0 fi diff --git a/docs/architecture.md b/docs/architecture.md index da1519a0d44..d1bcb83c565 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -89,7 +89,7 @@ On an unmarked return, `bin/fm-afk-return.sh` owns ordered shutdown, durable cat The runtime backend is the session-provider layer below firstmate's scripts. It owns task endpoint creation, bounded capture, text/key sends, current-path reads for spawn-time worktree discovery when the backend does not create the worktree itself, live-window fallback lookup, agent-process liveness probes where verified, and endpoint teardown. -`bin/fm-backend.sh` centralizes backend selection, `state/<id>.meta` helpers, selector resolution, and operation dispatch; `bin/backends/tmux.sh` is the verified reference adapter ([`docs/tmux-backend.md`](tmux-backend.md)), and `bin/backends/herdr.sh` (P2), `bin/backends/zellij.sh` (P3), `bin/backends/orca.sh` (P4), and `bin/backends/cmux.sh` (P5) are experimental task-spawn adapters. +`bin/fm-backend.sh` centralizes backend selection, `state/<id>.meta` helpers, metadata-only cleanup identity validation, selector resolution, and operation dispatch; `bin/backends/tmux.sh` is the verified reference adapter ([`docs/tmux-backend.md`](tmux-backend.md)), and `bin/backends/herdr.sh` (P2), `bin/backends/zellij.sh` (P3), `bin/backends/orca.sh` (P4), and `bin/backends/cmux.sh` (P5) are experimental task-spawn adapters. New spawns select a backend from `--backend`, then `FM_BACKEND`, then local `config/backend`, then runtime auto-detection from `$TMUX`, `HERDR_ENV=1`, or cmux runtime signals, then default `tmux`. Runtime auto-detection is innermost-first: `$TMUX` wins over `HERDR_ENV=1`, which wins over cmux's primary `CMUX_WORKSPACE_ID` marker and documented fallback signals; auto-detected herdr or cmux prints a one-time opt-out notice, auto-detected tmux stays silent, and zellij and orca are never auto-detected (only explicit selection). Unknown backend names fail loudly. diff --git a/docs/configuration.md b/docs/configuration.md index 7f23866c334..d9a06bf4cad 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -67,6 +67,7 @@ A zellij spawn additionally version-gates against the installed `zellij` binary' A cmux spawn additionally version-gates against the installed `cmux` binary's version, requires `jq`, and requires the control socket to be reachable and accessible (see [`docs/cmux-backend.md`](cmux-backend.md) "Setup" for the one-time socket-access configuration this needs; Automation mode is the recommended socket control mode, with Password mode supported via `config/cmux-socket-password`), refusing loudly and non-retryably on a `cmuxOnly`/unauthenticated socket. A backend spawn refusal from a missing dependency, version gate, or unauthenticated socket is terminal for that selected backend; firstmate surfaces it as a blocker instead of silently retrying another backend. Task meta records `backend=` only for a non-default backend; an absent `backend=` means `tmux`, preserving existing default-path meta files. +Every new task records `endpoint_task_id=` as the cleanup binding between the metadata filename and its opaque runtime endpoint. A herdr task additionally records `herdr_session=`, `herdr_workspace_id=`, `herdr_tab_id=`, and `herdr_pane_id=`. A zellij task additionally records `zellij_session=`, `zellij_tab_id=`, and `zellij_pane_id=`. An Orca task additionally records `orca_worktree_id=` and `terminal=`, with `window=fm-<id>` kept as the shared firstmate alias. @@ -77,7 +78,9 @@ Otherwise an exact task id matching `state/<id>.meta` wins before the legacy `fm A metadata-routed selector returns the recorded backend target (`terminal=` for Orca, otherwise `window=`), and matching explicit targets can still recover the recorded backend when metadata contains the same endpoint. Only metadata-routed task selectors carry secondmate-marker and Codex-harness context; explicit endpoint escape hatches do not. These five sentences are the single owner of the task-selector vocabulary; backend guides and other documents point here instead of restating the resolution order. -`fm-teardown.sh <id>` takes a task id directly and uses the same recorded backend target fields after loading `state/<id>.meta`. +`fm-teardown.sh <id>` takes a task id directly and validates the complete metadata-only endpoint identity before any runtime dispatch or cleanup mutation. +Missing, empty, duplicate, malformed, backend-inconsistent, or task-mismatched endpoint records are preserved and refused. +Legacy tmux metadata remains cleanup-compatible when its exact window name is `fm-<id>`; opaque non-tmux endpoints require their recorded `endpoint_task_id=` binding. By default, Herdr workspaces are derived from `FM_HOME`: the primary home uses `firstmate`, and a secondmate home marked by `.fm-secondmate-home` uses `2ndmate-<secondmate-id>`. The default-container spawn, list-live, and recovery paths read that label from the active home, so a secondmate's own crewmates stay inside that secondmate home's herdr space. The optional local `config/herdr-presentation-spaces` presence flag instead enables Herdr's default-off disposable single-task visual projection; [Optional presentation spaces](herdr-backend.md#optional-presentation-spaces) owns its behavior, safety limits, recovery contract, and narrow locked session-start cleanup of exact restored idle-shell children. diff --git a/docs/turnend-guard.md b/docs/turnend-guard.md index 30690bb887e..8ee750de397 100644 --- a/docs/turnend-guard.md +++ b/docs/turnend-guard.md @@ -42,8 +42,8 @@ If `jq` is missing or hook stdin is empty, the guard exits 0 because it cannot s - Codex registers a `Stop` hook in `.codex/hooks.json`, anchors the executable to the hook process working directory, verifies a Firstmate-shaped hook-bearing root, and passes the original payload to the shared guard. - OpenCode listens for `session.idle` in `.opencode/plugins/fm-primary-turnend-guard.js`, lets the watcher coordinator act first, and calls `client.session.promptAsync` once when the guard returns 2. - Pi listens for `agent_settled` in `.pi/extensions/fm-primary-turnend-guard.ts`, runs once per logical agent run, and calls `pi.sendUserMessage(..., { deliverAs: "followUp" })` once when the guard returns 2. -- Grok registers a `Stop` hook in `.grok/hooks/fm-primary-turnend-guard.json` and uses `bin/fm-turnend-guard-grok.sh` to resume the reported session once when the shared guard returns 2. - The adapter intentionally omits `--permission-mode`, so a passive hook cannot grant stronger permissions than the resumed session default. +- Grok registers a `Stop` hook in `.grok/hooks/fm-primary-turnend-guard.json` and delegates capability selection to `bin/fm-turnend-guard-grok.sh`. + The tracked Claude Stop entries are inert when `GROK_AGENT` is present, so Grok's Claude-compatible settings loading cannot create a second continuation path. Claude and Codex can block a Stop directly with exit status 2 and stderr. Both payloads carry `stop_hook_active`. @@ -55,15 +55,22 @@ The Claude mode waits up to `FM_CLAUDE_AUTOARM_SYNC_WAIT_MS` (default 800 millis When none of those proofs appears, it re-blocks up to `FM_CLAUDE_TURNEND_BLOCK_BUDGET` times (default 3, below Claude's 8-block override), then allows degraded with a visible `systemMessage`. Any allow resets the budget. -OpenCode, Pi, pi-signed, and Grok expose passive callbacks for this purpose. +OpenCode, Pi, and pi-signed expose passive callbacks for this purpose. Their adapters fail open at the hook boundary to protect the user session but schedule one bounded follow-up when the predicate blocks. The generated prompts use the canonical `turn-end-guard` kind after the U+2063 `FIRSTMATE_OP: ` prefix, so Ahoy does not treat them as captain messages. -Each adapter owns a loop latch. +Each passive adapter owns a loop latch. Pi keeps the latch across internal tool turns and clears it only when the generated follow-up settles or delivery fails. -Grok's project hook requires the checkout to be trusted with `/hooks-trust` or launch-time `--trust`. OpenCode's forced follow-up is supported for persistent TUI sessions and remains fail-open in headless `opencode run`. -If a passive adapter cannot invoke its SDK, find `grok`, or recover a Grok session id, the next pull-based `fm-guard.sh` call reports the problem. +Grok makes exactly one typed capability decision from each running Stop payload. +A boolean `stopHookActive` selects native blocking, including both false on the initial stop and true on the bounded continuation. +The camel-case field has precedence when both spellings appear; when it is absent, a boolean `stop_hook_active` selects the same native path for compatibility. +The native path returns the shared guard's status and stderr to the same Grok process and never starts `grok --resume`. +When both capability spellings are absent, the adapter preserves one pre-native `grok --resume` fallback guarded by `GROK_TURNEND_GUARD_ACTIVE` and intentionally omits `--permission-mode`. +Malformed JSON, a selected field with a non-boolean type, missing `jq`, missing hook prerequisites, or an already-active legacy guard allows the stop without starting either continuation path. +Grok's project hook requires the checkout to be trusted with `/hooks-trust` or launch-time `--trust`; genuine pre-native builds can run the same tracked hook from an isolated global hook directory. + +If a passive adapter cannot invoke its SDK, or the Grok legacy fallback cannot find `grok` or a session id, the next pull-based `fm-guard.sh` call reports the problem. That warning uses `bin/fm-supervision-instructions.sh --repair-line`, so it always points to the active harness protocol rather than embedding another repair command. ## Compatibility limits @@ -83,7 +90,7 @@ That warning uses `bin/fm-supervision-instructions.sh --repair-line`, so it alwa ## Regression coverage -`tests/fm-turnend-guard.test.sh` covers the predicate, main and secondmate primary scope, child-worktree exclusion, `FM_HOME` and `FM_STATE_OVERRIDE` precedence, the cooperative `--claude` claim wait, epoch allow, re-block budget, Pi logical-run latching, missing-`jq` behavior, all five primary registrations, and Grok resume permission and recursion safety. +`tests/fm-turnend-guard.test.sh` covers the predicate, main and secondmate primary scope, child-worktree exclusion, `FM_HOME` and `FM_STATE_OVERRIDE` precedence, the cooperative `--claude` claim wait, epoch allow, re-block budget, Pi logical-run latching, missing-`jq` behavior, all five primary registrations, Grok native and legacy selection, typed field precedence, malformed input, and exactly-one-path safety. `tests/fm-kimi-harness.test.sh` covers the separate Kimi crew hook's format preservation, idempotence, refusal cases, token guard, spawn registration, and teardown cleanup. `tests/fm-supervision-instructions.test.sh` covers recovery-line ownership and pi-signed's identity-preserving reuse of Pi's protocol. `FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh` is the opt-in isolated Pi path. diff --git a/docs/verification/runtime-backends.md b/docs/verification/runtime-backends.md index a711d84ee5b..65152100f4e 100644 --- a/docs/verification/runtime-backends.md +++ b/docs/verification/runtime-backends.md @@ -89,6 +89,34 @@ tests/fm-tmux-submit-busy.test.sh Expected structural matrix: real text on any content row is pending; all-empty complete boxes are empty; unreadable, incomplete, or unsafe boxes are unknown; and non-bordered panes retain cursor-row compatibility. Expected submit matrix: proven pending plus busy is accepted as queued; proven pending plus idle remains pending; ambiguous pending is never converted by the busy exception; and only a proven empty composer succeeds directly. +### Cleanup endpoint identity + +The cleanup identity boundary was validated on 2026-07-28 with tmux 3.6a and metadata fixtures for every supported backend. + +```sh +tests/fm-teardown-endpoint-safety.test.sh +tests/fm-teardown.test.sh +tests/fm-backend-herdr.test.sh +tests/fm-backend-zellij.test.sh +tests/fm-backend-orca.test.sh +tests/fm-backend-cmux.test.sh +``` + +Bounded output from the incident regression: + +```text +ok - fm-teardown: missing, empty, malformed, ambiguous, and task-mismatched endpoints refuse before every mutation or runtime call +ok - cleanup identity: valid tmux, Herdr, Zellij, Orca, and cmux records validate while every empty backend target refuses +ok - tmux backend: direct empty target returns nonzero without invoking tmux +ok - process cleanup: creation-time PID identity removes only the exact child and preserves the control child +ok - fm-teardown: dedicated-socket invalid cleanup preserves target/control and valid cleanup removes only the exact target +``` + +The dedicated tmux cell removed ambient tmux variables, required a socket-bound wrapper, kept one target and one independent control window, and proved the wrapper was not called for invalid metadata or a direct empty target. +Valid cleanup removed only the exact task-bound target and left the control window live. +The metadata-only validation covers tmux, Herdr, Zellij, Orca, and cmux before backend dispatch. +Claude, Codex, OpenCode, Pi, pi-signed, Grok, and Kimi share that backend cleanup boundary; their harness-specific hook files and token cleanup run only after it, so no harness needs a separate endpoint parser. + ## Herdr The compatibility floor is protocol 14. diff --git a/docs/verification/supervision.md b/docs/verification/supervision.md index 6945b3491dc..326d21d73ed 100644 --- a/docs/verification/supervision.md +++ b/docs/verification/supervision.md @@ -71,7 +71,26 @@ The direct and passive mechanisms were validated across all five harnesses on 20 | Codex | 0.142.1 | Blocking `Stop` hook | Hook process root stayed anchored to the trusted checkout and one continuation ran. | | OpenCode | 1.17.6 | Passive `session.idle` callback | Throwing could not block, while `promptAsync` scheduled one TUI follow-up; headless remained fail-open. | | Pi | 0.80.5 | Passive `agent_settled` callback | Exactly one guard follow-up ran for an unhealthy cycle, with no recursion across tool turns. | -| Grok | 0.2.93 | Passive `Stop` plus bounded resume | Project hook ran under trust, resumed once without inherited bypass permissions, and the environment latch prevented recursion. | +| Grok | 0.2.112 native and 0.2.73 pre-native | Running-payload adaptive `Stop` | Native false-to-true continuation stayed in one process with two model turns and zero resume launches; the field-absent pre-native process launched exactly one guarded resume. | + +The Grok adaptive matrix ran on 2026-07-28 with separate scratch repositories and homes, dedicated tmux sockets, one target plus one control window, ambient tmux variables removed, and a socket-bound wrapper first in `PATH`. + +```sh +FM_GROK_STOP_LIVE_E2E=1 \ + FM_GROK_NATIVE_BIN="$native_grok_0_2_112" \ + FM_GROK_LEGACY_BIN="$official_pre_native_grok_0_2_73" \ + tests/fm-grok-stop-live-e2e.test.sh +``` + +Observed bounded output: + +```text +ok - grok 0.2.112 (9bbd559437aa) [stable] native Stop kept one session across false->true, two model turns, and zero resume processes +ok - grok 0.2.73 (9ff14c43bbe5) [stable] legacy Stop omitted capability, resumed exactly once, and stopped normally +ok - Grok adaptive Stop real-process matrix passed with exact target cleanup and control-window survival +``` + +The same run proved the Claude-compatible Stop entries stay inert under `GROK_AGENT`, the legacy resume carries `GROK_TURNEND_GUARD_ACTIVE=1`, and every replacement root is removed after exact target cleanup while its control window survives. The secondmate-home scope and manual-repair wake path were measured with Claude Code 2.1.207 on 2026-07-12, when a native background completion re-invoked the idle model with no human input. The current Stop-owned main/secondmate inclusion and child-worktree exclusion are covered deterministically by `tests/fm-claude-stop-autoarm.test.sh`. @@ -96,6 +115,7 @@ Current entry points: tests/fm-turnend-guard.test.sh tests/fm-supervision-instructions.test.sh FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh +FM_GROK_STOP_LIVE_E2E=1 FM_GROK_NATIVE_BIN="$native_grok" FM_GROK_LEGACY_BIN="$pre_native_grok" tests/fm-grok-stop-live-e2e.test.sh ``` ## Watcher continuity diff --git a/tests/fm-backend.test.sh b/tests/fm-backend.test.sh index 7ac873f0878..323cd4f5e51 100755 --- a/tests/fm-backend.test.sh +++ b/tests/fm-backend.test.sh @@ -12,10 +12,7 @@ # binaries and fixtures as the REFACTORED versions in this checkout, then # diffs the two command logs byte-for-byte - the report's P1 checklist # item "run current main scripts and refactored scripts against the same -# fake tools and compare command logs". The teardown old-vs-new case also -# overlays a content-historical permissive tmux kill fixture: after the -# exact-selector change lands on the default branch, merge-base with main -# collapses to HEAD and can no longer supply that baseline. +# fake tools and compare command logs". # 3. Asserts the `--backend`/`FM_BACKEND` selection refuses unknown backends # and the blocked `codex-app` backend loudly. # @@ -83,9 +80,6 @@ SH } # The commit this branch started from - the P1 "current main" baseline. -# Suitable for byte-identical old-vs-new checks while a branch still diverges -# from main. After a squash lands, merge-base(HEAD, main) collapses to HEAD, so -# callers that need a true pre-change fixture must not rely on this alone. resolve_base_ref() { local ref base for ref in main refs/heads/main origin/main refs/remotes/origin/main origin/HEAD refs/remotes/origin/HEAD; do @@ -101,30 +95,6 @@ resolve_base_ref() { BASE_REF=$(resolve_base_ref) \ || fail "fm-backend baseline requires local main or origin/main; fetch the default branch before running this test" -# Newest first-parent revision whose bin/backends/tmux.sh still uses the -# pre-exact permissive kill-window target. Content-addressed from history so the -# fixture stays historical on default-branch CI and on branches cut after the -# exact-selector change, where merge-base with main is self-referential. -resolve_permissive_tmux_kill_ref() { - local commit body - while IFS= read -r commit; do - [ -n "$commit" ] || continue - body=$(git -C "$ROOT" show "$commit:bin/backends/tmux.sh" 2>/dev/null) || continue - # shellcheck disable=SC2016 - case "$body" in - *'tmux kill-window -t "=$session:=$window"'*) continue ;; - esac - # shellcheck disable=SC2016 - case "$body" in - *'tmux kill-window -t "$1"'*|*'tmux kill-window -t "$target"'*) - printf '%s\n' "$commit" - return 0 - ;; - esac - done < <(git -C "$ROOT" log --first-parent --format='%H' HEAD -- bin/backends/tmux.sh) - return 1 -} - # --- shared: a pre-refactor bin/ shim -------------------------------------- # # build_old_bin echoes a directory whose bin/ subdir holds the PRE-REFACTOR @@ -960,20 +930,9 @@ run_teardown_case() { } test_teardown_conformance_old_vs_new() { - local old_bin fb proj wt id old_tmux_ref saved_base_ref + local old_bin fb proj wt id local state_old state_new config_old config_new data log_old log_new out_old out_new rc_old rc_new - # Force the post-squash topology inside this case: merge-base with main may - # equal HEAD on default-branch CI, and that must not make the legacy kill - # fixture self-referential. build_old_bin still uses BASE_REF for entrypoints; - # only the tmux kill adapter is pinned to the content-historical permissive ref. - saved_base_ref=$BASE_REF - BASE_REF=$(git -C "$ROOT" rev-parse HEAD) - old_tmux_ref=$(resolve_permissive_tmux_kill_ref) \ - || { BASE_REF=$saved_base_ref; fail "unable to locate a historical bin/backends/tmux.sh with permissive kill-window selectors"; } old_bin=$(build_old_bin teardown-old) - git -C "$ROOT" show "$old_tmux_ref:bin/backends/tmux.sh" > "$old_bin/bin/backends/tmux.sh" \ - || { BASE_REF=$saved_base_ref; fail "could not materialize historical tmux adapter from $old_tmux_ref"; } - BASE_REF=$saved_base_ref proj="$TMP_ROOT/teardown-project"; wt="$TMP_ROOT/teardown-wt" id="teardownconform1" fm_git_worktree "$proj" "$wt" "fm/$id" @@ -1005,15 +964,8 @@ test_teardown_conformance_old_vs_new() { expect_code 0 "$rc_new" "new fm-teardown.sh (scout, report present) should succeed"$'\n'"$out_new" assert_contains "$(cat "$log_new")" "treehouse"$'\x1f''return'$'\x1f''--force'$'\x1f'"$wt" \ "teardown did not call treehouse return --force <worktree>" - # The legacy fixture's adapter comes from BASE_REF, so its selector form is - # whatever the merge-base carried: permissive while the exact-selector change - # was still on a branch, exact for every branch cut after it landed on main. - # Pinning the old form here would make this case pass once and then fail - # forever, so the '=' exactness markers are normalized away and the legacy run - # is only required to have reached tmux window cleanup for this task. The - # exact-selector contract belongs to the current script, asserted below. - assert_contains "$(tr -d '=' < "$log_old")" "tmux"$'\x1f''kill-window'$'\x1f''-t'$'\x1f'"firstmate:fm-$id" \ - "legacy teardown fixture did not exercise tmux window cleanup for the task" + assert_contains "$(cat "$log_old")" "tmux"$'\x1f''kill-window'$'\x1f''-t'$'\x1f'"firstmate:fm-$id" \ + "legacy teardown fixture did not exercise tmux's permissive target selector" assert_contains "$(cat "$log_new")" "tmux"$'\x1f''kill-window'$'\x1f''-t'$'\x1f'"=firstmate:=fm-$id" \ "teardown did not call tmux kill-window with exact session and window selectors" diff --git a/tests/fm-secondmate-liveness.test.sh b/tests/fm-secondmate-liveness.test.sh index ff5c07a948c..ed356638962 100755 --- a/tests/fm-secondmate-liveness.test.sh +++ b/tests/fm-secondmate-liveness.test.sh @@ -349,7 +349,7 @@ test_sweep_respawns_confirmed_dead_secondmate() { assert_not_contains "$out" "SECONDMATE_LIVENESS: secondmate sm1: respawned" \ "a successfully respawned secondmate should be handled silently" - assert_contains "$(cat "$log")" "kill-window -t firstmate:fm-sm1" \ + assert_contains "$(cat "$log")" "kill-window -t =firstmate:=fm-sm1" \ "the stale endpoint must be killed before respawn (tmux refuses a same-named window over a live one)" assert_contains "$(cat "$log")" "new-window" \ "a confirmed-dead secondmate should actually be relaunched" diff --git a/tests/fm-turnend-guard.test.sh b/tests/fm-turnend-guard.test.sh index 813709d73a9..2b82165178b 100755 --- a/tests/fm-turnend-guard.test.sh +++ b/tests/fm-turnend-guard.test.sh @@ -604,17 +604,119 @@ EOF expect_code 0 "$status" "grok adapter must allow its own forced resume turn to end" [ -z "$out" ] || fail "grok adapter printed output while loop-guarded: $out" [ ! -e "$log" ] || fail "grok adapter spawned another resume while loop-guarded: $(cat "$log")" - pass "fm-turnend-guard-grok: loop guard prevents a nested resume loop" + pass "fm-turnend-guard-grok: legacy environment loop guard prevents a nested resume loop" +} + +test_grok_adapter_native_false_blocks_without_resume() { + local dir fakebin log out status + dir=$(make_primary_dir "$TMP_ROOT/grok-native-false") + : > "$dir/state/task1.meta" + fakebin=$(fm_fakebin "$TMP_ROOT/grok-native-false-bin") + log="$TMP_ROOT/grok-native-false.log" + printf '#!/usr/bin/env bash\nprintf called >> %q\n' "$log" > "$fakebin/grok" + chmod +x "$fakebin/grok" + out=$(printf '%s' '{"sessionId":"native","stopHookActive":false}' | PATH="$fakebin:$PATH" GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 2 "$status" "native stopHookActive=false must return the shared blocking status" + assert_contains "$out" 'TURN WOULD END BLIND' "native block must pass shared guard feedback to Grok" + [ ! -e "$log" ] || fail "native path started grok --resume" + pass "fm-turnend-guard-grok: native false delegates blocking feedback with zero resume processes" +} + +test_grok_adapter_native_true_allows_without_resume() { + local dir fakebin log out status + dir=$(make_primary_dir "$TMP_ROOT/grok-native-true") + : > "$dir/state/task1.meta" + fakebin=$(fm_fakebin "$TMP_ROOT/grok-native-true-bin") + log="$TMP_ROOT/grok-native-true.log" + printf '#!/usr/bin/env bash\nprintf called >> %q\n' "$log" > "$fakebin/grok" + chmod +x "$fakebin/grok" + out=$(printf '%s' '{"sessionId":"native","stopHookActive":true}' | PATH="$fakebin:$PATH" GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 0 "$status" "native stopHookActive=true must allow the bounded continuation to stop" + [ -z "$out" ] || fail "native true produced output: $out" + [ ! -e "$log" ] || fail "native true started grok --resume" + pass "fm-turnend-guard-grok: native true remains bounded and starts no resume process" +} + +test_grok_adapter_snake_case_native_and_camel_precedence() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/grok-native-spellings") + : > "$dir/state/task1.meta" + out=$(printf '%s' '{"sessionId":"native","stop_hook_active":false}' | GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 2 "$status" "typed snake_case false must select native blocking" + assert_contains "$out" 'TURN WOULD END BLIND' "snake_case native block lost feedback" + out=$(printf '%s' '{"sessionId":"native","stopHookActive":true,"stop_hook_active":false}' | GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 0 "$status" "camelCase true must win over snake_case false" + out=$(printf '%s' '{"sessionId":"native","stopHookActive":false,"stop_hook_active":true}' | GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 2 "$status" "camelCase false must win over snake_case true" + pass "fm-turnend-guard-grok: both spellings are typed and camelCase has deterministic precedence" +} + +test_grok_adapter_invalid_inputs_start_neither_path() { + local dir fakebin log payload out status + dir=$(make_primary_dir "$TMP_ROOT/grok-invalid-inputs") + : > "$dir/state/task1.meta" + fakebin=$(fm_fakebin "$TMP_ROOT/grok-invalid-bin") + log="$TMP_ROOT/grok-invalid.log" + printf '#!/usr/bin/env bash\nprintf called >> %q\n' "$log" > "$fakebin/grok" + chmod +x "$fakebin/grok" + for payload in \ + ' ' \ + '{' \ + '{"sessionId":"x","stopHookActive":"false"}' \ + '{"sessionId":"x","stop_hook_active":1}' \ + '{"sessionId":"x"}{"sessionId":"y"}' \ + '{"sessionId":"x","stopHookActive":false}{"sessionId":"y","stopHookActive":false}' \ + '{"sessionId":"x","stopHookActive":"bad","stopHookActive":false}' \ + '{"sessionId":"x","stop_hook_active":false,"stop_hook_active":false}' \ + '{"sessionId":"x","sessionId":"y"}' + do + out=$(printf '%s' "$payload" | PATH="$fakebin:$PATH" GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 0 "$status" "invalid Grok payload must conservatively allow without choosing a path" + [ -z "$out" ] || fail "invalid Grok payload produced output: $out" + done + [ ! -e "$log" ] || fail "invalid Grok payload started a resume process" + out=$(printf '%s' '{"sessionId":"x","stopHookActive":false}' | PATH="$fakebin:$PATH" GROK_WORKSPACE_ROOT="$TMP_ROOT/missing-grok-root" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 0 "$status" "missing shared-guard prerequisite must conservatively allow" + [ -z "$out" ] || fail "missing prerequisite produced output: $out" + [ ! -e "$log" ] || fail "missing prerequisite started a resume process" + pass "fm-turnend-guard-grok: malformed, invalidly typed, and missing-prerequisite payloads start neither path" +} + +test_grok_adapter_missing_jq_and_no_supervision_allow() { + local dir fakebin log out status tool tool_path + dir=$(make_primary_dir "$TMP_ROOT/grok-nojq") + : > "$dir/state/task1.meta" + fakebin=$(fm_fakebin "$TMP_ROOT/grok-nojq-bin") + log="$TMP_ROOT/grok-nojq.log" + for tool in bash cat printf; do + tool_path=$(command -v "$tool") || fail "test host must provide $tool" + ln -s "$tool_path" "$fakebin/$tool" + done + printf '#!/usr/bin/env bash\nprintf called >> %q\n' "$log" > "$fakebin/grok" + chmod +x "$fakebin/grok" + out=$(printf '%s' '{"sessionId":"x","stopHookActive":false}' | PATH="$fakebin" GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 0 "$status" "missing jq must conservatively allow" + [ -z "$out" ] || fail "missing jq produced output: $out" + [ ! -e "$log" ] || fail "missing jq started a resume process" + + dir=$(make_primary_dir "$TMP_ROOT/grok-native-no-work") + out=$(printf '%s' '{"sessionId":"x","stopHookActive":false}' | GROK_WORKSPACE_ROOT="$dir" bash "$dir/bin/fm-turnend-guard-grok.sh" 2>&1); status=$? + expect_code 0 "$status" "healthy no-supervision-needed native stop must allow" + [ -z "$out" ] || fail "no-supervision-needed native stop produced output: $out" + pass "fm-turnend-guard-grok: missing jq and no-supervision-needed stops stay silent and bounded" } test_settings_hook_uses_claude_project_dir() { - local settings command + local settings command autoarm settings="$ROOT/.claude/settings.json" [ -f "$settings" ] || fail "tracked .claude/settings.json is missing" command=$(jq -r '.hooks.Stop[0].hooks[0].command // empty' "$settings") + autoarm=$(jq -r '.hooks.Stop[0].hooks[1].command // empty' "$settings") [ -n "$command" ] || fail "Stop hook command is missing from .claude/settings.json" assert_contains "$command" 'CLAUDE_PROJECT_DIR' "Stop hook must resolve via CLAUDE_PROJECT_DIR, not a cwd-relative path" assert_contains "$command" 'fm-turnend-guard.sh --claude' "Stop hook must invoke fm-turnend-guard.sh in cooperative --claude mode" + assert_contains "$command" 'GROK_AGENT' "Claude blocking Stop hook must stay inert when Grok loads Claude-compatible settings" + assert_contains "$autoarm" 'GROK_AGENT' "Claude auto-arm Stop hook must stay inert when Grok loads Claude-compatible settings" case "$command" in bin/fm-turnend-guard.sh|./bin/fm-turnend-guard.sh) fail "Stop hook must not use a bare relative path (cwd-dependent): $command" @@ -1117,6 +1219,11 @@ test_hook_silent_without_stdin test_hook_runs_fast test_grok_adapter_forces_one_resume_when_unhealthy test_grok_adapter_loop_guard_skips_resume +test_grok_adapter_native_false_blocks_without_resume +test_grok_adapter_native_true_allows_without_resume +test_grok_adapter_snake_case_native_and_camel_precedence +test_grok_adapter_invalid_inputs_start_neither_path +test_grok_adapter_missing_jq_and_no_supervision_allow test_settings_hook_uses_claude_project_dir test_codex_hook_invokes_shared_guard test_codex_hook_uses_process_pwd_when_payload_cwd_is_outside_root diff --git a/tests/lib.sh b/tests/lib.sh index d33062915ff..ee3b1d1476c 100644 --- a/tests/lib.sh +++ b/tests/lib.sh @@ -152,13 +152,16 @@ fm_write_meta() { } # fm_write_secondmate_meta <file> <home> [window] [projects] [harness]: write the -# standard kind=secondmate meta block used across the secondmate suites. window -# is explicit and defaults to firstmate:fm-domain, projects defaults to alpha, -# and harness defaults to echo to match the common case. +# standard kind=secondmate meta block used across the secondmate suites. Window +# defaults to firstmate:fm-<id>, projects defaults to alpha, and harness defaults +# to echo to match the common case. fm_write_secondmate_meta() { - local file=$1 home=$2 window=${3:-firstmate:fm-domain} projects=${4:-alpha} harness=${5:-echo} + local file=$1 home=$2 id window projects=${4:-alpha} harness=${5:-echo} + id=$(basename "$file" .meta) + window=${3:-firstmate:fm-$id} fm_write_meta "$file" \ "window=$window" \ + "endpoint_task_id=$id" \ "worktree=$home" \ "project=$home" \ "harness=$harness" \ From 37e169285e191d9ee0a92adeb1a3f08ad72acdd8 Mon Sep 17 00:00:00 2001 From: Christopher McKay <101884182+karotkriss@users.noreply.github.com> Date: Tue, 28 Jul 2026 14:34:33 -0400 Subject: [PATCH 22/52] fix: restore stock macOS Bash 3.2 brief scaffolding (#1093) * fix(brief): make DOD scaffolding parse-safe on stock macOS Bash 3.2 fm-brief.sh built each Definition-of-done block and the not-enabled Herdr declaration with `VAR=$(cat <<EOF ... EOF)`. On Bash 3.2 (macOS /bin/bash) the lexer scans for the command substitution's closing `)` textually and tracks quote state through the heredoc body, so a single apostrophe, unbalanced quote, or unbalanced paren in that prose breaks parsing of the whole script. Every ship-brief scaffold (no-mistakes, direct-PR, local-only) failed with `unexpected EOF while looking for matching )`. Bash 4+ parses it fine, so the breakage stayed invisible everywhere except stock macOS. Replace all four command-substitution heredocs with `IFS= read -r -d '' VAR <<EOF || true`. That removes the `$(...)` wrapper and the entire defect class regardless of future prose, and preserves the variable expansion the direct-PR and local-only bodies need. `read` keeps the heredoc's trailing newline that `$(...)` used to strip, so trim one newline to keep every generated brief byte-identical to prior output. Guard the structure, not one historical phrase: a new test rejects any heredoc nested in a command substitution anywhere in fm-brief.sh, where the old assertion pinned a single apostrophe phrase and so missed the reintroduction. Extend the stock-macOS Bash CI job from parsing one script to the whole maintained shell surface (bin/*.sh, bin/backends/*.sh, tests/*.sh), matching bin/fm-lint.sh's canonical file set so parse scope and lint scope cannot drift apart. * no-mistakes(review): Captain: harden Bash structure and inventory guards * no-mistakes(document): Align stock macOS Bash contributor checks * no-mistakes(lint): Suppress deliberate SC2016 literal fixture warnings --- bin/fm-brief.sh | 143 +++++------------------------------------- tests/fm-lint.test.sh | 84 +++++++++++++++++++++++-- 2 files changed, 94 insertions(+), 133 deletions(-) diff --git a/bin/fm-brief.sh b/bin/fm-brief.sh index 2ac2dbdbcc6..8125aa2e9e4 100755 --- a/bin/fm-brief.sh +++ b/bin/fm-brief.sh @@ -6,8 +6,8 @@ # description, acceptance criteria, and context, and may adjust other sections # when the task genuinely deviates (e.g. working an existing external PR instead # of shipping a new one). -# Usage: fm-brief.sh <task-id> <repo-name> [--scout] [--herdr-lab] [--force-regenerate] -# fm-brief.sh <task-id> --secondmate {<project>...|--no-projects} [--force-regenerate] +# Usage: fm-brief.sh <task-id> <repo-name> [--scout] [--herdr-lab] +# fm-brief.sh <task-id> --secondmate {<project>...|--no-projects} # --scout writes the scout contract instead: the deliverable is a report at # data/<task-id>/report.md (no branch, no push, no PR) and the worktree is scratch. # --secondmate writes a persistent secondmate charter. The project list @@ -26,12 +26,6 @@ # The flag must be explicit because {TASK} is filled after scaffolding and the # caller-supplied repo string cannot reliably identify this repo. Briefs made # without it carry a loud declaration so an omitted contract cannot be silent. -# Every generated brief carries a versioned scaffold safety marker. -# When an existing brief is present, the refusal reports whether that marker -# is current but never treats the marker as proof that the task text is fresh. -# --force-regenerate renders a fresh scaffold before archiving an existing -# brief beside it and installing the replacement; it never silently clobbers -# the previous content. # For ship tasks, the definition of done is shaped by the project's delivery mode # (data/projects.md via fm-project-mode.sh; see the project-management skill # and AGENTS.md task lifecycle): @@ -50,7 +44,7 @@ # it carries the AGENTS.md authoring bar (widely useful knowledge only, pointers # over copied detail) and has the crewmate add the fm-ensure-agents-md.sh # self-governance section when a touched project AGENTS.md lacks it. -# Refuses to overwrite or silently reuse an existing brief. +# Refuses to overwrite an existing brief. set -eu SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -72,35 +66,13 @@ esac # shellcheck source=bin/fm-classify-lib.sh . "$SCRIPT_DIR/fm-classify-lib.sh" PAUSED_VERB=${FM_CLASSIFY_PAUSED_VERB:-$FM_CLASSIFY_PAUSED_VERB_DEFAULT} - -resolve_directory_input() { - local name=$1 path=$2 resolved - case "$path" in - /*) printf '%s\n' "$path"; return 0 ;; - esac - resolved=$(CDPATH='' cd -- "$path" 2>/dev/null && pwd -P) || { - echo "error: $name directory cannot be resolved: $path" >&2 - return 1 - } - printf '%s\n' "$resolved" -} - FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" -FM_HOME=$(resolve_directory_input FM_HOME "${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}") || exit 1 -if [ -n "${FM_DATA_OVERRIDE:-}" ]; then - DATA=$(resolve_directory_input FM_DATA_OVERRIDE "$FM_DATA_OVERRIDE") || exit 1 -else - DATA="$FM_HOME/data" -fi -if [ -n "${FM_STATE_OVERRIDE:-}" ]; then - STATE=$(resolve_directory_input FM_STATE_OVERRIDE "$FM_STATE_OVERRIDE") || exit 1 -else - STATE="$FM_HOME/state" -fi +FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" +DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" +STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" KIND=ship HERDR_LAB=0 NO_PROJECTS=0 -FORCE_REGENERATE=0 POS=() for a in "$@"; do case "$a" in @@ -108,16 +80,10 @@ for a in "$@"; do --secondmate) KIND=secondmate ;; --herdr-lab) HERDR_LAB=1 ;; --no-projects) NO_PROJECTS=1 ;; - --force-regenerate) FORCE_REGENERATE=1 ;; *) POS+=("$a") ;; esac done -ID=${POS[0]:-} - -if [ -z "$ID" ]; then - echo "error: task id is required" >&2 - exit 1 -fi +ID=${POS[0]} if [ "$KIND" = secondmate ] && [ "$HERDR_LAB" -eq 1 ]; then echo "error: --herdr-lab applies only to crewmate ship or scout briefs" >&2 @@ -130,73 +96,8 @@ if [ "$NO_PROJECTS" -eq 1 ] && [ "$KIND" != secondmate ]; then fi BRIEF="$DATA/$ID/brief.md" -BRIEF_SAFETY_MARKER='<!-- firstmate-brief-scaffold-safety:v1 -->' -BRIEF_OUTPUT="$BRIEF" - -cleanup_staged_brief() { - if [ "$BRIEF_OUTPUT" != "$BRIEF" ]; then - rm -f -- "$BRIEF_OUTPUT" - fi -} - -brief_has_current_safety_marker() { - [ -f "$BRIEF" ] && grep -Fqx "$BRIEF_SAFETY_MARKER" "$BRIEF" -} - -prepare_brief_path() { - mkdir -p "$DATA/$ID" - if [ ! -e "$BRIEF" ]; then - return 0 - fi - - if [ "$FORCE_REGENERATE" -ne 1 ]; then - echo "error: $BRIEF already exists; refusing to overwrite or silently reuse it" >&2 - if brief_has_current_safety_marker; then - echo "error: current scaffold safety marker is present, but task freshness is unverified" >&2 - echo "error: inspect $BRIEF and verify it intentionally, or rerun with --force-regenerate to archive it and write a fresh scaffold" >&2 - else - echo "error: missing current scaffold safety marker; this brief may predate current safety contracts" >&2 - echo "error: Do not launch this brief unchanged; rerun the same scaffold command with --force-regenerate to archive it and write a fresh scaffold" >&2 - fi - return 1 - fi - - BRIEF_OUTPUT=$(mktemp "$DATA/$ID/.brief.md.XXXXXX") || { - echo "error: could not stage regenerated brief: $BRIEF" >&2 - return 1 - } - trap cleanup_staged_brief EXIT -} - -install_staged_brief() { - local archive_base archive timestamp suffix - [ "$BRIEF_OUTPUT" != "$BRIEF" ] || return 0 - - timestamp=$(date -u +%Y%m%dT%H%M%SZ) - archive_base="$BRIEF.archive-$timestamp" - archive=$archive_base - suffix=1 - while [ -e "$archive" ]; do - archive="$archive_base.$suffix" - suffix=$((suffix + 1)) - done - if [ -e "$BRIEF" ]; then - mv -- "$BRIEF" "$archive" || { - echo "error: could not archive existing brief: $BRIEF" >&2 - return 1 - } - fi - if ! mv -- "$BRIEF_OUTPUT" "$BRIEF"; then - if [ -e "$archive" ]; then - mv -- "$archive" "$BRIEF" || echo "error: could not restore existing brief: $BRIEF" >&2 - fi - echo "error: could not install regenerated brief: $BRIEF" >&2 - return 1 - fi - BRIEF_OUTPUT="$BRIEF" - trap - EXIT - [ -e "$archive" ] && echo "archived existing brief: $archive" -} +[ -e "$BRIEF" ] && { echo "error: $BRIEF already exists" >&2; exit 1; } +mkdir -p "$DATA/$ID" shell_quote() { printf "'" @@ -218,7 +119,6 @@ if [ "$NO_PROJECTS" -eq 1 ]; then else [ -n "$SECONDMATE_PROJECTS" ] || { echo "error: --secondmate requires at least one project, or --no-projects for a project-less home" >&2; exit 1; } fi -prepare_brief_path || exit 1 SECONDMATE_CHARTER=${FM_SECONDMATE_CHARTER:-"{TASK}"} SECONDMATE_SCOPE=${FM_SECONDMATE_SCOPE:-${FM_SECONDMATE_CHARTER:-"{TASK}"}} if [ "$NO_PROJECTS" -eq 1 ]; then @@ -228,8 +128,7 @@ else PROJECT_CLONES_BODY=$(printf '%s\n' "$SECONDMATE_PROJECTS" | tr ' ' '\n' | sed 's/^/- /') PROJECT_CLONES_NOTE="The projects above are local clones for work you supervise; they are not an exclusive ownership claim." fi -cat > "$BRIEF_OUTPUT" <<EOF -$BRIEF_SAFETY_MARKER +cat > "$BRIEF" <<EOF You are a persistent second mate managed by the main firstmate. Work on your own; do not wait for a human. # Charter @@ -285,7 +184,6 @@ When you have no assigned or in-flight work after that reconciliation, go idle a An empty queue is a healthy resting state, not a cue to invent work: never spawn a survey, audit, or any self-directed "find work" task on your own initiative. If this charter cannot be carried out, append \`blocked: {why}\` or \`failed: {why}\` to the main status file and stop. EOF -install_staged_brief || exit 1 if [ "$SECONDMATE_CHARTER" = "{TASK}" ]; then echo "scaffolded: $BRIEF (secondmate charter; replace {TASK})" else @@ -294,12 +192,7 @@ fi exit 0 fi -REPO=${POS[1]:-} -if [ -z "$REPO" ]; then - echo "error: repo name is required for ship and scout briefs" >&2 - exit 1 -fi -prepare_brief_path || exit 1 +REPO=${POS[1]} if [ "$HERDR_LAB" -eq 1 ]; then HERDR_LAB_HELPER=$(shell_quote "$FM_ROOT/bin/fm-herdr-lab.sh") @@ -334,8 +227,7 @@ HERDR_SECTION=${HERDR_SECTION%$'\n'} fi if [ "$KIND" = scout ]; then -cat > "$BRIEF_OUTPUT" <<EOF -$BRIEF_SAFETY_MARKER +cat > "$BRIEF" <<EOF You are a crewmate: an autonomous worker agent managed by firstmate. Work on your own; do not wait for a human. # Task @@ -378,7 +270,6 @@ Before reporting done, read and follow \`$FM_ROOT/.agents/skills/decision-hold-l When the report is complete, append \`done: {one-line conclusion}\` to the status file and stop. If your findings reveal work that should ship (e.g. you reproduced a bug and the fix is clear), say so in the report; firstmate may promote this task in place, and you would then receive mode-specific ship instructions as a follow-up message. EOF -install_staged_brief || exit 1 echo "scaffolded: $BRIEF (scout; replace {TASK})" exit 0 fi @@ -386,12 +277,8 @@ fi # Ship task: shape Setup / Rule 1 / Definition of done by the project's delivery mode. # yolo does not affect the brief because the worker never owns approval decisions; # firstmate applies the authority contract in AGENTS.md section 7, so discard it. -MODE_OUTPUT=$("$FM_ROOT/bin/fm-project-mode.sh" "$REPO") || { - echo "error: could not resolve delivery mode for $REPO" >&2 - exit 1 -} read -r MODE _ <<EOF -$MODE_OUTPUT +$("$FM_ROOT/bin/fm-project-mode.sh" "$REPO") EOF case "$MODE" in @@ -448,8 +335,7 @@ esac # briefs stay byte-identical to the historical Bash 5 output. DOD=${DOD%$'\n'} -cat > "$BRIEF_OUTPUT" <<EOF -$BRIEF_SAFETY_MARKER +cat > "$BRIEF" <<EOF You are a crewmate: an autonomous worker agent managed by firstmate. Work on your own; do not wait for a human. # Task @@ -500,5 +386,4 @@ Keep it proportionate: skip \`AGENTS.md\` edits for trivial tasks that produced $DOD EOF -install_staged_brief || exit 1 echo "scaffolded: $BRIEF (ship, mode=$MODE; replace {TASK})" diff --git a/tests/fm-lint.test.sh b/tests/fm-lint.test.sh index 17fb097f758..a2b3c8fb296 100755 --- a/tests/fm-lint.test.sh +++ b/tests/fm-lint.test.sh @@ -18,7 +18,11 @@ set -u . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" LINT="$ROOT/bin/fm-lint.sh" +CI="$ROOT/.github/workflows/ci.yml" +NM="$ROOT/.no-mistakes.yaml" INSTALLER="$ROOT/bin/fm-install-shellcheck.sh" +# The authoritative file set the one owner must run. +CANON='ROOTS=(bin/*.sh bin/backends/*.sh tests/*.sh)' # The pinned version, read from the single source (the one owner itself). REQUIRED=$("$LINT" --required-version) @@ -29,13 +33,48 @@ pinned_ready() { [ "$(shellcheck --version | awk '/^version:/ {print $2; exit}')" = "$REQUIRED" ] } -test_list_files_reports_the_shell_inventory() { +test_owner_exists_and_executable() { + assert_present "$LINT" "bin/fm-lint.sh is missing" + [ -x "$LINT" ] || fail "bin/fm-lint.sh must be executable so CI/gate can run it directly" + pass "one-owner lint script exists and is executable" +} + +test_owner_defines_canonical_set() { + assert_grep "$CANON" "$LINT" "fm-lint.sh must run the canonical shellcheck file set" + # It must not weaken CI: no severity downgrade and no blanket disable/exclude + # that would hide findings CI fails on. + assert_no_grep '--severity' "$LINT" "fm-lint.sh must not lower severity below the CI default" + assert_no_grep '--exclude' "$LINT" "fm-lint.sh must not blanket-exclude checks CI enforces" + assert_grep "\"\$FM_LINT_SHELLCHECK\" --norc --external-sources -- \"\${roots[@]}\"" "$LINT" "every bounded worker must ignore ambient config and preserve annotated production sources" + [ "$(grep -Fc -- '--norc --external-sources' "$LINT")" -eq 1 ] || fail "the one worker command must own ShellCheck configuration" + assert_grep "JOBS=\${FM_LINT_JOBS:-2}" "$LINT" "canonical lint must default to two bounded workers" + pass "fm-lint.sh is the sole authoritative definition at CI-default severity" +} + +test_ci_invokes_the_owner() { + grep -Eq '^ - run: bin/fm-lint\.sh$' "$CI" || fail "CI lint job must invoke the one-owner script as a run step" + # Guard against regression to an inline re-spelling of the command. + assert_no_grep 'run: shellcheck' "$CI" "CI must call fm-lint.sh, not re-spell shellcheck inline" + pass "CI lint job calls the one-owner script, not an inline command" +} + +test_stock_bash_parse_uses_owner_inventory() { local listed expected listed=$("$LINT" --list-files) expected=$(find bin bin/backends tests -maxdepth 1 -type f -name '*.sh' -print | LC_ALL=C sort) [ "$(printf '%s\n' "$listed" | LC_ALL=C sort)" = "$expected" ] \ - || fail "fm-lint.sh --list-files did not return the complete shell inventory" - pass "fm-lint.sh --list-files reports the complete shell inventory" + || fail "fm-lint.sh --list-files did not return the complete canonical shell inventory" + # shellcheck disable=SC2016 # Literal assertion must remain unexpanded. + assert_grep 'bin/fm-lint.sh --list-files > "$shell_inventory"' "$CI" \ + "stock macOS Bash parse sweep must consume fm-lint.sh's canonical inventory" + assert_no_grep 'for f in bin/*.sh bin/backends/*.sh tests/*.sh' "$CI" \ + "stock macOS Bash parse sweep must not duplicate the canonical inventory" + pass "stock macOS Bash parse sweep consumes the canonical lint inventory" +} + +test_nomistakes_invokes_the_owner() { + grep -Fqx " lint: 'bin/fm-lint.sh'" "$NM" || fail "no-mistakes commands.lint must map exactly to the one-owner script" + pass "no-mistakes pre-push lint calls the one-owner script" } test_pins_an_explicit_version() { @@ -46,6 +85,17 @@ test_pins_an_explicit_version() { pass "fm-lint.sh pins an explicit ShellCheck version ($REQUIRED)" } +test_ci_installs_and_logs_the_pinned_version() { + # CI must derive the version from the one owner (never hardcode a divergent + # number) and log the resolved version as parity evidence. + assert_grep "VERSION=\"\$(\"\$ROOT/bin/fm-lint.sh\" --required-version)\"" "$INSTALLER" "installer must read the version fm-lint.sh pins" + [ "$(grep -Fc "bin/fm-install-shellcheck.sh \"\$RUNNER_TEMP/bin\"" "$CI")" -eq 4 ] || fail "lint and all three portable behavior jobs must use the shared ShellCheck installer" + assert_grep "ACTUAL_SHA256=\$(sha256sum" "$INSTALLER" "installer must calculate the ShellCheck archive checksum" + assert_grep "[ \"\$ACTUAL_SHA256\" = \"\$SHA256\" ]" "$INSTALLER" "installer must verify the ShellCheck archive checksum" + assert_grep "\"\$DESTINATION/shellcheck\" --version" "$INSTALLER" "installer must log the resolved ShellCheck version as evidence" + pass "CI installs and logs the pinned ShellCheck version from the one owner" +} + test_installer_retries_transient_download_failure() { local tmp fakebin destination out tmp=$(fm_test_tmproot fm-shellcheck-download) @@ -202,6 +252,26 @@ SH pass "fm-lint.sh passes a clean fixture" } +test_source_graph_boundaries_keep_every_owner() { + local adapter file production_context_tests="" + [ "$(grep -Fc '# shellcheck source=/dev/null' "$ROOT/bin/fm-backend.sh")" -eq 5 ] \ + || fail "the dispatcher must stop static source following at all five dynamic adapters" + for adapter in tmux herdr zellij orca cmux; do + assert_present "$ROOT/bin/backends/$adapter.sh" "canonical adapter root is missing: $adapter" + done + assert_present "$ROOT/bin/fm-push-transition-lib.sh" "narrow push-transition owner is missing" + assert_grep '# shellcheck source=bin/fm-push-transition-lib.sh' "$ROOT/bin/fm-watch.sh" "the watcher must consume the narrow push-transition owner" + assert_grep ". \"\$ROOT/bin/fm-push-transition-lib.sh\"" "$ROOT/tests/fm-backend-herdr-eventwait-smoke.test.sh" "the Herdr event-wait smoke must consume the narrow production owner" + assert_no_grep '# shellcheck source=bin/fm-watch.sh' "$ROOT/tests/fm-backend-herdr-eventwait-smoke.test.sh" "the event-wait smoke must not re-import the whole watcher graph" + for file in "$ROOT"/tests/*.sh; do + grep -q '^[[:space:]]*# shellcheck source=bin/' "$file" || continue + production_context_tests="${production_context_tests}$(basename "$file")|" + done + [ "$production_context_tests" = 'fm-backend-herdr.test.sh|fm-daemon.test.sh|fm-pending-reply.test.sh|fm-secondmate-sync.test.sh|' ] \ + || fail "only callback/variable interop tests may retain production source context: $production_context_tests" + pass "dispatcher, adapters, production owner, and tests have explicit lint boundaries" +} + test_jobs_are_deterministic_and_complete() { if ! pinned_ready; then pass "SKIP (ShellCheck $REQUIRED not resolved): deterministic bounded jobs check" @@ -426,13 +496,19 @@ SH pass "seeded dispatcher, adapter, production-owner, and test-local diagnostics preserve parity" } -test_list_files_reports_the_shell_inventory +test_owner_exists_and_executable +test_owner_defines_canonical_set +test_ci_invokes_the_owner +test_stock_bash_parse_uses_owner_inventory +test_nomistakes_invokes_the_owner test_pins_an_explicit_version +test_ci_installs_and_logs_the_pinned_version test_installer_retries_transient_download_failure test_rejects_wrong_shellcheck_version test_catches_a_real_lint_defect test_ignores_ambient_shellcheck_opts test_clean_fixture_passes +test_source_graph_boundaries_keep_every_owner test_jobs_are_deterministic_and_complete test_worker_trees_stop_on_signal test_seeded_module_boundary_parity From 30765128645b7f0acc78ce4a1ff5735a32da1849 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Tue, 28 Jul 2026 12:05:56 -0700 Subject: [PATCH 23/52] test: stabilize tmux teardown conformance baseline (#1209) * fix(test): pin teardown tmux baseline to historical kill selectors merge-base HEAD main collapses to HEAD after the exact-selector change lands on the default branch, so the old teardown fixture was accidentally exercising current exact targets. Resolve a content-historical permissive tmux adapter from first-parent history and force that post-squash topology inside the conformance case so main and feature branches keep the same old-vs-new contract. * no-mistakes(lint): Suppress intentional literal-pattern ShellCheck warnings --- tests/fm-backend.test.sh | 77 ++++++++++++++++++++++++++++++++++++++-- 1 file changed, 75 insertions(+), 2 deletions(-) diff --git a/tests/fm-backend.test.sh b/tests/fm-backend.test.sh index 323cd4f5e51..922b227cec8 100755 --- a/tests/fm-backend.test.sh +++ b/tests/fm-backend.test.sh @@ -12,7 +12,10 @@ # binaries and fixtures as the REFACTORED versions in this checkout, then # diffs the two command logs byte-for-byte - the report's P1 checklist # item "run current main scripts and refactored scripts against the same -# fake tools and compare command logs". +# fake tools and compare command logs". The teardown old-vs-new case also +# overlays a content-historical permissive tmux kill fixture: after the +# exact-selector change lands on the default branch, merge-base with main +# collapses to HEAD and can no longer supply that baseline. # 3. Asserts the `--backend`/`FM_BACKEND` selection refuses unknown backends # and the blocked `codex-app` backend loudly. # @@ -80,6 +83,9 @@ SH } # The commit this branch started from - the P1 "current main" baseline. +# Suitable for byte-identical old-vs-new checks while a branch still diverges +# from main. After a squash lands, merge-base(HEAD, main) collapses to HEAD, so +# callers that need a true pre-change fixture must not rely on this alone. resolve_base_ref() { local ref base for ref in main refs/heads/main origin/main refs/remotes/origin/main origin/HEAD refs/remotes/origin/HEAD; do @@ -95,6 +101,30 @@ resolve_base_ref() { BASE_REF=$(resolve_base_ref) \ || fail "fm-backend baseline requires local main or origin/main; fetch the default branch before running this test" +# Newest first-parent revision whose bin/backends/tmux.sh still uses the +# pre-exact permissive kill-window target. Content-addressed from history so the +# fixture stays historical on default-branch CI and on branches cut after the +# exact-selector change, where merge-base with main is self-referential. +resolve_permissive_tmux_kill_ref() { + local commit body + while IFS= read -r commit; do + [ -n "$commit" ] || continue + body=$(git -C "$ROOT" show "$commit:bin/backends/tmux.sh" 2>/dev/null) || continue + # shellcheck disable=SC2016 + case "$body" in + *'tmux kill-window -t "=$session:=$window"'*) continue ;; + esac + # shellcheck disable=SC2016 + case "$body" in + *'tmux kill-window -t "$1"'*|*'tmux kill-window -t "$target"'*) + printf '%s\n' "$commit" + return 0 + ;; + esac + done < <(git -C "$ROOT" log --first-parent --format='%H' HEAD -- bin/backends/tmux.sh) + return 1 +} + # --- shared: a pre-refactor bin/ shim -------------------------------------- # # build_old_bin echoes a directory whose bin/ subdir holds the PRE-REFACTOR @@ -929,10 +959,52 @@ run_teardown_case() { "$script" "$id" } +test_permissive_tmux_kill_ref_stays_historical() { + local ref body_hist body_head head + head=$(git -C "$ROOT" rev-parse HEAD) + ref=$(resolve_permissive_tmux_kill_ref) \ + || fail "unable to locate a historical bin/backends/tmux.sh with permissive kill-window selectors" + body_hist=$(git -C "$ROOT" show "$ref:bin/backends/tmux.sh") \ + || fail "could not read historical tmux adapter at $ref" + body_head=$(cat "$ROOT/bin/backends/tmux.sh") + + # shellcheck disable=SC2016 + case "$body_hist" in + *'tmux kill-window -t "=$session:=$window"'*) + fail "resolve_permissive_tmux_kill_ref returned exact selectors at $ref" + ;; + esac + # shellcheck disable=SC2016 + case "$body_hist" in + *'tmux kill-window -t "$1"'*|*'tmux kill-window -t "$target"'*) ;; + *) fail "historical tmux adapter at $ref lacks a permissive kill-window target" ;; + esac + # shellcheck disable=SC2016 + case "$body_head" in + *'tmux kill-window -t "=$session:=$window"'*) ;; + *) fail "current tmux adapter lost exact kill-window selectors" ;; + esac + [ "$ref" != "$head" ] \ + || fail "permissive tmux baseline collapsed to HEAD; fixture is no longer historical" + + pass "historical permissive tmux kill baseline stays distinct from current exact selectors" +} + test_teardown_conformance_old_vs_new() { - local old_bin fb proj wt id + local old_bin fb proj wt id old_tmux_ref saved_base_ref local state_old state_new config_old config_new data log_old log_new out_old out_new rc_old rc_new + # Force the post-squash topology inside this case: merge-base with main may + # equal HEAD on default-branch CI, and that must not make the legacy kill + # fixture self-referential. build_old_bin still uses BASE_REF for entrypoints; + # only the tmux kill adapter is pinned to the content-historical permissive ref. + saved_base_ref=$BASE_REF + BASE_REF=$(git -C "$ROOT" rev-parse HEAD) + old_tmux_ref=$(resolve_permissive_tmux_kill_ref) \ + || { BASE_REF=$saved_base_ref; fail "unable to locate a historical bin/backends/tmux.sh with permissive kill-window selectors"; } old_bin=$(build_old_bin teardown-old) + git -C "$ROOT" show "$old_tmux_ref:bin/backends/tmux.sh" > "$old_bin/bin/backends/tmux.sh" \ + || { BASE_REF=$saved_base_ref; fail "could not materialize historical tmux adapter from $old_tmux_ref"; } + BASE_REF=$saved_base_ref proj="$TMP_ROOT/teardown-project"; wt="$TMP_ROOT/teardown-wt" id="teardownconform1" fm_git_worktree "$proj" "$wt" "fm/$id" @@ -1106,6 +1178,7 @@ test_backend_of_selector_matches_explicit_target_meta test_send_conformance_old_vs_new test_peek_conformance_old_vs_new test_spawn_symlinked_project_prefix_avoids_false_refusal +test_permissive_tmux_kill_ref_stays_historical test_teardown_conformance_old_vs_new test_spawn_refuses_unknown_backend_flag test_spawn_refuses_codex_app_backend_flag From a8a68434f17a69e4d9034d4e8258051a69137169 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Tue, 28 Jul 2026 12:25:52 -0700 Subject: [PATCH 24/52] docs: slim quota-array-dispatch to the pace selection core (#1197) Cut the runtime skill to the compact pace-aware selection procedure plus minimum owner pointers. Keep every distinct decision rule and move expanded acceptance scenarios to deterministic fixture ownership assertions. Size: 170/1374/10187 -> 63/544/4068 (about 63%/60%/60% reduction). --- .agents/skills/quota-array-dispatch/SKILL.md | 173 ++++--------------- tests/fm-quota-array-dispatch.test.sh | 39 +++-- 2 files changed, 57 insertions(+), 155 deletions(-) diff --git a/.agents/skills/quota-array-dispatch/SKILL.md b/.agents/skills/quota-array-dispatch/SKILL.md index a5fe06d6ec6..d9de90ffba2 100644 --- a/.agents/skills/quota-array-dispatch/SKILL.md +++ b/.agents/skills/quota-array-dispatch/SKILL.md @@ -12,159 +12,52 @@ metadata: # quota-array-dispatch This skill is the single owner of the pace-aware profile-array selection procedure. -The concise always-loaded intake boundary remains in `AGENTS.md` section 4. -`docs/configuration.md` owns the `config/crew-dispatch.json` schema only. +`AGENTS.md` section 4 owns the always-loaded intake boundary, load trigger, malformed-config refusal, every-candidate accounting, and strongest-reasoning/tie safety rules. +`harness-adapters` owns harness verification, model/provider discovery, and effort fallback. `quota-axi` remains data-only and never recommends a route. -Firstmate owns the judgment. Do not add a daemon, opaque composite score, routing wrapper, hard-coded model-specific policy, or producer-side route recommendation. -## When to load +## Collect facts -Load this skill whenever a matched dispatch rule or the configured default resolves to a profile array (more than one candidate), before choosing the concrete `--harness`, `--model`, and `--effort` passed to `fm-spawn`. -Keep using `harness-adapters` for harness verification, model/provider discovery, and effort fallback. +Run `quota-axi --json` once per intake and reuse that snapshot for every candidate. +For each candidate, establish the harness/model/provider relationship from `harness-adapters`, then record only inspectable facts: -## Intake boundary this skill does not relax +- task/profile fit and required reasoning class +- raw applicable headroom (`effectivePercentRemaining` or the tightest applicable remaining percentage) +- effective pace status, signed reserve per applicable window, and worst applicable reserve (`worstReservePercentPoints` when present, else the minimum signed reserve) +- whether any applicable window or effective summary is ahead of reset, or any applicable pace is `unknown` +- schema note when pace fields are absent -1. Explicit per-task captain overrides still win over configured profiles. -2. Configured profile matching precedence is unchanged: best-fit rule, then configured default, then static crewmate harness. -3. Malformed `config/crew-dispatch.json` remains an actionable error; never select around it. -4. Every configured candidate in the matched array must be accounted for. -5. If any harness/model/provider relationship, applicable quota data, or interpretation cannot be established, stop and report that candidate instead of omitting it, guessing, falling back, or calling the result quota-informed. -6. When every candidate is tight, preserve the captain's strongest-reasoning class rather than silently downgrading it solely to conserve quota; stop and report the tight choice if that class cannot proceed. -7. Genuine ties must remain free of array-order or harness bias. +Stale raw windows are diagnostic only, never current headroom. +Read every bounding window named by `boundedBy`, `limitingWindowIds`, `aheadWindowIds`, `behindWindowIds`, `onPaceWindowIds`, and `unknownWindowIds`. -## Collect inspectable facts for every candidate +## Pace semantics -For each candidate profile: +`reservePercentPoints = percentRemaining - timeRemainingPercent`. +Negative reserve means usage is ahead of reset pace and creates conservation pressure. +Positive reserve means usage is behind reset pace. +`on_pace` is neutral. +Conservation pressure is present when effective pace status is `ahead`, effective pace status is `mixed` and any `aheadWindowIds` remain, or any applicable bounding window itself has pace status `ahead`. +`unknown` is valid explicit uncertainty from quota-axi, not a parser failure and not permission to assume the window is healthy or exhausted. -1. Establish the harness/model/provider relationship from current authoritative discovery owned by `harness-adapters`. - Fail loudly on an unresolved relationship. -2. Run `quota-axi --json` once per intake and reuse that snapshot for every candidate. -3. Require a current provider report with known quota semantics and a known applicable effective-availability record for that candidate's provider and model scope. - Stale raw windows remain diagnostic evidence only and are never current headroom. -4. Read every bounding window relevant to that candidate, including windows named by `boundedBy`, `limitingWindowIds`, `aheadWindowIds`, `behindWindowIds`, `onPaceWindowIds`, and `unknownWindowIds` on the effective record. -5. Record these inspectable facts, never a hidden score: - - task/profile fit - - reasoning class required by the captain request or task ambiguity - - raw applicable headroom (`effectivePercentRemaining` or the tightest applicable remaining percentage) - - effective pace status when present - - signed reserve for each applicable window and the effective worst reserve when present - - whether any applicable window or effective summary is ahead of reset - - whether any applicable pace is `unknown` - - schema compatibility note when pace fields are absent +## Selection order -## Pace signals +Apply only among candidates that already satisfy required fit and the strongest reasoning class the request needs. +Never use pace or raw headroom to silently replace that reasoning class. -quota-axi `schemaVersion` 3 window pace uses: - -- `reservePercentPoints = percentRemaining - timeRemainingPercent` -- Negative reserve means usage is ahead of reset pace and creates conservation pressure. -- Positive reserve means usage is behind reset pace. -- `on_pace` is neutral. - -Effective-availability pace summaries may report `ahead`, `behind`, `on_pace`, `mixed`, or `unknown`. - -Treat conservation pressure as present when: - -- effective pace status is `ahead`, or -- effective pace status is `mixed` and any `aheadWindowIds` remain, or -- any applicable bounding window itself has pace status `ahead`. - -An effective `mixed` result is never healthy merely because one window is behind. -Any remaining `aheadWindowIds` keep conservation pressure. - -Signed reserve comparison uses the worst applicable reserve, preferring the producer field `worstReservePercentPoints` when present and otherwise the minimum signed reserve across applicable bounding windows. - -## Selection procedure - -Apply these steps only among candidates that already satisfy required task/profile fit and the strongest reasoning class the request genuinely needs. -Never use pace or raw headroom to silently replace that reasoning class with a weaker one. - -1. **Unresolved relationship or quota data** - Stop and report the blocked candidate. -2. **Strongest-reasoning / all-tight** - If every remaining candidate is tight, keep the strongest-reasoning class and either dispatch inside that class or stop and report that the tight choice cannot proceed. - Do not conserve quota through an unapproved downgrade. -3. **Conservation pressure vs sustainable pace** - When fit and reasoning class are comparable, prefer a candidate without ahead-of-reset conservation pressure over one with conservation pressure, even when the pressured candidate has somewhat higher raw remaining percentage. -4. **Among pressured candidates** - Prefer the least-negative worst applicable reserve. - Example: worst reserve `-4` is safer than `-18` when other inspectable facts are comparable. -5. **Among sustainable candidates** - Use known behind/on-pace evidence plus raw headroom transparently. - Do not collapse those facts into an opaque composite score. +1. Unresolved relationship or quota data: stop and report the blocked candidate. +2. All-tight: keep the strongest-reasoning class; dispatch inside it or stop and report that the tight choice cannot proceed. +3. When fit and reasoning are comparable, prefer a candidate without ahead-of-reset conservation pressure over one with conservation pressure, even when the pressured candidate has somewhat higher raw remaining percentage. +4. Among pressured candidates, prefer the least-negative worst applicable reserve. +5. Among sustainable candidates, use known behind/on-pace evidence plus raw headroom transparently. Prefer known sustainable evidence over `unknown` pace when otherwise comparable. - Between known sustainable candidates, prefer the clearly better inspectable pair of pace reserve and raw headroom; state both facts in the choice rationale. -6. **Unknown pace** - `unknown` is valid explicit uncertainty from quota-axi, not a parser failure and not permission to assume the window is healthy or exhausted. - Inspect `unknownWindowIds` and each window's pace `reason` so the rationale preserves the producer's stated uncertainty. - Prefer known sustainable evidence when otherwise comparable. - If the dispatch choice materially hinges on unresolved pace, report the uncertainty rather than inventing a conclusion. -7. **Absent pace / older schema** - `schemaVersion` 2 payloads or missing pace fields must degrade explicitly and safely. - Do not crash, fabricate pace, or silently reinterpret absence as healthy/`on_pace`. - Compare raw applicable headroom only, using known effective availability rather than stale or isolated window percentages, state that pace is unavailable, and keep every other safety rule above. -8. **Genuine ties** - If every inspectable selection fact is equal, stop and report every tied candidate for captain choice. + Do not collapse those facts into an opaque composite score. +6. If the dispatch choice materially hinges on unresolved pace, report the uncertainty rather than inventing a conclusion. +7. Absent pace or older schema: do not crash, fabricate pace, or silently reinterpret absence as healthy/`on_pace`. + Compare raw applicable headroom only, state that pace is unavailable, and keep every other safety rule. +8. Genuine ties: stop and report every tied candidate for captain choice. Do not select by array order, harness name, or another arbitrary identity ordering. Report duplicate concrete profiles as a configuration error. -The intake rationale must name the inspectable facts used for every candidate. +Name the inspectable facts used for every candidate. Never conclude with an unexplained "best quota" label. - -## Acceptance scenarios - -These scenarios are normative examples of the procedure above. - -### Higher raw quota but materially ahead vs lower raw quota on/behind pace - -Candidate A has higher `effectivePercentRemaining` but conservation pressure from an ahead bounding window. -Candidate B has lower raw headroom, no conservation pressure, and known behind or on-pace evidence. -Choose B when fit and reasoning class are comparable. - -### Mixed effective pace with an ahead bound - -Effective pace status is `mixed` and `aheadWindowIds` is non-empty. -Treat the candidate as conservation-pressured even if another window is behind or on pace. - -### Both candidates ahead with different worst reserves - -Both candidates have conservation pressure. -Choose the least-negative worst applicable reserve when fit and reasoning class are comparable. - -### Known sustainable versus unknown - -Candidate A has known behind or on-pace evidence. -Candidate B has comparable fit, reasoning class, and raw headroom but `unknown` pace. -Prefer A. -If the only way to prefer one side depends on unresolved pace and no known sustainable candidate remains, report the uncertainty. - -### Every candidate tight while strongest-reasoning applies - -All candidates are tight on real headroom. -Keep the strongest reasoning class required by the request. -Do not pick a weaker class only to save quota. -Dispatch inside that class or stop and report that the tight strongest-class choice cannot proceed. - -### Genuine tie without array-order or harness bias - -Two candidates match on fit, reasoning class, conservation pressure, worst reserve, pace class, raw headroom, and unknown flags. -Choosing either array order or a standing harness preference is forbidden. -Stop and report both tied candidates for captain choice. - -### schemaVersion 2 or absent-pace compatibility - -Older quota-axi output or missing pace fields still allow array resolution. -Compare raw headroom only, state that pace is unavailable, and do not invent ahead/behind/on_pace. - -## Sanitized producer shape - -Validate consumers against a sanitized `schemaVersion` 3 shape derived from quota-axi 0.1.15: - -- top level: `schemaVersion`, `generatedAt`, `providers[]` -- each provider: `provider`, `state`, `windows[]`, and optional `quotaSemantics` with `status` and `effectiveAvailability[]` -- each window: `id`, `label`, `kind`, and optional `percentRemaining` and `pace`; pace has `status` plus optional `reason`, `timeRemainingPercent`, and `reservePercentPoints` -- each effective-availability entry: `scope`, `status`, `boundedBy`, optional `effectivePercentRemaining`, optional `limitingWindowIds`, and optional pace summary -- each effective pace summary: `status` plus optional `aheadWindowIds`, `behindWindowIds`, `onPaceWindowIds`, `unknownWindowIds`, `worstReservePercentPoints`, and `worstReserveWindowId` - -Never persist live provider balances, reset timestamps, account identifiers, or other private account details in tracked fixtures. diff --git a/tests/fm-quota-array-dispatch.test.sh b/tests/fm-quota-array-dispatch.test.sh index a958e56c306..0c0d848fa05 100755 --- a/tests/fm-quota-array-dispatch.test.sh +++ b/tests/fm-quota-array-dispatch.test.sh @@ -153,8 +153,8 @@ test_owner_and_always_loaded_boundary() { pass "quota-array-dispatch has one conditional owner and a concise always-loaded boundary" } -test_owner_contains_acceptance_procedure() { - local phrase +test_owner_contains_selection_procedure() { + local phrase lines words bytes for phrase in \ 'reservePercentPoints = percentRemaining - timeRemainingPercent' \ 'Negative reserve means usage is ahead of reset pace and creates conservation pressure' \ @@ -163,30 +163,39 @@ test_owner_contains_acceptance_procedure() { 'effective pace status is `mixed` and any `aheadWindowIds` remain' \ 'prefer a candidate without ahead-of-reset conservation pressure over one with conservation pressure' \ 'even when the pressured candidate has somewhat higher raw remaining percentage' \ - 'Prefer the least-negative worst applicable reserve' \ - 'Use known behind/on-pace evidence plus raw headroom transparently' \ + 'prefer the least-negative worst applicable reserve' \ + 'use known behind/on-pace evidence plus raw headroom transparently' \ 'Do not collapse those facts into an opaque composite score' \ '`unknown` is valid explicit uncertainty from quota-axi' \ 'Prefer known sustainable evidence over `unknown` pace when otherwise comparable' \ 'If the dispatch choice materially hinges on unresolved pace, report the uncertainty' \ - 'Do not crash, fabricate pace, or silently reinterpret absence as healthy' \ + 'do not crash, fabricate pace, or silently reinterpret absence as healthy' \ 'stop and report every tied candidate for captain choice' \ 'Do not select by array order, harness name, or another arbitrary identity ordering' \ - 'Do not add a daemon, opaque composite score, routing wrapper, hard-coded model-specific policy'; do + 'Do not add a daemon, opaque composite score, routing wrapper, hard-coded model-specific policy' \ + 'Report duplicate concrete profiles as a configuration error' \ + 'Name the inspectable facts used for every candidate'; do assert_grep "$phrase" "$OWNER" "quota-array-dispatch procedure lost '$phrase'" done + # Expanded acceptance scenarios live in deterministic fixtures, not runtime prose. for phrase in \ 'Higher raw quota but materially ahead vs lower raw quota on/behind pace' \ - 'Mixed effective pace with an ahead bound' \ - 'Both candidates ahead with different worst reserves' \ - 'Known sustainable versus unknown' \ - 'Every candidate tight while strongest-reasoning applies' \ - 'Genuine tie without array-order or harness bias' \ - 'schemaVersion 2 or absent-pace compatibility'; do - assert_grep "$phrase" "$OWNER" "acceptance scenario missing: $phrase" + 'Sanitized producer shape' \ + '## When to load' \ + '## Intake boundary this skill does not relax'; do + if grep -Fq -- "$phrase" "$OWNER"; then + fail "quota-array-dispatch should not keep removed runtime prose: $phrase" + fi done - pass "quota-array-dispatch owns the full pace procedure and acceptance scenarios" + + lines=$(wc -l < "$OWNER" | tr -d ' ') + words=$(wc -w < "$OWNER" | tr -d ' ') + bytes=$(wc -c < "$OWNER" | tr -d ' ') + [ "$lines" -le 65 ] || fail "quota-array-dispatch skill is too long: $lines lines (want <= 65)" + [ "$words" -le 550 ] || fail "quota-array-dispatch skill is too wordy: $words words (want <= 550)" + [ "$bytes" -le 4600 ] || fail "quota-array-dispatch skill is too large: $bytes bytes (want <= 4600)" + pass "quota-array-dispatch owns the compact pace procedure ($lines lines, $words words, $bytes bytes)" } test_cross_references_stay_pointers() { @@ -271,7 +280,7 @@ test_no_duplicate_procedure_in_agents() { } test_owner_and_always_loaded_boundary -test_owner_contains_acceptance_procedure +test_owner_contains_selection_procedure test_cross_references_stay_pointers test_schema_v3_shape_fixture test_deterministic_acceptance_cases From c53c9e955cc60be4a9ce3117c233627516560f6b Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Tue, 28 Jul 2026 16:54:20 -0700 Subject: [PATCH 25/52] feat(bin): inherit backend config into secondmate homes (#1219) * Inherit config/backend into secondmate homes with deliberate-override preservation Add backend to the shared inheritable config allowlist so launch, locked bootstrap, and config-push converge a primary pin into secondmate homes as each home local future-spawn default. Track last-inherited bytes in a private state provenance marker so deliberate per-home overrides survive present and absent primary convergence, keep --backend and FM_BACKEND stronger, and extend the existing inheritance tests plus docs and skill claims. * no-mistakes(review): Preserve equal unprovenanced backend overrides * no-mistakes(review): Preserve symlink overrides and verify spawn precedence * no-mistakes(review): Snapshot backend inheritance for consistent provenance * no-mistakes(review): Simplify backend inheritance to primary-authoritative convergence * no-mistakes(document): Document inherited backend override preservation * fix: restore primary-authoritative backend inheritance after document regression The document step reintroduced provenance and deliberate per-home override semantics after review had simplified config/backend to plain primary-authoritative allowlist membership. Restore the primary-always-wins path: present overwrites, absent removes, no provenance marker, and docs/tests match that contract. * no-mistakes(review): Add divergent backend precedence regression fixtures * no-mistakes(document): Document backend inheritance contract --- .../skills/secondmate-provisioning/SKILL.md | 4 +- AGENTS.md | 2 +- bin/fm-config-inherit-lib.sh | 49 +---- docs/configuration.md | 4 +- .../fm-backend-herdr-presentation-e2e.test.sh | 2 +- tests/fm-secondmate-harness.test.sh | 173 +++++++++++++++--- 6 files changed, 160 insertions(+), 74 deletions(-) diff --git a/.agents/skills/secondmate-provisioning/SKILL.md b/.agents/skills/secondmate-provisioning/SKILL.md index 44caa0cb1c1..f9e68937ab9 100644 --- a/.agents/skills/secondmate-provisioning/SKILL.md +++ b/.agents/skills/secondmate-provisioning/SKILL.md @@ -78,7 +78,7 @@ This section is the single owner of the secondmate sync and inherited-local-mate Before launch, `fm-spawn.sh --secondmate` locally fast-forwards the home to the primary firstmate checkout's current default-branch commit when it is safe; dirty, diverged, or in-flight homes launch unchanged with a warning. The locked session-start bootstrap sweep runs the same guarded fast-forward for every live secondmate home, discovered from `state/<id>.meta` records with `kind=secondmate` (`data/secondmates.md` only backfills `home=` for older records). That no-fetch path is a purely local fast-forward of tracked files, never an origin fetch, and it never touches the gitignored operational dirs, so a secondmate's backlog, projects, and in-flight work are never disturbed; a linked worktree advances immediately, while a standalone clone that lacks the target receives firstmate updates through `/updatefirstmate`'s origin refresh. -The same launch and the same locked bootstrap sweep also propagate the primary's declared inherited local material: `config/crew-dispatch.json`, `config/crew-harness`, `config/backlog-backend`, `config/backend`, `config/herdr-presentation-spaces`, `config/startup-memory-budget`, and the one shared captain-preference file `data/captain-shared.md`. +The same launch and the same locked bootstrap sweep also propagate the primary's declared inherited local material: `config/crew-dispatch.json`, `config/crew-harness`, `config/backlog-backend`, `config/backend`, `config/herdr-presentation-spaces`, and the one shared captain-preference file `data/captain-shared.md`. Because these paths are gitignored, that propagation is a separate, primary-authoritative copy independent of the tracked-files fast-forward: it re-converges every live home whether or not its tracked files advanced, and it touches only the declared items. Propagation failures warn without blocking secondmate launch or session-start continuation, and the destination keeps whatever safely validated state the helper left behind. Inheritance copies the literal `config/crew-harness` file, so a secondmate's own crewmates use the primary's crewmate harness only when it names a concrete adapter such as `codex`; an unset or `default` value has nothing concrete to inherit, and the secondmate's own crewmates fall back to the secondmate's own or detected harness instead. @@ -99,7 +99,7 @@ Keep every `data/learnings.md` fully local by captain decision; route fleet-gene No AGENTS.md reread nudge is needed at spawn or respawn because the agent reads instructions fresh on launch; only the bootstrap sweep's running-home instruction-surface advance needs that AGENTS.md re-read. Bootstrap reports successful AGENTS.md re-read sends as `BOOTSTRAP_INFO:` and only emits `NUDGE_SECONDMATES:` when that send fails and needs retry. A separate, literal-content config reread is required whenever inherited `config/*` material changes under an already-running secondmate. -After each successful allowlisted config write, both the locked bootstrap convergence path and mid-session `bin/fm-config-push.sh` use the shared propagation report to build one per-home generation-specific private instruction file from the validated destination post-write bytes for only the allowlisted config items that actually changed for that home (`config/crew-dispatch.json`, `config/crew-harness`, `config/backlog-backend`, `config/backend`, `config/herdr-presentation-spaces`, `config/startup-memory-budget`), in deterministic allowlist order. +After each successful allowlisted config write, both the locked bootstrap convergence path and mid-session `bin/fm-config-push.sh` use the shared propagation report to build one per-home generation-specific private instruction file from the validated destination post-write bytes for only the allowlisted config items that actually changed for that home (`config/crew-dispatch.json`, `config/crew-harness`, `config/backlog-backend`, `config/backend`, `config/herdr-presentation-spaces`), in deterministic allowlist order. Each changed path is printed with clear begin/end delimiters and the destination file's full exact new bytes unparsed, or the explicit token `ABSENT` when propagation removed the destination copy. The instruction uses only minimal framing that these are defaults/rules and do not remove judgment; it never includes SHA values, selected profiles, parsed summaries, or any other generated interpretation. `data/captain-shared.md` is not a config file and is never inlined into this instruction file or message. diff --git a/AGENTS.md b/AGENTS.md index 4991118cb4b..82bcbe8e55a 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -67,7 +67,7 @@ config/crew-harness crewmate harness override; LOCAL, gitignored; absent or "de config/crew-dispatch.json optional crewmate dispatch profiles; LOCAL, gitignored; firstmate-maintained but human-editable natural-language rules that choose a per-task harness/model/effort profile (section 4). Inherited by secondmate homes config/secondmate-harness harness the PRIMARY uses to launch SECONDMATE agents, optionally followed by a model and effort token on the same line ("<harness> [<model>] [<effort>]"; section 4); LOCAL, gitignored; absent or "default" harness falls back to config/crew-harness then firstmate's own. The primary's own setting; NOT inherited into secondmate homes (secondmates do not spawn secondmates) config/backlog-backend backlog backend override; LOCAL, gitignored; absent or "tasks-axi" = default tasks-axi backend, "manual" = force routine backlog updates to hand-editing; inherited by secondmate homes (section 10) -config/backend runtime session-provider backend override for new tasks; LOCAL, gitignored; absent = falls through to runtime auto-detection (the runtime firstmate itself is executing inside), then tmux; tmux is the verified reference backend (docs/tmux-backend.md), while herdr, zellij, orca, and cmux are experimental spawn backends (docs/herdr-backend.md, docs/zellij-backend.md, docs/orca-backend.md, docs/cmux-backend.md) - herdr and cmux can also be selected by runtime auto-detection, zellij and orca never are (always explicit), and codex-app is not accepted; see docs/codex-app-backend.md; not inherited into secondmate homes +config/backend runtime session-provider backend override for new tasks; LOCAL, gitignored; absent = falls through to runtime auto-detection (the runtime firstmate itself is executing inside), then tmux; tmux is the verified reference backend (docs/tmux-backend.md), while herdr, zellij, orca, and cmux are experimental spawn backends (docs/herdr-backend.md, docs/zellij-backend.md, docs/orca-backend.md, docs/cmux-backend.md) - herdr and cmux can also be selected by runtime auto-detection, zellij and orca never are (always explicit), and codex-app is not accepted; see docs/codex-app-backend.md; inherited by secondmate homes under the primary-authoritative contract in secondmate-provisioning config/calm Pi Calm presentation preference; LOCAL, gitignored, and not inherited; see docs/configuration.md "Pi Calm preference" config/herdr-presentation-spaces optional presence flag for Herdr's default-off disposable single-task visual projection; LOCAL, gitignored; inherited by secondmate homes; see docs/herdr-backend.md "Optional presentation spaces" config/cmux-socket-password optional cmux control-socket password; LOCAL, gitignored; read fresh on every cmux CLI call and passed through without ever overriding an operator's own ambient CMUX_SOCKET_PASSWORD when absent (docs/cmux-backend.md "Setup") diff --git a/bin/fm-config-inherit-lib.sh b/bin/fm-config-inherit-lib.sh index bffbd5234d7..22109aa87aa 100644 --- a/bin/fm-config-inherit-lib.sh +++ b/bin/fm-config-inherit-lib.sh @@ -6,8 +6,7 @@ # profile rules, primary config/crew-harness=codex makes a secondmate's crewmates # spawn on codex too, primary config/backlog-backend=manual makes that home # hand-edit backlog files too, primary config/backend pins that home's local -# runtime-backend default for future spawns, primary config/startup-memory-budget -# bounds that home's startup-memory curation, and primary +# runtime-backend default for future spawns, and primary # config/herdr-presentation-spaces enables the same default-off Herdr presentation # projection). It also pushes the one primary-authoritative shared # captain-preference file, data/captain-shared.md, into each secondmate home's @@ -33,9 +32,6 @@ # is deliberately NOT in the list: it is the primary's own setting for launching # secondmates, and a secondmate never spawns secondmates, so it must not flow # downstream. -# -# shellcheck source=bin/fm-startup-memory-budget-lib.sh -. "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/fm-startup-memory-budget-lib.sh" # The one shared data file in this inheritance contract. There is deliberately # no shared learnings file. @@ -46,7 +42,7 @@ FM_SHARED_CAPTAIN_MODE="444" # The declared inheritable set (space-separated, config-dir-relative item paths). # Extend here to inherit more of the primary's local config; override via the # environment only in tests. Items must not contain whitespace. -FM_INHERITABLE_CONFIG="${FM_INHERITABLE_CONFIG:-crew-dispatch.json crew-harness backlog-backend backend herdr-presentation-spaces startup-memory-budget}" +FM_INHERITABLE_CONFIG="${FM_INHERITABLE_CONFIG:-crew-dispatch.json crew-harness backlog-backend backend herdr-presentation-spaces}" fm_inherit_file_mode() { if [ "$(uname)" = Darwin ]; then @@ -405,47 +401,6 @@ propagate_inheritable_config() { esac src="$src_config/$item" dest="$dest_config/$item" - # This one scalar config is consumed as a local safety boundary, so reject - # every unsafe or malformed source/destination artifact before the generic - # byte-copy behavior below can treat it as ordinary inherited material. - if [ "$item" = "$FM_STARTUP_MEMORY_BUDGET_FILE" ]; then - if [ -e "$src_config" ] || [ -L "$src_config" ]; then - if ! fm_startup_memory_budget_config_dir_safe "$src_config"; then - reason="unsafe primary config directory: $FM_STARTUP_MEMORY_BUDGET_ERROR" - warn_inheritable_config_error "$item" "$src_config" "$reason" - record_inheritable_config_result "$item" error "$reason" - rc=1 - continue - fi - fi - if [ -e "$dest_config" ] || [ -L "$dest_config" ]; then - if ! fm_startup_memory_budget_config_dir_safe "$dest_config"; then - reason="unsafe destination config directory: $FM_STARTUP_MEMORY_BUDGET_ERROR" - warn_inheritable_config_error "$item" "$dest_config" "$reason" - record_inheritable_config_result "$item" error "$reason" - rc=1 - continue - fi - fi - if [ -e "$src" ] || [ -L "$src" ]; then - if ! fm_startup_memory_budget_file_valid "$src"; then - reason="unsafe or invalid primary source: $FM_STARTUP_MEMORY_BUDGET_ERROR" - warn_inheritable_config_error "$item" "$src" "$reason" - record_inheritable_config_result "$item" error "$reason" - rc=1 - continue - fi - fi - if [ -e "$dest" ] || [ -L "$dest" ]; then - if ! fm_startup_memory_budget_file_valid "$dest"; then - reason="unsafe or invalid destination: $FM_STARTUP_MEMORY_BUDGET_ERROR" - warn_inheritable_config_error "$item" "$dest" "$reason" - record_inheritable_config_result "$item" error "$reason" - rc=1 - continue - fi - fi - fi if [ -f "$src" ]; then if ! destination_allows_inherited_item "$dest_config" "$item"; then reason=$(inheritable_config_skip_reason) diff --git a/docs/configuration.md b/docs/configuration.md index d9a06bf4cad..1b8ea3bb592 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -93,7 +93,7 @@ Use the guarded cleanup path described in [`docs/zellij-backend.md`](zellij-back cmux has no session layer at all - one workspace per task, in whatever cmux window is open - and its socket password (when configured) is read from local, gitignored `config/cmux-socket-password` under the effective config directory, never committed. The caller-facing label remains `fm-<id>`, but the actual cmux workspace title is scoped by the active `FM_HOME` readable label plus a short hash of the resolved `FM_ROOT` path as `fm-<home-label>-<id>`. Test cleanup must use the guarded path in [`docs/cmux-backend.md`](cmux-backend.md#current-operation-and-safety), never enumerate-and-close every workspace. -The `config/backend` file is not inherited by secondmate homes. +`config/backend` is inherited into secondmate homes under the primary-authoritative contract owned by [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md). ## Away-mode supervisor backend (FM_SUPERVISOR_BACKEND / FM_SUPERVISOR_TARGET) @@ -283,7 +283,7 @@ When a running home advances and its loaded instruction surface (`AGENTS.md`, `b If that send fails, bootstrap keeps an idempotent retry marker and emits `NUDGE_SECONDMATES:` with the failure reason. The same bootstrap run emits `SECONDMATE_LIVENESS:` only when a registered secondmate is skipped or its relaunch fails; already-live and successfully relaunched secondmates are handled silently. For a mid-session inherited local-material edit where tracked-file sync is not needed, run `bin/fm-config-push.sh`. -It uses the same live secondmate discovery and propagation helper as bootstrap, prints each live home's `crew-dispatch.json`, `crew-harness`, `backlog-backend`, `herdr-presentation-spaces`, and `data/captain-shared.md` result as `pushed`, `unchanged`, `skipped`, or `error`, and exits non-zero for real propagation errors or config-reread send failures. +It uses the same live secondmate discovery and propagation helper as bootstrap, prints each live home's `crew-dispatch.json`, `crew-harness`, `backlog-backend`, `backend`, `herdr-presentation-spaces`, and `data/captain-shared.md` result as `pushed`, `unchanged`, `skipped`, or `error`, and exits non-zero for real propagation errors or config-reread send failures. When an allowlisted config item changes for an already-running home, it sends the literal-content reread pointer described in [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md); unchanged allowlisted config sends no pointer unless a previous delivery is pending. The locked bootstrap inheritance pass uses the same per-home changed-set and reread path for already-running homes; see `secondmate-provisioning` for the single contract owner. That live discovery starts from `state/*.meta` records with `kind=secondmate`; `data/secondmates.md` only backfills `home=` for older or incomplete meta records. diff --git a/tests/fm-backend-herdr-presentation-e2e.test.sh b/tests/fm-backend-herdr-presentation-e2e.test.sh index fb44305e99e..158e943211c 100755 --- a/tests/fm-backend-herdr-presentation-e2e.test.sh +++ b/tests/fm-backend-herdr-presentation-e2e.test.sh @@ -866,7 +866,7 @@ touch "$SECOND_HOME_A/state/.last-watcher-beat" "$SECOND_HOME_B/state/.last-watc # may write config/herdr-presentation-spaces. git -C "$SECOND_HOME_A" init -q git -C "$SECOND_HOME_B" init -q -printf 'config/herdr-presentation-spaces\nconfig/crew-harness\nconfig/crew-dispatch.json\nconfig/backlog-backend\nconfig/backend\nconfig/startup-memory-budget\n' \ +printf 'config/herdr-presentation-spaces\nconfig/crew-harness\nconfig/crew-dispatch.json\nconfig/backlog-backend\nconfig/backend\n' \ > "$SECOND_HOME_A/.gitignore" cp "$SECOND_HOME_A/.gitignore" "$SECOND_HOME_B/.gitignore" git -C "$SECOND_HOME_A" add .gitignore diff --git a/tests/fm-secondmate-harness.test.sh b/tests/fm-secondmate-harness.test.sh index 39ca2021bf1..b9efe54033d 100755 --- a/tests/fm-secondmate-harness.test.sh +++ b/tests/fm-secondmate-harness.test.sh @@ -14,11 +14,12 @@ # explicit per-spawn harness arg still wins. # B) Inheritance. The primary pushes a declared, extensible set of LOCAL # (gitignored) config items - config/crew-dispatch.json, config/crew-harness, -# config/backlog-backend, and config/herdr-presentation-spaces - down into -# each secondmate home's config/, so the secondmate's OWN crewmates, -# dispatch profiles, backlog backend, and Herdr presentation opt-in inherit -# the primary's settings. It is primary-authoritative (re-pushed at -# secondmate spawn, on the bootstrap secondmate sweep, and by config push). +# config/backlog-backend, config/backend, and config/herdr-presentation-spaces - +# down into each secondmate home's config/, so the secondmate's OWN crewmates, +# dispatch profiles, backlog backend, runtime-backend default, and Herdr +# presentation opt-in inherit the primary's settings. It is primary-authoritative +# (re-pushed at secondmate spawn, on the bootstrap secondmate sweep, and by +# config push). # config/secondmate-harness is deliberately NOT inherited (secondmates do # not spawn secondmates). After a successful push that changes allowlisted # config under an already-running home, a literal-content reread instruction @@ -191,16 +192,18 @@ SH # B) propagate_inheritable_config unit behavior # =========================================================================== test_propagate_lib() { - local d src dest m1 m2 outside stdout stderr guard_repo err_text + local d src dest home m1 m2 outside stdout stderr guard_repo err_text d="$TMP_ROOT/prop-lib" src="$d/src" - dest="$d/dest" - mkdir -p "$src" "$dest" + home="$d/home1" + dest="$home/config" + mkdir -p "$src" "$dest" "$home/state" # 1. present source is copied printf '{"default":{"harness":"codex"}}\n' > "$src/crew-dispatch.json" printf 'codex\n' > "$src/crew-harness" printf 'manual\n' > "$src/backlog-backend" + printf 'tmux\n' > "$src/backend" : > "$src/herdr-presentation-spaces" stdout="$d/clean-copy.out" stderr="$d/clean-copy.err" @@ -210,7 +213,11 @@ test_propagate_lib() { [ "$(cat "$dest/crew-dispatch.json")" = '{"default":{"harness":"codex"}}' ] || fail "crew-dispatch.json not propagated" [ "$(cat "$dest/crew-harness")" = codex ] || fail "crew-harness not propagated" [ "$(cat "$dest/backlog-backend")" = manual ] || fail "backlog-backend not propagated" + [ "$(cat "$dest/backend")" = tmux ] || fail "backend not propagated" [ -f "$dest/herdr-presentation-spaces" ] || fail "herdr-presentation-spaces not propagated" + printf 'herdr\n' > "$dest/backend" + propagate_inheritable_config "$src" "$dest" + [ "$(cat "$dest/backend")" = tmux ] || fail "primary backend did not overwrite a divergent destination" # 2. idempotent: an unchanged re-run does not churn the mtime m1=$(date -r "$dest/crew-harness" +%s 2>/dev/null || stat -c %Y "$dest/crew-harness") @@ -227,10 +234,12 @@ test_propagate_lib() { printf '{"default":{"harness":"claude"}}\n' > "$src/crew-dispatch.json" printf 'claude\n' > "$src/crew-harness" printf 'tasks-axi\n' > "$src/backlog-backend" + printf 'zellij\n' > "$src/backend" propagate_inheritable_config "$src" "$dest" [ "$(cat "$dest/crew-dispatch.json")" = '{"default":{"harness":"claude"}}' ] || fail "changed dispatch profile did not converge" [ "$(cat "$dest/crew-harness")" = claude ] || fail "changed value did not converge" [ "$(cat "$dest/backlog-backend")" = tasks-axi ] || fail "changed backlog backend did not converge" + [ "$(cat "$dest/backend")" = zellij ] || fail "changed backend did not converge" outside="$d/outside-target" rm -f "$dest/crew-harness" "$outside" @@ -243,11 +252,14 @@ test_propagate_lib() { [ "$(cat "$outside")" = outside ] || fail "destination symlink target was overwritten" # 4. removing the source mirrors absence downstream (primary-authoritative) - rm -f "$src/crew-dispatch.json" "$src/crew-harness" "$src/backlog-backend" "$src/herdr-presentation-spaces" + printf 'herdr\n' > "$dest/backend" + rm -f "$src/crew-dispatch.json" "$src/crew-harness" "$src/backlog-backend" \ + "$src/backend" "$src/herdr-presentation-spaces" propagate_inheritable_config "$src" "$dest" [ -e "$dest/crew-dispatch.json" ] && fail "dispatch profile absence not mirrored downstream" [ -e "$dest/crew-harness" ] && fail "absence not mirrored downstream" [ -e "$dest/backlog-backend" ] && fail "backlog-backend absence not mirrored downstream" + [ -e "$dest/backend" ] && fail "backend absence not mirrored downstream" [ -e "$dest/herdr-presentation-spaces" ] && fail "herdr-presentation-spaces absence not mirrored downstream" rm -f "$dest/crew-harness" @@ -265,22 +277,25 @@ test_propagate_lib() { [ -d "$dest/crew-harness" ] || fail "failed absence mirror removed the wrong path" rm -rf "$dest/crew-harness" - # 5. secondmate-harness is never inherited + # 5. secondmate-harness is never inherited; backend still is printf 'grok\n' > "$src/secondmate-harness" printf '{"default":{"harness":"codex"}}\n' > "$src/crew-dispatch.json" printf 'codex\n' > "$src/crew-harness" printf 'manual\n' > "$src/backlog-backend" - rm -rf "$d/dest2" - mkdir -p "$d/dest2" - propagate_inheritable_config "$src" "$d/dest2" - [ -e "$d/dest2/secondmate-harness" ] && fail "secondmate-harness was inherited (must not be)" - [ "$(cat "$d/dest2/crew-dispatch.json")" = '{"default":{"harness":"codex"}}' ] || fail "crew-dispatch.json not propagated alongside" - [ "$(cat "$d/dest2/crew-harness")" = codex ] || fail "crew-harness not propagated alongside" - [ "$(cat "$d/dest2/backlog-backend")" = manual ] || fail "backlog-backend not propagated alongside" + printf 'herdr\n' > "$src/backend" + rm -rf "$d/home2" + mkdir -p "$d/home2/config" "$d/home2/state" + propagate_inheritable_config "$src" "$d/home2/config" + [ -e "$d/home2/config/secondmate-harness" ] && fail "secondmate-harness was inherited (must not be)" + [ "$(cat "$d/home2/config/crew-dispatch.json")" = '{"default":{"harness":"codex"}}' ] || fail "crew-dispatch.json not propagated alongside" + [ "$(cat "$d/home2/config/crew-harness")" = codex ] || fail "crew-harness not propagated alongside" + [ "$(cat "$d/home2/config/backlog-backend")" = manual ] || fail "backlog-backend not propagated alongside" + [ "$(cat "$d/home2/config/backend")" = herdr ] || fail "backend not propagated alongside" # 6. nothing to propagate -> destination dir is never created (a true no-op) rm -rf "$d/src3" "$d/dest3" mkdir -p "$d/src3" + # Keep backend out of the empty-source case by clearing it from src3 only. propagate_inheritable_config "$d/src3" "$d/dest3/config" [ -e "$d/dest3/config" ] && fail "empty-source propagation created a destination dir" @@ -371,6 +386,7 @@ test_spawn_split_and_inherit() { printf 'claude\n' > "$w/home/config/crew-harness" printf 'codex\n' > "$w/home/config/secondmate-harness" printf 'manual\n' > "$w/home/config/backlog-backend" + printf 'zellij\n' > "$w/home/config/backend" make_seeded_home "$sm" sm spawn_secondmate "$w" sm "$sm" @@ -385,6 +401,8 @@ test_spawn_split_and_inherit() { || fail "split: home crew-dispatch.json not inherited" [ "$(cat "$sm/config/backlog-backend" 2>/dev/null)" = manual ] \ || fail "split: home backlog-backend not inherited as manual" + [ "$(cat "$sm/config/backend" 2>/dev/null)" = zellij ] \ + || fail "split: home backend not inherited as zellij" [ -e "$sm/config/secondmate-harness" ] \ && fail "split: secondmate-harness leaked into the secondmate home" pass "B2 spawn: secondmate runs the secondmate harness; its home inherits declared config" @@ -537,6 +555,50 @@ spawn_secondmate_capture() { "$ROOT/bin/fm-spawn.sh" "$id" "$home" "$@" --secondmate } +test_spawn_backend_precedence_over_inherited_config() { + local w sm meta launchlog out status + w="$TMP_ROOT/spawn-backend-env-precedence" + sm="$w/sm" + launchlog="$w/launch.log" + mkdir -p "$w/home/config" + printf 'herdr\n' > "$w/home/config/backend" + make_seeded_home "$sm" sm + + out=$(FM_BACKEND=tmux spawn_secondmate_capture \ + "$w" sm "$sm" "$launchlog" 2>&1); status=$? + expect_code 0 "$status" \ + "FM_BACKEND=tmux should beat inherited config/backend=herdr"$'\n'"$out" + + meta="$w/home/state/sm.meta" + [ "$(cat "$sm/config/backend")" = herdr ] \ + || fail "backend precedence fixture did not inherit config/backend=herdr" + assert_no_grep '^backend=' "$meta" \ + "FM_BACKEND=tmux did not beat inherited config/backend=herdr" + pass "B5b spawn: FM_BACKEND wins over inherited config/backend" +} + +test_spawn_explicit_backend_precedence_over_env_and_inherited_config() { + local w sm meta launchlog out status + w="$TMP_ROOT/spawn-backend-flag-precedence" + sm="$w/sm" + launchlog="$w/launch.log" + mkdir -p "$w/home/config" + printf 'herdr\n' > "$w/home/config/backend" + make_seeded_home "$sm" sm + + out=$(FM_BACKEND=zellij spawn_secondmate_capture \ + "$w" sm "$sm" "$launchlog" --backend tmux 2>&1); status=$? + expect_code 0 "$status" \ + "explicit --backend tmux should beat FM_BACKEND=zellij and inherited config/backend=herdr"$'\n'"$out" + + meta="$w/home/state/sm.meta" + [ "$(cat "$sm/config/backend")" = herdr ] \ + || fail "explicit backend precedence fixture did not inherit config/backend=herdr" + assert_no_grep '^backend=' "$meta" \ + "explicit --backend tmux did not beat FM_BACKEND=zellij and inherited config/backend=herdr" + pass "B5c spawn: explicit --backend wins over FM_BACKEND and inherited config/backend" +} + # A bare "<harness>" secondmate-harness file (today's format) must launch with # NO --model/--effort flag at all, and meta must keep recording model=default, # effort=default - the core backward-compat requirement of the new format. @@ -770,6 +832,7 @@ new_world() { printf 'projects/\nstate/\ndata/\n.no-mistakes/\n' [ "$dispatch_ignore" = no ] || printf 'config/crew-dispatch.json\n' printf 'config/crew-harness\nconfig/secondmate-harness\nconfig/backlog-backend\n' + printf 'config/backend\nconfig/herdr-presentation-spaces\n' } > "$w/main/.gitignore" printf 'v1\n' > "$w/main/AGENTS.md" printf 'r1\n' > "$w/main/README.md" @@ -965,6 +1028,7 @@ test_bootstrap_sweep_propagates_and_reconverges() { printf '{"default":{"harness":"codex"}}\n' > "$w/home/config/crew-dispatch.json" printf 'codex\n' > "$w/home/config/crew-harness" printf 'manual\n' > "$w/home/config/backlog-backend" + printf 'tmux\n' > "$w/home/config/backend" printf 'grok\n' > "$w/home/config/secondmate-harness" run_bootstrap "$w" >/dev/null [ "$(cat "$w/sm/config/crew-harness" 2>/dev/null)" = codex ] \ @@ -973,6 +1037,8 @@ test_bootstrap_sweep_propagates_and_reconverges() { || fail "sweep: crew-dispatch.json not pushed into the live home" [ "$(cat "$w/sm/config/backlog-backend" 2>/dev/null)" = manual ] \ || fail "sweep: backlog-backend not pushed into the live home" + [ "$(cat "$w/sm/config/backend" 2>/dev/null)" = tmux ] \ + || fail "sweep: backend not pushed into the live home" [ -e "$w/sm/config/secondmate-harness" ] \ && fail "sweep: secondmate-harness was inherited (must not be)" @@ -980,6 +1046,7 @@ test_bootstrap_sweep_propagates_and_reconverges() { printf '{"default":{"harness":"claude"}}\n' > "$w/home/config/crew-dispatch.json" printf 'claude\n' > "$w/home/config/crew-harness" printf 'tasks-axi\n' > "$w/home/config/backlog-backend" + printf 'zellij\n' > "$w/home/config/backend" run_bootstrap "$w" >/dev/null [ "$(cat "$w/sm/config/crew-harness" 2>/dev/null)" = claude ] \ || fail "sweep: home did not re-converge to the primary's new crew-harness" @@ -987,9 +1054,12 @@ test_bootstrap_sweep_propagates_and_reconverges() { || fail "sweep: home did not re-converge to the primary's new crew-dispatch.json" [ "$(cat "$w/sm/config/backlog-backend" 2>/dev/null)" = tasks-axi ] \ || fail "sweep: home did not re-converge to the primary's new backlog-backend" + [ "$(cat "$w/sm/config/backend" 2>/dev/null)" = zellij ] \ + || fail "sweep: home did not re-converge to the primary's new backend" # Mirror absence: primary clears inherited config; the home's copies are removed. - rm -f "$w/home/config/crew-dispatch.json" "$w/home/config/crew-harness" "$w/home/config/backlog-backend" + rm -f "$w/home/config/crew-dispatch.json" "$w/home/config/crew-harness" \ + "$w/home/config/backlog-backend" "$w/home/config/backend" run_bootstrap "$w" >/dev/null [ -e "$w/sm/config/crew-dispatch.json" ] \ && fail "sweep: home crew-dispatch.json not removed after the primary cleared it" @@ -997,6 +1067,8 @@ test_bootstrap_sweep_propagates_and_reconverges() { && fail "sweep: home crew-harness not removed after the primary cleared it" [ -e "$w/sm/config/backlog-backend" ] \ && fail "sweep: home backlog-backend not removed after the primary cleared it" + [ -e "$w/sm/config/backend" ] \ + && fail "sweep: home backend not removed after the primary cleared it" pass "B7 bootstrap sweep pushes, re-converges, and mirrors absence; never inherits secondmate-harness" } @@ -1011,6 +1083,7 @@ test_bootstrap_sweep_propagates_when_tracked_current() { printf '{"default":{"harness":"codex"}}\n' > "$w/home/config/crew-dispatch.json" printf 'codex\n' > "$w/home/config/crew-harness" printf 'manual\n' > "$w/home/config/backlog-backend" + printf 'tmux\n' > "$w/home/config/backend" run_bootstrap "$w" >/dev/null [ "$(cat "$w/sm/config/crew-dispatch.json" 2>/dev/null)" = '{"default":{"harness":"codex"}}' ] \ || fail "crew-dispatch.json did not propagate to a tracked-current home" @@ -1018,6 +1091,8 @@ test_bootstrap_sweep_propagates_when_tracked_current() { || fail "config did not propagate to a tracked-current home" [ "$(cat "$w/sm/config/backlog-backend" 2>/dev/null)" = manual ] \ || fail "backlog-backend did not propagate to a tracked-current home" + [ "$(cat "$w/sm/config/backend" 2>/dev/null)" = tmux ] \ + || fail "backend did not propagate to a tracked-current home" pass "B8 bootstrap sweep propagates config even when the home's tracked files are already current" } @@ -1069,12 +1144,49 @@ test_bootstrap_sweep_no_inheritance_is_noop() { [ -e "$w/sm/config/crew-dispatch.json" ] && fail "no-inheritance sweep created a home crew-dispatch.json" [ -e "$w/sm/config/crew-harness" ] && fail "no-inheritance sweep created a home crew-harness" + [ -e "$w/sm/config/backend" ] && fail "no-inheritance sweep created a home backend" [ -e "$w/sm/config" ] && fail "no-inheritance sweep created a home config/ dir" [ "$(git -C "$w/sm" rev-parse HEAD)" = "$head" ] \ || fail "no-inheritance sweep did not still fast-forward the tracked files" pass "B10 bootstrap sweep with no inherited config is a config no-op and still fast-forwards" } +# config/backend: present and absent primary state converges exactly. +test_backend_inheritance_present_and_absent() { + local w head out err status instruction + w=$(new_world backend-inherit) + head=$(git -C "$w/main" rev-parse HEAD) + add_sm_worktree "$w" sm "$head" + + printf 'tmux\n' > "$w/home/config/backend" + err="$w/backend-inherit.err" + out=$(run_config_push "$w" 2>"$err"); status=$? + expect_code 0 "$status" "backend present push should succeed" + assert_contains "$out" "backend: pushed" "backend present value should report pushed" + [ "$(cat "$w/sm/config/backend")" = tmux ] || fail "backend present value not pushed" + instruction=$(reread_instruction_path "$w/sm") || fail "backend present reread instruction missing" + assert_contains "$(cat "$instruction")" $'-----BEGIN config/backend-----\ntmux\n-----END config/backend-----' \ + "backend present reread must include exact bytes" + + printf 'herdr\n' > "$w/sm/config/backend" + printf 'zellij\n' > "$w/home/config/backend" + out=$(run_config_push "$w" 2>"$err"); status=$? + expect_code 0 "$status" "backend changed push should succeed" + assert_contains "$out" "backend: pushed" "backend changed value should report pushed" + [ "$(cat "$w/sm/config/backend")" = zellij ] \ + || fail "primary backend did not overwrite the divergent destination" + + rm -f "$w/home/config/backend" + out=$(run_config_push "$w" 2>"$err"); status=$? + expect_code 0 "$status" "backend absence push should succeed" + assert_contains "$out" "backend: pushed - mirrored primary absence" "backend should mirror primary absence" + [ -e "$w/sm/config/backend" ] && fail "backend not removed on primary absence" + instruction=$(reread_instruction_path "$w/sm") || fail "backend absence reread instruction missing" + assert_contains "$(cat "$instruction")" $'-----BEGIN config/backend-----\nABSENT\n-----END config/backend-----' \ + "backend absence reread must use ABSENT token" + pass "B12b backend inheritance: present values and primary absence converge exactly" +} + test_bootstrap_sweep_surfaces_config_propagation_failure() { local w c1 out fail_line w=$(new_world boot-prop-fail) @@ -1113,7 +1225,7 @@ test_bootstrap_rereads_after_partial_propagation() { } test_config_push_propagates_reports_without_ff_or_nudge() { - local w c1 sm_real old_head out err status out2 tmp log + local w c1 sm_real old_head out err status out2 tmp log instruction w=$(new_world config-push-basic) c1=$(git -C "$w/main" rev-parse HEAD) add_sm_worktree "$w" sm "$c1" @@ -1131,6 +1243,7 @@ test_config_push_propagates_reports_without_ff_or_nudge() { printf '{"default":{"harness":"codex"}}\n' > "$w/home/config/crew-dispatch.json" printf 'codex\n' > "$w/home/config/crew-harness" printf 'manual\n' > "$w/home/config/backlog-backend" + printf 'tmux\n' > "$w/home/config/backend" err="$w/config-push-basic.err" log="$w/config-push-basic.tmux.log" out=$(run_config_push "$w" "$log" 2>"$err"); status=$? @@ -1146,12 +1259,18 @@ test_config_push_propagates_reports_without_ff_or_nudge() { "config push did not report crew-harness as pushed" assert_contains "$out" "backlog-backend: pushed" \ "config push did not report backlog-backend as pushed" + assert_contains "$out" "backend: pushed" \ + "config push did not report backend as pushed" assert_contains "$out" "config-reread: sent" \ "config push with changed config must send a literal reread instruction" assert_not_contains "$out" "NUDGE_SECONDMATES" \ "config push must not use the AGENTS.md instruction-surface nudge channel" [ "$(git -C "$w/sm" rev-parse HEAD)" = "$old_head" ] \ || fail "config push fast-forwarded tracked files" + [ "$(cat "$w/sm/config/backend")" = tmux ] || fail "config push did not write backend" + instruction=$(reread_instruction_path "$w/sm") || fail "config-push reread instruction missing" + assert_contains "$(cat "$instruction")" $'-----BEGIN config/backend-----\ntmux\n-----END config/backend-----' \ + "config-push reread must include exact backend bytes" [ ! -s "$err" ] || fail "clean config push wrote unexpected stderr: $(cat "$err")" assert_contains "$(cat "$log")" "[fm-from-firstmate]" \ "config reread must use the marked routed secondmate path" @@ -1165,6 +1284,8 @@ test_config_push_propagates_reports_without_ff_or_nudge() { "idempotent config push did not report crew-harness as unchanged" assert_contains "$out2" "backlog-backend: unchanged" \ "idempotent config push did not report backlog-backend as unchanged" + assert_contains "$out2" "backend: unchanged" \ + "idempotent config push did not report backend as unchanged" assert_not_contains "$out2" "config-reread: sent" \ "unchanged config must not send a reread message" [ ! -s "$log" ] || fail "unchanged config push still invoked tmux send: $(cat "$log")" @@ -1301,6 +1422,7 @@ test_config_reread_per_home_changed_sets_and_exact_bytes() { printf '%s' "$multiline_json" > "$w/home/config/crew-dispatch.json" printf 'codex\n' > "$w/home/config/crew-harness" printf 'manual\n' > "$w/home/config/backlog-backend" + printf 'tmux\n' > "$w/home/config/backend" { shared_captain_header_for_tests printf '%s\n' "shared secret preference body that must never appear in a config reread" @@ -1319,6 +1441,7 @@ test_config_reread_per_home_changed_sets_and_exact_bytes() { || fail "beta did not receive multiline dispatch" [ "$(cat "$w/alpha/config/crew-harness")" = codex ] || fail "alpha harness not updated" [ "$(cat "$w/alpha/config/backlog-backend")" = manual ] || fail "alpha backlog-backend not updated" + [ "$(cat "$w/alpha/config/backend")" = tmux ] || fail "alpha backend not updated" instr_a=$(reread_instruction_path "$w/alpha") || fail "alpha instruction missing after config push" instr_b=$(reread_instruction_path "$w/beta") || fail "beta instruction missing after config push" @@ -1328,19 +1451,21 @@ test_config_reread_per_home_changed_sets_and_exact_bytes() { [ "$(reread_mode "$instr_b")" = 600 ] || fail "beta instruction is not private" # Deterministic allowlist path order and exact destination bytes for alpha - # (all three config items were missing/stale and therefore pushed). + # (allowlisted config items were missing/stale and therefore pushed). assert_grep "These inherited config files changed" "$instr_a" "alpha framing missing" assert_grep "defaults/rules" "$instr_a" "alpha must preserve agent judgment framing" assert_contains "$(cat "$instr_a")" "config/crew-dispatch.json" "alpha missing dispatch path" assert_contains "$(cat "$instr_a")" "config/crew-harness" "alpha missing harness path" assert_contains "$(cat "$instr_a")" "config/backlog-backend" "alpha missing backlog path" + assert_contains "$(cat "$instr_a")" "config/backend" "alpha missing backend path" # Path order follows FM_INHERITABLE_CONFIG. awk ' /config\/crew-dispatch\.json/ { d=NR } /config\/crew-harness/ { h=NR } /config\/backlog-backend/ { b=NR } + /config\/backend/ && !/backlog-backend/ { k=NR } END { - if (!(d && h && b && d < h && h < b)) exit 1 + if (!(d && h && b && k && d < h && h < b && b < k)) exit 1 } ' "$instr_a" || fail "alpha instruction path order is not deterministic allowlist order" @@ -1351,6 +1476,8 @@ test_config_reread_per_home_changed_sets_and_exact_bytes() { "alpha instruction must include exact harness scalar bytes" assert_contains "$(cat "$instr_a")" $'-----BEGIN config/backlog-backend-----\nmanual\n-----END config/backlog-backend-----' \ "alpha instruction must include exact backlog-backend scalar bytes" + assert_contains "$(cat "$instr_a")" $'-----BEGIN config/backend-----\ntmux\n-----END config/backend-----' \ + "alpha instruction must include exact backend scalar bytes" # No parsed/effective summary, no SHA, no captain-shared dump. assert_not_contains "$(cat "$instr_a")" "Default worker" "must not emit parsed worker summary" @@ -1434,6 +1561,7 @@ test_config_reread_isolation_and_absent_and_send_failure() { printf '%s\n' $'crew-dispatch.json\tpushed\tmirrored primary absence' printf '%s\n' $'crew-harness\tunchanged\t' printf '%s\n' $'backlog-backend\tunchanged\t' + printf '%s\n' $'backend\tunchanged\t' printf '%s\n' $'data/captain-shared.md\tpushed\t' } > "$report" rm -f "$w/beta/config/crew-dispatch.json" @@ -2124,6 +2252,8 @@ test_spawn_backward_compat_crew_fallback test_spawn_bare_backward_compat test_spawn_explicit_harness_wins test_spawn_unverified_secondmate_harness_refused +test_spawn_backend_precedence_over_inherited_config +test_spawn_explicit_backend_precedence_over_env_and_inherited_config test_spawn_bare_harness_no_model_effort_flag test_spawn_secondmate_harness_model_token test_spawn_secondmate_harness_model_and_effort_tokens @@ -2136,6 +2266,7 @@ test_bootstrap_sweep_propagates_and_reconverges test_bootstrap_sweep_propagates_when_tracked_current test_bootstrap_sweep_defers_dispatch_on_stale_unignored_home test_bootstrap_sweep_no_inheritance_is_noop +test_backend_inheritance_present_and_absent test_bootstrap_sweep_surfaces_config_propagation_failure test_bootstrap_rereads_after_partial_propagation test_config_push_propagates_reports_without_ff_or_nudge From 2ff421487dcf0988ca59f9fbd6c37260d08554b4 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Tue, 28 Jul 2026 18:31:23 -0700 Subject: [PATCH 26/52] fix(pi): remove Calm's upper version ceiling (#1226) * fix(pi): remove Calm's exclusive Pi upper-version ceiling tests/fm-calm-pi-extension.test.sh gated on a closed PI_COMPAT_VERSIONS allowlist ("0.81.1 0.82.0") that refused any other installed Pi, and docs described that range as "supported" rather than verified evidence. The Calm CHANGELOG shows no API introduced at either version, so there is no evidence for a real minimum; the presentation adapters already probe the exact method they patch rather than checking a version. Replace the allowlist with dated version evidence that never rejects a newer Pi, and make each presentation adapter degrade independently with a diagnostic if a future Pi removes its API, instead of the whole Calm extension failing to load. Rewrite the feasibility doc's "Pi 0.81.1 through 0.82.0" phrasing to state it as verified evidence, not a ceiling. * no-mistakes(review): Probe missing Calm adapter exports safely * no-mistakes(document): Document Calm's unbounded Pi compatibility --- .pi/extensions/fm-calm.ts | 25 ++- .../lib/fm-calm-operational-user-layout.ts | 25 ++- docs/calm-mode-feasibility.md | 28 ++- docs/calm.md | 6 + tests/fm-calm-pi-extension.test.sh | 202 ++++++++++++++++-- 5 files changed, 249 insertions(+), 37 deletions(-) diff --git a/.pi/extensions/fm-calm.ts b/.pi/extensions/fm-calm.ts index eb009fd8e3b..f78c1b5acd9 100644 --- a/.pi/extensions/fm-calm.ts +++ b/.pi/extensions/fm-calm.ts @@ -1,11 +1,13 @@ // Firstmate's home-persistent Pi transcript presentation toggle. // -// Compatibility boundary: Pi 0.81.1 and 0.82.0 expose built-in ToolDefinitions, per-slot +// Verified against Pi 0.81.1 and 0.82.0, which expose built-in ToolDefinitions, per-slot // renderers, renderShell: "self", session_start replacement reasons, // ExtensionUIContext.setToolsExpanded(), setWorkingVisible(), and -// setHiddenThinkingLabel(). The focused tests pin those assumptions. Version-bounded -// presentation adapters cover collapsed assistant thinking and operational user rows; -// Pi still exposes no global renderer for arbitrary built-in or custom rows. +// setHiddenThinkingLabel(). The focused tests pin those assumptions but never reject a +// newer Pi solely for its version. The collapsed-thinking and operational-user +// presentation adapters probe the exact API they patch and degrade independently with a +// diagnostic (see installCalmPresentationAdapter below) if a future Pi removes it; Pi +// still exposes no global renderer for arbitrary built-in or custom rows. // docs/configuration.md owns the home-local Calm preference contract. import { randomUUID } from "node:crypto"; import { @@ -74,9 +76,20 @@ const extensionFile = fileURLToPath(import.meta.url); const extensionDir = dirname(extensionFile); const root = resolve(extensionDir, "../.."); +// Each presentation adapter probes the exact Pi API it patches. If a future Pi removes +// that API, only the affected adapter degrades; the rest of Calm keeps working. +function installCalmPresentationAdapter(name: string, install: () => void): void { + try { + install(); + } catch (error) { + const reason = error instanceof Error ? error.message : String(error); + console.error(`Firstmate Calm: ${name} presentation adapter unavailable, skipping. ${reason}`); + } +} + export default function (pi: ExtensionAPI) { - installCalmAssistantLayout(); - installCalmOperationalUserLayout(); + installCalmPresentationAdapter("collapsed-thinking", installCalmAssistantLayout); + installCalmPresentationAdapter("operational-user-row", installCalmOperationalUserLayout); let exportRendering = false; let removeTerminalInputHandler: (() => void) | undefined; diff --git a/.pi/extensions/lib/fm-calm-operational-user-layout.ts b/.pi/extensions/lib/fm-calm-operational-user-layout.ts index 82c69eda01f..ca9b0bbcc0a 100644 --- a/.pi/extensions/lib/fm-calm-operational-user-layout.ts +++ b/.pi/extensions/lib/fm-calm-operational-user-layout.ts @@ -1,13 +1,14 @@ -// Pi 0.81.1 and 0.82.0 add the ordinary-user spacer and row together. -// This version-bounded adapter changes only that presentation and never message delivery. -import { - InteractiveMode, - UserMessageComponent, -} from "@earendil-works/pi-coding-agent"; +// Verified against Pi 0.81.1 and 0.82.0, which add the ordinary-user spacer and row +// together via InteractiveMode.addMessageToChat. This adapter probes that exact method +// and throws if it is missing; fm-calm.ts catches that and skips only this adapter with a +// diagnostic instead of blocking Calm or Pi. It changes only that presentation and never +// message delivery. +import type { UserMessageComponent as PiUserMessageComponent } from "@earendil-works/pi-coding-agent"; +import * as PiCodingAgent from "@earendil-works/pi-coding-agent"; import { calmPresentationHides } from "./fm-calm-visibility.ts"; import { classifyFirstmateCurrentOperationalText } from "./fm-operational-input.ts"; -type UserMessageConstructorArgs = ConstructorParameters<typeof UserMessageComponent>; +type UserMessageConstructorArgs = ConstructorParameters<typeof PiUserMessageComponent>; type UserMessageLike = { role: string; content: unknown; @@ -18,7 +19,7 @@ type AddMessageOptions = { type InteractiveModePresentation = { chatContainer: { children: unknown[]; - addChild(component: UserMessageComponent): void; + addChild(component: PiUserMessageComponent): void; }; editor: { addToHistory?(text: string): void; @@ -81,12 +82,20 @@ export function installCalmOperationalUserLayout(): void { hidesOperationalInput, isOperationalInput, }; + const InteractiveMode = PiCodingAgent.InteractiveMode; + if (typeof InteractiveMode !== "function") { + throw new Error("Firstmate Calm requires Pi InteractiveMode"); + } const prototype = InteractiveMode.prototype as unknown as InteractiveModePrototype; const originalAddMessageToChat = prototype.addMessageToChat; if (typeof originalAddMessageToChat !== "function") { throw new Error("Firstmate Calm requires Pi InteractiveMode.addMessageToChat"); } + const UserMessageComponent = PiCodingAgent.UserMessageComponent; + if (typeof UserMessageComponent !== "function") { + throw new Error("Firstmate Calm requires Pi UserMessageComponent"); + } class CalmOperationalUserMessageComponent extends UserMessageComponent { private readonly hasLeadingSpacer: boolean; diff --git a/docs/calm-mode-feasibility.md b/docs/calm-mode-feasibility.md index c4a051b9cc6..b94b6a6aef9 100644 --- a/docs/calm-mode-feasibility.md +++ b/docs/calm-mode-feasibility.md @@ -9,9 +9,17 @@ A qualifying implementation must auto-load from the trusted project, persist the The governing presentation policy allows genuine original user prompts, genuine user-facing assistant text, and Pi's native working activity. Changing persisted context to remove hidden content, filtering provider context, patching installed harness code, or claiming coverage outside a supported renderer does not satisfy that boundary. +## Compatibility evidence + +[`calm.md`](calm.md#pi-compatibility) owns the current Pi compatibility contract. +Pi 0.81.1 was installed when Calm was first built, and Pi 0.82.0 was the later reverification target. +The inspected Pi CHANGELOG shows no relevant presentation API introduced at either version, so those versions remain verification evidence rather than compatibility bounds. +The exported classes used by the adapters (`AssistantMessageComponent` and `InteractiveMode`) are undocumented internals with no stated version guarantee. +`tests/fm-calm-pi-extension.test.sh` records the installed Pi version as evidence without gating on it and covers both newer synthetic versions and an unavailable adapter seam. + ## Pi 0.81.1 end-to-end reproduction -The current installed and regression-supported Pi version was verified on 2026-07-22. +The Pi version installed at the time was verified on 2026-07-22. ```text $ pi --version @@ -57,7 +65,8 @@ The single-thinking, tool-call-only, tool-result, Calm-off, and `clearOnShrink` PR 927 made Calm persistent and described controlled rows as gapless while retaining a documented unsupported boundary for collapsed-thinking spacing. PR 936 removed the unsafe operational-input reroute and preserved legacy zero-height entries but did not change assistant-message layout. -The fix installs one idempotent Pi 0.81.1 through 0.82.0 presentation adapter on the exported `AssistantMessageComponent.updateContent` method. +The fix installs one idempotent presentation adapter, verified on Pi 0.81.1 through 0.82.0, on the exported `AssistantMessageComponent.updateContent` method. +The adapter probes for that exact method and, per the [compatibility contract](calm.md#pi-compatibility), degrades independently with a diagnostic rather than gating on a version number. Only while Calm is active and Pi has collapsed thinking does the adapter pass a shallow thinking-free presentation copy into Pi's ordinary layout calculation, then retain the original message on the component for invalidation and thinking expansion. The persisted assistant message, provider context, tool execution, export data, and expansion history remain unchanged. Collapsed thinking-only assistant messages now render zero rows, thinking before visible assistant text adds no spacing beyond the text-only baseline, and expanding thinking still renders the original reasoning. @@ -114,7 +123,8 @@ The real Pi viewport moved the unchanged assistant text from row 7 to row 2, ren The leading cause would have been falsified if the row or height remained, the provider lost or duplicated the message, or the persisted role or bytes changed. None occurred. -The fix installs a separate idempotent Pi 0.81.1 through 0.82.0 presentation adapter on the exported `InteractiveMode.addMessageToChat` method. +The fix installs a separate idempotent presentation adapter, verified on Pi 0.81.1 through 0.82.0, on the exported `InteractiveMode.addMessageToChat` method. +The adapter probes for that exact method and, per the [compatibility contract](calm.md#pi-compatibility), degrades independently with a diagnostic rather than gating on a version number. It delegates current recognition to `bin/fm-operational-input.sh`, adds only the evidence-backed bare-U+2063 `Supervisor escalate (` presentation compatibility shape, mounts a `UserMessageComponent` subclass that preserves Pi's stock row plus leading spacer while Calm is off, and returns zero rendered lines while Calm is on. It never intercepts the input event, rewrites the message, changes its role, filters model context, or changes session data. Messages containing an image are left on Pi's ordinary path even when their text equals an operational envelope because Firstmate's authoritative producers are text-only. @@ -148,7 +158,7 @@ Serialized session data and Pi 0.81.1's sidebar tree also retain legacy hidden o The taxonomy was derived from Pi 0.81.1's installed public declarations, documentation, examples, `interactive-mode.js`, and its exported component implementations. The test fixture enumerates every class below through the centralized policy, and the interactive fixture exercises the screenshot classes, current user-role operational input, and legacy synthetic presentation entries. -| Policy class | Pi transcript path | Calm result on Pi 0.81.1 through 0.82.0 | +| Policy class | Pi transcript path | Calm result (verified on Pi 0.81.1 through 0.82.0) | | --- | --- | --- | | `genuine-user-prompt` | `UserMessageComponent` | Visible, including every tested operational near miss. | | `genuine-agent-response` | Assistant text in `AssistantMessageComponent` | Visible. | @@ -167,12 +177,12 @@ The test fixture enumerates every class below through the centralized policy, an | `system-notice` | `showStatus`, `showError`, compaction, retry, and startup warning rows | Unsupported boundary; remains visible. | | `cache-notice` | Non-persisted cache-miss `Text` row | Unsupported boundary; remains visible. | | `project-trust-warning` | Non-persisted startup `Text` row | Unsupported boundary; remains visible. | -| `synthetic-user` | Firstmate extension `sendUserMessage`, terminal-injected input, Firstmate-generated Pi positional brief, or the already non-displayed session-start nudge | Canonically classified text-only operational user messages stay ordinary semantic user messages but render through the zero-height Pi 0.81.1 through 0.82.0 adapter under Calm; legacy entries stay gaplessly controllable, and the session-start nudge retains its existing non-displayed custom-message path. | +| `synthetic-user` | Firstmate extension `sendUserMessage`, terminal-injected input, Firstmate-generated Pi positional brief, or the already non-displayed session-start nudge | Canonically classified text-only operational user messages stay ordinary semantic user messages but render through the zero-height adapter (verified on Pi 0.81.1 through 0.82.0) under Calm; legacy entries stay gaplessly controllable, and the session-start nudge retains its existing non-displayed custom-message path. | | `synthetic-assistant` | No authoritative Firstmate source found | Policy-hidden, but Pi exposes no generic assistant-role renderer. | | `unknown` | Future or unclassified transcript component | Policy-hidden, but no generic renderer exists; never claimed as covered. | The installed extension API has no supported global transcript filter, user-message renderer, assistant-message renderer, chat-container API, or generic custom-tool wrapper. -Pi 0.81.1 through 0.82.0 export `AssistantMessageComponent` and `InteractiveMode`, so Calm uses separate version-bounded, idempotent adapters for assistant thinking layout and the complete operational-user transcript row while leaving all message data and non-Calm rendering unchanged. +Pi 0.81.1 through 0.82.0 export `AssistantMessageComponent` and `InteractiveMode`, so Calm uses separate idempotent, API-probed adapters for assistant thinking layout and the complete operational-user transcript row while leaving all message data and non-Calm rendering unchanged; see the [compatibility contract](calm.md#pi-compatibility) for how a future Pi lacking one of those exports is handled. General component replacement, ANSI cursor erasure, provider-context mutation, and installed-file patching remain rejected as unsupported or preservation-breaking workarounds. ## Cross-harness verification record @@ -197,7 +207,7 @@ grok 0.2.106 (bde89716f679) | Claude Code 2.1.218 | Not feasible through the inspected supported project surface. | Project hooks can observe lifecycle and tool events, while the plugin CLI packages supported components; neither inspected surface exposes a transcript-row renderer or transcript-wide redraw API. | | Codex CLI 0.144.6 | Not feasible through the inspected supported project surface. | The tracked hooks expose session, pre-tool, and stop handling, while the plugin and feature inventories expose no TUI tool-row renderer or transcript redraw control. | | OpenCode 1.17.18 | Not feasible without violating the preservation boundary. | Plugins expose events and tool execution hooks, not a built-in transcript-row renderer; same-name tool replacement changes execution rather than presentation alone. | -| Pi 0.81.1 through 0.82.0 | Partially feasible with two version-bounded exported-class adapters. | Public APIs control working visibility, collapsed labels, known tool slots, custom entries, and expansion redraws; exported assistant and interactive-mode classes provide the version-pinned collapsed-thinking and operational-user layout boundaries, while generic user, tool, and status filtering remains unavailable. | +| Pi (verified 0.81.1 through 0.82.0) | Partially feasible with two API-probed exported-class adapters. | Public APIs control working visibility, collapsed labels, known tool slots, custom entries, and expansion redraws; exported assistant and interactive-mode classes provide the collapsed-thinking and operational-user layout boundaries, gated on the exact method's presence rather than a version number, while generic user, tool, and status filtering remains unavailable. | | Grok CLI 0.2.106 | Not feasible through the inspected supported project surface. | Project hooks expose lifecycle and tool interception, while the plugin CLI exposes no row-renderer contract; `--minimal` changes the whole screen mode rather than selected transcript rows. | These conclusions are deliberately limited to the named versions and supported surfaces. @@ -259,8 +269,8 @@ skip: set FM_PI_LIVE_E2E=1 to run the isolated interactive Pi regression ## 2026-07-26 Pi 0.82.0 compatibility verification -Pi 0.82.0 preserved both version-bounded presentation seams and every deterministic Calm TUI guarantee. -The globally installed declaration package remained 0.81.1, so the strict typecheck continued to cover that lower supported boundary while the real CLI exercised 0.82.0. +Pi 0.82.0 preserved both API-probed presentation seams and every deterministic Calm TUI guarantee. +The globally installed declaration package remained 0.81.1, so the strict typecheck continued to cover that earlier declaration-evidence version while the real CLI exercised 0.82.0. ```text $ pi --version diff --git a/docs/calm.md b/docs/calm.md index 6a2c1d14b9c..8d63b6d0b56 100644 --- a/docs/calm.md +++ b/docs/calm.md @@ -18,6 +18,12 @@ Pi's supported presentation API does not expose a global transcript filter. Expanded reasoning and its reserved spacing, built-in tool images, user-bash rows, skill and summary rows, generic status notices, and arbitrary custom-tool or extension rows remain visible. These are supported-API boundaries rather than hidden-content failures. +## Pi compatibility + +Calm has no numeric Pi version minimum or maximum and never refuses Pi solely because its version is newer than a previously verified version. +The collapsed-thinking and operational-user-row presentation adapters probe the exact Pi API seam they patch when Calm loads. +If Pi removes one of those seams, Calm logs a diagnostic naming the unavailable adapter and skips only that adapter; `/calm`, the other adapter, and unrelated Pi extensions remain available. + [`calm-mode-feasibility.md`](calm-mode-feasibility.md) owns the version-scoped renderer taxonomy and empirical evidence. [`configuration.md`](configuration.md#pi-calm-preference-configcalm) owns the persisted preference file and resolution rules. `.pi/extensions/lib/fm-calm-visibility.ts` owns the visibility policy, and `.pi/extensions/lib/fm-calm-operational-user-layout.ts` owns the zero-height operational-user row adapter. diff --git a/tests/fm-calm-pi-extension.test.sh b/tests/fm-calm-pi-extension.test.sh index b63661ad7a4..46b945e23f7 100755 --- a/tests/fm-calm-pi-extension.test.sh +++ b/tests/fm-calm-pi-extension.test.sh @@ -16,14 +16,15 @@ PI_OPERATIONAL_INPUT="$ROOT/.pi/extensions/lib/fm-operational-input.ts" PI_PACKAGE_DIR=${FM_PI_PACKAGE_DIR:-"$(npm root -g 2>/dev/null)/@earendil-works/pi-coding-agent"} TMUX_SOCKET="fm-calm-$$" TMUX_SESSION="fm-calm-e2e" -PI_COMPAT_VERSIONS="0.81.1 0.82.0" - -require_pi_compat_version() { +# Verified against Pi 0.81.1 and 0.82.0 (docs/calm-mode-feasibility.md). This is +# known-good evidence, not a support ceiling: the fixtures below run against whatever +# Pi is actually installed, and record_pi_version_evidence never rejects a newer +# version. The tracked presentation adapters probe the exact API they patch (see +# .pi/extensions/fm-calm.ts) instead of relying on version inference, so a version +# string is evidence for the record, not a gate. +record_pi_version_evidence() { local version=$1 context=$2 - case " $PI_COMPAT_VERSIONS " in - *" $version "*) return 0 ;; - *) fail "$context requires Pi $PI_COMPAT_VERSIONS, found $version" ;; - esac + [ -n "$version" ] || fail "$context could not determine the installed Pi version" } cleanup() { @@ -91,10 +92,13 @@ test_static_contract() { assert_contains "$text" 'ctx.ui.setWorkingVisible(true)' "Pi calm extension does not preserve Pi's live working row" assert_not_contains "$text" 'ctx.ui.setWorkingVisible(!active)' "Pi calm extension still hides Pi's live working row" assert_contains "$text" 'ctx.ui.setHiddenThinkingLabel(active ? "" : undefined)' "Pi calm extension does not hide collapsed thinking labels" - assert_contains "$text" 'installCalmAssistantLayout()' "Pi Calm extension does not install its zero-height assistant layout" - assert_contains "$text" 'installCalmOperationalUserLayout()' "Pi Calm extension does not install its operational-user layout" + assert_contains "$text" 'installCalmPresentationAdapter("collapsed-thinking", installCalmAssistantLayout)' "Pi Calm extension does not install its zero-height assistant layout" + assert_contains "$text" 'installCalmPresentationAdapter("operational-user-row", installCalmOperationalUserLayout)' "Pi Calm extension does not install its operational-user layout" + assert_contains "$text" 'function installCalmPresentationAdapter' "Pi Calm extension does not degrade a missing presentation adapter independently with a diagnostic" + assert_contains "$assistant_layout" 'import * as PiCodingAgent' "Pi Calm assistant layout still requires its optional runtime class as a named import" assert_contains "$assistant_layout" 'AssistantMessageComponent.prototype.updateContent' "Pi Calm assistant layout does not control the exported component presentation path" assert_contains "$assistant_layout" 'block.type !== "thinking"' "Pi Calm assistant layout does not remove thinking from its presentation copy" + assert_contains "$operational_user_layout" 'import * as PiCodingAgent' "Pi Calm operational-user layout still requires its optional runtime class as a named import" assert_contains "$operational_user_layout" 'InteractiveMode.prototype' "Pi Calm operational-user layout does not control the transcript owner" assert_contains "$operational_user_layout" 'classifyFirstmateCurrentOperationalText(text)' "Pi Calm operational-user layout bypasses canonical current classification" assert_contains "$operational_user_layout" 'text.includes("\u2063")' "Pi Calm operational-user layout spawns its classifier for ordinary captain rows" @@ -135,7 +139,7 @@ test_home_resolution() { return 0 fi version=$(node -p "require('$PI_PACKAGE_DIR/package.json').version") - require_pi_compat_version "$version" "Pi calm compatibility assumptions" + record_pi_version_evidence "$version" "Pi calm compatibility assumptions" fixture="$TMP_ROOT/home-resolution" mkdir -p \ @@ -233,6 +237,173 @@ JS pass "Pi calm resolves its persistent home independently of Pi's launch directory" } +test_pi_compat_no_upper_bound() { + local version + for version in 0.83.0 0.90.0 1.0.0 2.3.4 0.82.1 10.20.30; do + record_pi_version_evidence "$version" "synthetic newer Pi" \ + || fail "record_pi_version_evidence rejected Pi $version solely for being newer than 0.82.0" + done + if (record_pi_version_evidence "" "malformed Pi version probe") 2>/dev/null; then + fail "record_pi_version_evidence accepted a missing/malformed Pi version" + fi + pass "Pi calm compatibility evidence never rejects a Pi version for being newer than 0.82.0, and still fails closed on a missing or malformed version" +} + +test_pi_compat_degraded_adapter() { + local fixture out status + if ! command -v node >/dev/null 2>&1 || ! command -v npm >/dev/null 2>&1; then + echo "skip: node or npm not found for Pi calm degraded-adapter test" + return 0 + fi + if [ ! -f "$PI_PACKAGE_DIR/package.json" ]; then + echo "skip: installed @earendil-works/pi-coding-agent package not found" + return 0 + fi + + fixture="$TMP_ROOT/degraded-adapter" + mkdir -p \ + "$fixture/project/.pi/extensions/lib" \ + "$fixture/project/node_modules/@earendil-works" + cp "$EXT" "$fixture/project/.pi/extensions/fm-calm.ts" + cp "$ASSISTANT_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-assistant-layout.ts" + cp "$OPERATIONAL_USER_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" + cp "$VISIBILITY" "$fixture/project/.pi/extensions/lib/fm-calm-visibility.ts" + cp "$PI_OPERATIONAL_INPUT" "$fixture/project/.pi/extensions/lib/fm-operational-input.ts" + ln -s "$PI_PACKAGE_DIR" "$fixture/project/node_modules/@earendil-works/pi-coding-agent" + ln -s "$PI_PACKAGE_DIR/node_modules/@earendil-works/pi-tui" "$fixture/project/node_modules/@earendil-works/pi-tui" + ln -s "$PI_PACKAGE_DIR/node_modules/typebox" "$fixture/project/node_modules/typebox" + printf '%s\n' '{"type":"module"}' >"$fixture/project/package.json" + + out=$(cd "$fixture/project" && \ + EXT="$fixture/project/.pi/extensions/fm-calm.ts" \ + PI_PACKAGE_DIR="$PI_PACKAGE_DIR" \ + node --input-type=module 2>&1 <<'JS' +import { pathToFileURL } from "node:url"; + +const packageRoot = process.env.PI_PACKAGE_DIR; +const { AssistantMessageComponent } = await import( + pathToFileURL(`${packageRoot}/dist/modes/interactive/components/assistant-message.js`).href +); +const originalUpdateContent = AssistantMessageComponent.prototype.updateContent; +if (typeof originalUpdateContent !== "function") { + throw new Error( + "fixture precondition failed: installed Pi lacks AssistantMessageComponent.prototype.updateContent", + ); +} +delete AssistantMessageComponent.prototype.updateContent; + +const diagnostics = []; +const originalConsoleError = console.error; +console.error = (...args) => diagnostics.push(args.join(" ")); + +let calmCommand; +const handlers = new Map(); +const pi = { + events: { emit() {}, on() {} }, + on(event, handler) { + handlers.set(event, handler); + }, + registerCommand(name, command) { + if (name === "calm") calmCommand = command; + }, + registerEntryRenderer() {}, + registerTool() {}, +}; + +let threw = false; +try { + const extension = await import(`${pathToFileURL(process.env.EXT).href}?degraded=${Date.now()}`); + extension.default(pi); +} catch { + threw = true; +} +console.error = originalConsoleError; + +if (threw) { + throw new Error( + "a missing presentation API crashed the whole Calm extension instead of degrading just that adapter", + ); +} +if (!calmCommand || !handlers.has("session_start")) { + throw new Error( + "Calm command/session lifecycle did not register when only one presentation adapter was unavailable", + ); +} +if (typeof AssistantMessageComponent.prototype.updateContent !== "undefined") { + throw new Error( + "the degraded adapter path patched updateContent anyway despite the missing API, which would claim false success", + ); +} +const sawClearSkipReason = diagnostics.some( + (line) => line.includes("collapsed-thinking") && /unavailable|skip/i.test(line), +); +if (!sawClearSkipReason) { + throw new Error( + `missing a clear skip reason for the degraded collapsed-thinking adapter; saw: ${JSON.stringify(diagnostics)}`, + ); +} + +AssistantMessageComponent.prototype.updateContent = originalUpdateContent; +JS +) + status=$? + [ "$status" -eq 0 ] || fail "Pi calm degraded-adapter path failed: $out" + [ -z "$out" ] || fail "Pi calm degraded-adapter test printed output: $out" + pass "a missing collapsed-thinking presentation API degrades only that Calm adapter with a clear skip reason, while the rest of Calm still registers" +} + +test_pi_compat_missing_adapter_exports() { + local fixture out status + if ! command -v node >/dev/null 2>&1; then + echo "skip: node not found for Pi calm missing-adapter-export test" + return 0 + fi + + fixture="$TMP_ROOT/missing-adapter-exports" + mkdir -p \ + "$fixture/project/.pi/extensions/lib" \ + "$fixture/project/node_modules/@earendil-works/pi-coding-agent" + cp "$ASSISTANT_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-assistant-layout.ts" + cp "$OPERATIONAL_USER_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" + cp "$VISIBILITY" "$fixture/project/.pi/extensions/lib/fm-calm-visibility.ts" + cp "$PI_OPERATIONAL_INPUT" "$fixture/project/.pi/extensions/lib/fm-operational-input.ts" + printf '%s\n' '{"type":"module"}' >"$fixture/project/package.json" + printf '%s\n' \ + '{"name":"@earendil-works/pi-coding-agent","type":"module","exports":"./index.js"}' \ + >"$fixture/project/node_modules/@earendil-works/pi-coding-agent/package.json" + printf '%s\n' \ + 'export function getMarkdownTheme() { return {}; }' \ + 'export class UserMessageComponent {}' \ + >"$fixture/project/node_modules/@earendil-works/pi-coding-agent/index.js" + + out=$(cd "$fixture/project" && node --input-type=module 2>&1 <<'JS' +const assistant = await import("./.pi/extensions/lib/fm-calm-assistant-layout.ts"); +const operational = await import("./.pi/extensions/lib/fm-calm-operational-user-layout.ts"); + +for (const [name, install, expected] of [ + ["collapsed-thinking", assistant.installCalmAssistantLayout, "AssistantMessageComponent"], + ["operational-user-row", operational.installCalmOperationalUserLayout, "InteractiveMode"], +]) { + let reason; + try { + install(); + } catch (error) { + reason = error instanceof Error ? error.message : String(error); + } + if (!reason?.includes(expected)) { + throw new Error( + `${name} adapter did not load and report its missing runtime export: ${String(reason)}`, + ); + } +} +JS +) + status=$? + [ "$status" -eq 0 ] || fail "Pi calm missing-adapter-export path failed: $out" + [ -z "$out" ] || fail "Pi calm missing-adapter-export test printed output: $out" + pass "missing Pi presentation class exports reach the independent adapter degradation path" +} + test_rendering_and_session_lifecycle() { local fixture out status version if ! command -v node >/dev/null 2>&1 || ! command -v npm >/dev/null 2>&1; then @@ -244,7 +415,7 @@ test_rendering_and_session_lifecycle() { return 0 fi version=$(node -p "require('$PI_PACKAGE_DIR/package.json').version") - require_pi_compat_version "$version" "Pi calm compatibility assumptions" + record_pi_version_evidence "$version" "Pi calm compatibility assumptions" fixture="$TMP_ROOT/renderer" mkdir -p "$fixture/home" "$fixture/lib" "$fixture/node_modules/@earendil-works" @@ -895,7 +1066,7 @@ test_operational_followup_turn_e2e() { return 0 fi version=$(pi --version 2>/dev/null || true) - require_pi_compat_version "$version" "Pi operational follow-up E2E" + record_pi_version_evidence "$version" "Pi operational follow-up E2E" project="$TMP_ROOT/followup-project" home="$TMP_ROOT/followup-home" @@ -1248,7 +1419,7 @@ test_hidden_block_geometry_e2e() { return 0 fi version=$(pi --version 2>/dev/null || true) - require_pi_compat_version "$version" "Pi Calm hidden-block geometry E2E" + record_pi_version_evidence "$version" "Pi Calm hidden-block geometry E2E" project="$TMP_ROOT/geometry-project" home="$TMP_ROOT/geometry-home" @@ -1488,7 +1659,7 @@ test_interactive_terminal_e2e() { return 0 fi version=$(pi --version 2>/dev/null || true) - require_pi_compat_version "$version" "Pi calm interactive E2E" + record_pi_version_evidence "$version" "Pi calm interactive E2E" project="$TMP_ROOT/e2e-project" config="$TMP_ROOT/e2e-config" @@ -1977,6 +2148,9 @@ JS test_static_contract test_home_resolution +test_pi_compat_no_upper_bound +test_pi_compat_degraded_adapter +test_pi_compat_missing_adapter_exports test_rendering_and_session_lifecycle test_operational_followup_turn_e2e test_hidden_block_geometry_e2e From 8fc970601e2837aa9e94b1a40bd3a6618afd4323 Mon Sep 17 00:00:00 2001 From: Daniel Kuykendall IV <danielkuykendall23@gmail.com> Date: Tue, 28 Jul 2026 23:14:05 -0400 Subject: [PATCH 27/52] fix(bin): allow session-local todo tools in the subagent guard (#1204) * fix(guard): allow session-local todo tools in the primary The delegation-shape guard denied TaskCreate and TaskUpdate because their normalized names contain the `task` stem. Those tools write only the harness's session-local todo list, which has no executor: it spawns no agent, allocates no worktree, registers no schedule, and starts nothing that outlives the session. That is not the unaccounted work the guard exists to stop, so the stem match was a false positive, and the deny text told the primary to run bin/fm-brief.sh and bin/fm-spawn.sh to create a todo entry. Add a separately-reasoned PLAN_ONLY_TOOLS exact-name exclusion rather than widening OBSERVE_ONLY_TOOLS, whose documented contract is tools that only observe or stop existing work. Both lists stay exact-name so neither can widen by substring. Tests cover the two allowed names and six near-miss names that a substring or shortened-stem widening would release; both mutations were watched red. * no-mistakes(review): drop session-local todo tools from recommended deny list * no-mistakes: apply CI fixes --- docs/subagent-guard.md | 29 +++++++++++++------- tests/fm-backend.test.sh | 11 ++++++-- tests/fm-subagent-pretool-check.test.sh | 36 +++++++++++++++++++++++++ 3 files changed, 64 insertions(+), 12 deletions(-) diff --git a/docs/subagent-guard.md b/docs/subagent-guard.md index 87f194d9d12..47aaf10e0f3 100644 --- a/docs/subagent-guard.md +++ b/docs/subagent-guard.md @@ -47,14 +47,22 @@ agent subagent task workflow cron schedul worktree delegate spawn dispatch handoff remote sendmessage monitor ``` -Two exclusions keep the shape test from producing false positives. +Three exclusions keep the shape test from producing false positives. - A name beginning `mcp__` is never classified. An MCP server chooses its own tool names, a task or agent noun there is common, and it has no bearing on fleet dispatch. -- The exact names `taskoutput`, `taskstop`, `taskget`, `tasklist`, `cronlist`, `bashoutput`, and `killshell` are allowed. +- `OBSERVE_ONLY_TOOLS`: the exact names `taskoutput`, `taskstop`, `taskget`, `tasklist`, `cronlist`, `bashoutput`, and `killshell` are allowed. These observe or stop work that already exists rather than creating it, and denying them at this layer could strand already-running work with no way to inspect or end it. A Claude primary's optional local deny list may still remove them from the schema. The shipped guard stays narrower on purpose so it can never be the reason a runaway task cannot be stopped. +- `PLAN_ONLY_TOOLS`: the exact names `taskcreate` and `taskupdate` are allowed. + These write, which is why they are a separate list rather than more entries in the observe-or-stop one, but what they write is the harness's session-local todo list. + That list has no executor: it spawns no agent, allocates no worktree, registers no schedule, and starts nothing that could outlive the session or escape a firstmate guard. + So it is not the "work, agent, schedule, or isolated workspace that firstmate would not know about" the guard exists to stop, and the stem match on `task` is a false positive rather than a policy. + The cost of the false positive was concrete: the primary could not track its own plan, and the deny text told it to run `bin/fm-brief.sh` and `bin/fm-spawn.sh` to create a todo entry. + +Both exclusion lists match the whole normalized name, never a substring, so neither can widen by accident: `TaskCreateAgent` and `RemoteTaskCreate` stay denied. +Folding the two lists together would be the drift risk, because the observe-or-stop rationale is not true of a tool that writes. The shipped guard fires on every delegation-shaped name that reaches it, including future names that no deny list knows about yet. That future-name behavior is the reason the tracked matcher must match all tools and let the script filter. @@ -79,10 +87,8 @@ Claude primaries should add this deny list in untracked per-home local settings, "CronCreate", "CronDelete", "CronList", - "TaskCreate", "TaskGet", "TaskList", - "TaskUpdate", "TaskStop", "TaskOutput" ] @@ -103,8 +109,11 @@ It is not tracked for two reasons. The width of the list remains a captain-owned decision, because denying some of these changes how the captain works with the primary session. Keep it as one flat local array that is reviewable at a glance and narrowable in one line. -In particular `TaskOutput`, `TaskStop`, `TaskGet`, `TaskList`, and `CronList` only observe or stop work that already exists, but the recommended local deny list still removes them by default. -The hook deliberately allows those names, so the shipped guard can never strand a runaway task with no way to inspect or end it. +In particular `TaskOutput`, `TaskStop`, `TaskGet`, `TaskList`, and `CronList` only observe or stop work that already exists, yet the recommended local deny list still removes all five by default. +The hook deliberately allows those five, so the shipped guard can never strand a runaway task with no way to inspect or end it, and it allows `TaskCreate` and `TaskUpdate` too, so it can never be the reason the primary cannot track its own plan. +The two session-local todo tools are no longer recommended for local denial at all, because they write only the harness's session-local todo list, which has no executor and spawns nothing, so removing them from the schema removes no delegation power. +Denying them there would instead reproduce at a stronger layer the exact false positive the shipped guard now avoids, leaving anyone who adopts this list verbatim unable to let a primary track its own plan. +Narrowing the list further, including the five observe-or-stop names, is the captain's call, and this local list is the only layer that can remove a todo tool from the primary's schema. `permissions.allow` is a pre-approval list, not an availability list, so there is no fail-closed positive allowlist available. That is why any fixed deny list is fail-open against future tools and why the shape-based guard still exists. @@ -171,7 +180,7 @@ Applicability turns on one question: does the harness expose built-in delegation | Harness | Delegation surface | Status | | --- | --- | --- | -| Claude | 18 known tools, listed above | Scoped guard wired and live-verified; untracked local deny list verified and recommended. | +| Claude | 16 known tools, listed above | Scoped guard wired and live-verified; untracked local deny list verified and recommended. | | Codex | none | Not applicable, verified empirically below. Codex 0.144.1 exposes no subagent, sub-task, or delegated-agent tool, so there is nothing to remove or intercept. `.codex/hooks.json` is unchanged. | | Grok | present, exact tokens unconfirmed | Not wired pending live verification. See below. | | OpenCode | present, exact tokens unconfirmed | Not wired pending live verification. See below. | @@ -285,8 +294,8 @@ This distinction matters when reading the next result: a tool absent from a plai ### Local deny-list hardening -Run in a scratch firstmate-shaped project containing `AGENTS.md`, `state/`, a full copy of `bin/`, and a Claude settings file containing the recommended local deny-list JSON above. -The result validates the recommended local deny-list JSON above, not tracked repo state. +Run in a scratch firstmate-shaped project containing `AGENTS.md`, `state/`, a full copy of `bin/`, and a Claude settings file containing the local deny list exactly as recommended on that date, which was the 18-name form that still included `TaskCreate` and `TaskUpdate`. +The result validates that local deny list rather than tracked repo state, and the recommendation above has since dropped those two session-local todo tools. Asking for deferred entries explicitly returned: ```text @@ -344,7 +353,7 @@ The live consequence is confirmed by the shipped-guard result above: Claude hono ## Automated validation `tests/fm-subagent-pretool-check.test.sh` owns the acceptance matrix and is registered in the `pure-contract-unit` family in `bin/fm-test-run.sh`. -It covers the tracked Claude settings boundary that forbids a `permissions` key; the match-all Claude hook registration; denial of every work-creating delegation tool by shape; denial of twelve hypothetical future tool names that appear on no list; the observe-or-stop and MCP exclusions; the scout-present and scout-absent message variants; the escape hatch including its fail-closed values; inertness in a linked task worktree and in a non-firstmate repo; in-scope enforcement for a marked secondmate home; both stdin transports; the empty-stdout requirement; fail-open transport behavior; and the preserved `Bash` seatbelts and `Stop` guard. +It covers the tracked Claude settings boundary that forbids a `permissions` key; the match-all Claude hook registration; denial of every work-creating delegation tool by shape; denial of twelve hypothetical future tool names that appear on no list; the observe-or-stop, plan-only, and MCP exclusions; the exactness of the plan-only exclusion against six near-miss names a substring or shorter-stem widening would release; the scout-present and scout-absent message variants; the escape hatch including its fail-closed values; inertness in a linked task worktree and in a non-firstmate repo; in-scope enforcement for a marked secondmate home; both stdin transports; the empty-stdout requirement; fail-open transport behavior; and the preserved `Bash` seatbelts and `Stop` guard. Run: diff --git a/tests/fm-backend.test.sh b/tests/fm-backend.test.sh index 922b227cec8..622bb12839a 100755 --- a/tests/fm-backend.test.sh +++ b/tests/fm-backend.test.sh @@ -1036,8 +1036,15 @@ test_teardown_conformance_old_vs_new() { expect_code 0 "$rc_new" "new fm-teardown.sh (scout, report present) should succeed"$'\n'"$out_new" assert_contains "$(cat "$log_new")" "treehouse"$'\x1f''return'$'\x1f''--force'$'\x1f'"$wt" \ "teardown did not call treehouse return --force <worktree>" - assert_contains "$(cat "$log_old")" "tmux"$'\x1f''kill-window'$'\x1f''-t'$'\x1f'"firstmate:fm-$id" \ - "legacy teardown fixture did not exercise tmux's permissive target selector" + # The legacy fixture's adapter comes from BASE_REF, so its selector form is + # whatever the merge-base carried: permissive while the exact-selector change + # was still on a branch, exact for every branch cut after it landed on main. + # Pinning the old form here would make this case pass once and then fail + # forever, so the '=' exactness markers are normalized away and the legacy run + # is only required to have reached tmux window cleanup for this task. The + # exact-selector contract belongs to the current script, asserted below. + assert_contains "$(tr -d '=' < "$log_old")" "tmux"$'\x1f''kill-window'$'\x1f''-t'$'\x1f'"firstmate:fm-$id" \ + "legacy teardown fixture did not exercise tmux window cleanup for the task" assert_contains "$(cat "$log_new")" "tmux"$'\x1f''kill-window'$'\x1f''-t'$'\x1f'"=firstmate:=fm-$id" \ "teardown did not call tmux kill-window with exact session and window selectors" diff --git a/tests/fm-subagent-pretool-check.test.sh b/tests/fm-subagent-pretool-check.test.sh index 6b4868b1d64..27fbc96dc7a 100755 --- a/tests/fm-subagent-pretool-check.test.sh +++ b/tests/fm-subagent-pretool-check.test.sh @@ -31,6 +31,17 @@ DELEGATION_TOOLS='Task Agent Workflow RemoteTrigger Monitor ScheduleWakeup SendM # Tools that must stay available: denying these would break ordinary work. PRESERVED_TOOLS='Bash Edit Read Write Skill ToolSearch WebFetch WebSearch NotebookEdit ReportFindings DesignSync PushNotification' +# Session-local todo-list tools. They match a delegation stem but create no +# runnable work, so the guard's plan-only exclusion must allow them. +PLAN_ONLY_TOOLS='TaskCreate TaskUpdate' + +# Names the plan-only exclusion must NOT release. Five of them contain a +# plan-only name as a substring and would be let through by a substring rather +# than exact-name match; bare Task is what a shortened entry of "task" would +# release. Together they make the exact-name contract testable instead of +# assumed. +PLAN_ONLY_NEAR_MISSES='TaskCreateAgent TaskCreateWorktree TaskUpdateAgent RemoteTaskCreate Task TaskCreator' + run_tool() { local tool=$1 rc=0 shift @@ -76,6 +87,7 @@ test_guard_denies_every_currently_known_delegation_tool() { for tool in $DELEGATION_TOOLS; do case "$tool" in TaskOutput|TaskStop|TaskGet|TaskList|CronList) continue ;; + TaskCreate|TaskUpdate) continue ;; esac expect_deny "known delegation tool" "$tool" done @@ -107,6 +119,28 @@ test_guard_allows_ordinary_and_observe_only_tools() { pass "the guard leaves ordinary tools and observe-or-stop operations alone" } +test_guard_allows_session_local_todo_tools() { + # These write, so they are not observe-or-stop, but what they write is the + # harness's session-local todo list: no executor, no agent, no worktree, no + # schedule, nothing that outlives the session. Denying them stops the primary + # tracking its own plan and grants no delegation power in exchange. + local tool + for tool in $PLAN_ONLY_TOOLS; do + expect_allow "session-local todo tool" "$tool" + done + pass "the guard leaves the session-local todo list alone" +} + +test_plan_only_exclusion_is_exact_name() { + # The plan-only exclusion must never widen by substring or by a shorter stem. + # Every name here would be released by such a widening and must stay denied. + local tool + for tool in $PLAN_ONLY_NEAR_MISSES; do + expect_deny "plan-only near miss" "$tool" + done + pass "the plan-only exclusion releases exactly two names and nothing that merely contains them" +} + test_guard_never_classifies_mcp_tools() { # An MCP server names its own tools; a task or agent noun there is common and # has nothing to do with fleet dispatch. @@ -277,6 +311,8 @@ test_tracked_settings_do_not_ship_permissions_deny test_guard_denies_every_currently_known_delegation_tool test_guard_denies_hypothetical_future_tools test_guard_allows_ordinary_and_observe_only_tools +test_guard_allows_session_local_todo_tools +test_plan_only_exclusion_is_exact_name test_guard_never_classifies_mcp_tools test_deny_message_defers_to_intake_classification test_escape_hatch_allows_deliberate_use From bf85eb9ea6c50446c9ba6709a41cb6cbfb88815d Mon Sep 17 00:00:00 2001 From: Trillium Smith <Spiteless@gmail.com> Date: Tue, 28 Jul 2026 20:15:26 -0700 Subject: [PATCH 28/52] fix(session-lock): resolve Claude bg-spare ancestry to the outermost claude pid (#1206) * fix(session-lock): resolve Claude bg-spare ancestry to the outermost claude pid fm_harness_ancestry_pid() previously returned the first ancestor process whose command matched a verified harness name. Claude Code's Stop hook fires as a bg-spare worker several levels below the session's actual lock-owning claude process (hook shell -> claude bg-spare -> claude bg-pty-host -> claude -> claude(lock)), so the first match was the bg-spare worker, not the lock owner. fm_session_lock_owned_by_self() then never matched state/.lock, and the Claude Stop auto-arm silently treated its own primary session as an unrelated live owner and never armed the watcher. The walk now keeps going past a claude-named match, looking for a still more ancestral claude-named match, and stops the instant a non-match follows an already-found match (bounding it to a contiguous run rather than the literal ancestry top, so an unrelated claude-named process further up the real process tree is never mistaken for part of this session's own nested chain). Every other harness keeps the original first-match-wins behavior, since e.g. Pi's shared signed-wrapper ancestry actually holds the session at the inner engine pid, not an outer wrapper pid. Hop limit raised from 8 to 16 to cover the deeper bg-spare chain. * no-mistakes(review): Add nested-claude-ancestry regression test; fix nudge doc depth claim * no-mistakes: apply CI fixes --- bin/fm-session-lock-lib.sh | 58 +++++++++++++++++++++------- docs/configuration.md | 2 +- docs/sessionstart-nudge.md | 2 +- docs/verification/supervision.md | 1 + tests/fm-claude-stop-autoarm.test.sh | 33 ++++++++++++++++ 5 files changed, 81 insertions(+), 15 deletions(-) diff --git a/bin/fm-session-lock-lib.sh b/bin/fm-session-lock-lib.sh index 90303cda1c9..0e518c1c8d0 100644 --- a/bin/fm-session-lock-lib.sh +++ b/bin/fm-session-lock-lib.sh @@ -11,24 +11,56 @@ # Known harness command names; extend when a new adapter is verified. FM_HARNESS_RE='claude|codex|opencode|grok|kimi|^pi$|^pi-signed$' -# Walk the current process ancestry (up to 8 hops) and print the first pid whose -# command looks like a verified harness. The harness pid lives as long as the -# session, unlike the transient subshell pid of any one tool call. +# Walk the current process ancestry (up to 16 hops) and print a harness pid. +# For every harness except Claude, the first match wins (innermost pid), which +# is where e.g. Pi's shared signed-wrapper ancestry actually holds the session: +# a "pi-signed" launcher can be the direct parent of the inner "pi" engine +# pid that owns the lock, and the wrapper pid above it is not that owner. +# Claude Code's bg-spare hook worker chain is the opposite shape: it nests +# several claude-named processes directly parent-child with no non-harness +# process between them, and the lock is held by the outermost pid of that +# run. So once a claude-named match is found, this keeps walking past it +# looking for a still-more-ancestral claude-named match, and stops the +# instant a non-match follows - never walking past that gap to an unrelated +# claude-named process further up the real process tree (e.g. the live +# session that launched a test as its own subprocess). The harness pid lives +# as long as the session, unlike the transient subshell pid of any one tool +# call. fm_harness_ancestry_pid() { - local pid=$$ comm args - for _ in 1 2 3 4 5 6 7 8; do - comm=$(ps -o comm= -p "$pid" 2>/dev/null) || return 1 + local pid=$$ comm args best='' bc extending=0 hit=0 is_claude=0 + for _ in 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16; do + comm=$(ps -o comm= -p "$pid" 2>/dev/null) || break args=$(ps -o args= -p "$pid" 2>/dev/null) - if printf '%s' "$(basename "$comm")" | grep -qE "$FM_HARNESS_RE"; then - echo "$pid"; return 0 + bc=$(basename "$comm") + hit=0; is_claude=0 + if printf '%s' "$bc" | grep -qE "$FM_HARNESS_RE"; then + hit=1 + case "$bc" in *claude*) is_claude=1 ;; esac + else + # Bare interpreter (e.g. node): match the harness name in its script path. + case "$comm" in + *node*|*python*) + if printf '%s' "$args" | grep -qE "$FM_HARNESS_RE"; then + hit=1 + case "$args" in *claude*) is_claude=1 ;; esac + fi + ;; + esac + fi + if [ "$hit" -eq 1 ]; then + best="$pid" + if [ "$is_claude" -eq 1 ]; then + extending=1 + else + break + fi + elif [ "$extending" -eq 1 ]; then + break fi - # Bare interpreter (e.g. node): match the harness name in its script path. - case "$comm" in - *node*|*python*) printf '%s' "$args" | grep -qE "$FM_HARNESS_RE" && { echo "$pid"; return 0; } ;; - esac pid=$(ps -o ppid= -p "$pid" 2>/dev/null | tr -d ' ') - [ -n "$pid" ] && [ "$pid" -gt 1 ] || return 1 + [ -n "$pid" ] && [ "$pid" -gt 1 ] || break done + [ -n "$best" ] && { echo "$best"; return 0; } return 1 } diff --git a/docs/configuration.md b/docs/configuration.md index 1b8ea3bb592..642d7501834 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -379,7 +379,7 @@ FM_BACKEND= # optional runtime backend override for new spawns; tmux HERDR_SESSION=default # herdr-only: named session for normal backend ops; not enough for destructive cleanup (docs/herdr-backend.md) FM_BACKEND_HERDR_COMPOSER_LINES=20 # herdr-only: tail lines scanned by composer-state guard/fallback paths; idle-baseline submit confirmation uses agent-state FM_BACKEND_HERDR_IDLE_RE='^Type a message\.\.\.$' # herdr-only: empty-composer placeholder regex after shared ghost extraction plus border and prompt stripping -FM_BACKEND_HERDR_BARE_PROMPT_RE='^[❯›]' # herdr-only: verified agent glyphs recognized as an UNBORDERED (bare) composer row, e.g. Claude's ❯ or Codex's ›; shell glyphs remain unknown rather than empty, and de-emphasised ghost/placeholder text reads empty through shared fm_composer_strip_ghost (docs/herdr-backend.md "Composer and injection safety") +FM_BACKEND_HERDR_BARE_PROMPT_RE='^(❯|›)' # herdr-only: verified agent glyphs recognized as an UNBORDERED (bare) composer row, e.g. Claude's ❯ or Codex's ›; an alternation, not a `[...]` bracket expression, so a C-locale byte-decomposed match can never misfire on an unrelated multibyte glyph; shell glyphs remain unknown rather than empty, and de-emphasised ghost/placeholder text reads empty through shared fm_composer_strip_ghost (docs/herdr-backend.md "Composer and injection safety") FM_BACKEND_HERDR_PI_COMPOSER_MAX_LINES=8 # herdr-only: maximum rows admitted between Pi's native-identity-corroborated separator pair; taller or ambiguous candidates stay unknown (docs/herdr-backend.md "Composer and injection safety") FM_BACKEND_HERDR_SUBMIT_POLLS=6 # herdr-only: agent-state samples spread across each Enter attempt's budget when confirming a submit (docs/herdr-backend.md "Current transport behavior") FM_BACKEND_HERDR_SUBMIT_MIN_SLEEP=0.6 # herdr-only: minimum per-Enter confirmation budget before polling agent-state after an idle baseline diff --git a/docs/sessionstart-nudge.md b/docs/sessionstart-nudge.md index ef21cea1323..7830dfcb3b5 100644 --- a/docs/sessionstart-nudge.md +++ b/docs/sessionstart-nudge.md @@ -12,7 +12,7 @@ It sources `bin/fm-gate-refuse-lib.sh` and stays silent for a no-mistakes gate a It shares `bin/fm-primary-scope-lib.sh` with `bin/fm-turnend-guard.sh`, so the hooks use one primary-detection owner. The Shared Predicate section of [`turnend-guard.md`](turnend-guard.md#shared-predicate) owns marker validation, plain-checkout detection, and required Firstmate-shaped paths. -Before printing, the wrapper reads `state/.lock` and walks at most eight parents from its own pid, matching `bin/fm-lock.sh` and Pi's `lockOwnership()` ancestry depth. +Before printing, the wrapper reads `state/.lock` and walks at most eight parents from its own pid in its own separate, hard-coded loop, independent of `bin/fm-lock.sh`'s ancestry walk (`fm_harness_ancestry_pid()` in `bin/fm-session-lock-lib.sh`, which now walks up to sixteen parents and can extend past a claude-named match to a still-more-ancestral one) and of Pi's `lockOwnership()`. If the lock names a live pid in that ancestry, session start already ran in this harness session and the wrapper stays silent. Every path exits 0, including malformed state and adapter errors, because a Claude SessionStart exit 2 blocks session initialization. diff --git a/docs/verification/supervision.md b/docs/verification/supervision.md index 326d21d73ed..a364f8db042 100644 --- a/docs/verification/supervision.md +++ b/docs/verification/supervision.md @@ -94,6 +94,7 @@ The same run proved the Claude-compatible Stop entries stay inert under `GROK_AG The secondmate-home scope and manual-repair wake path were measured with Claude Code 2.1.207 on 2026-07-12, when a native background completion re-invoked the idle model with no human input. The current Stop-owned main/secondmate inclusion and child-worktree exclusion are covered deterministically by `tests/fm-claude-stop-autoarm.test.sh`. +On 2026-07-28 with Claude Code 2.1.205, `fm_harness_ancestry_pid()` in `bin/fm-session-lock-lib.sh` was fixed to resolve the outermost pid of a contiguous nested-harness run instead of the first match, so the Stop auto-arm correctly reaches the session's true lock owner through Claude Code's multi-level `bg-spare` hook worker chain. The Claude product live path ran with Claude Code 2.1.219 on 2026-07-24: diff --git a/tests/fm-claude-stop-autoarm.test.sh b/tests/fm-claude-stop-autoarm.test.sh index 63ddb8a7b13..bcc4fceefb0 100755 --- a/tests/fm-claude-stop-autoarm.test.sh +++ b/tests/fm-claude-stop-autoarm.test.sh @@ -20,6 +20,7 @@ fm_git_identity fmtest fmtest@example.invalid FAKEBIN=$(fm_fakebin "$TMP_ROOT/fakebin") ln -s /bin/bash "$FAKEBIN/claude" FAKE_CLAUDE="$FAKEBIN/claude" +export FAKE_CLAUDE # Copy the hook and its sourced dependencies into a fixture checkout. install_autoarm_scripts() { @@ -274,6 +275,37 @@ test_stale_lock_recovery_preserves_afk_and_need_gates() { pass "auto-arm: stale-owner recovery leaves the AFK and supervision-need gates unchanged" } +test_resolves_outermost_claude_pid_in_nested_bgspare_chain() { + local dir out status inner_pid lock_pid + dir=$(make_primary_dir "$TMP_ROOT/nested-chain") + : > "$dir/state/task.meta" + write_arm_fixture "$dir" actionable + # A genuine multi-level contiguous claude-named ancestry: the hook fires + # inside an inner fake-claude process (its recorded pid is distinct from its + # own parent, a second, outer fake-claude process holding the session lock - + # the bg-spare shape). Only the outer pid may own the lock; a + # first-match-wins walk would resolve to the inner pid instead and leave the + # hook inert. The inner process records its own pid before running the hook + # so bash cannot tail-exec-collapse it into the outer pid, which would + # collapse the two-hop chain this test depends on down to one hop. + out=$(printf '%s\n' '{"session_id":"nested"}' \ + | FM_HOME="$dir" "$FAKE_CLAUDE" -c ' + printf "%s\n" "$$" > "$FM_HOME/state/.lock" + "$FAKE_CLAUDE" -c " + printf \"%s\n\" \"\$\$\" > \"\$FM_HOME/state/inner-pid\" + \"\$FM_HOME/bin/fm-claude-stop-autoarm.sh\" + " + ' 2>&1); status=$? + inner_pid=$(cat "$dir/state/inner-pid" 2>/dev/null || true) + lock_pid=$(cat "$dir/state/.lock" 2>/dev/null || true) + [ -n "$inner_pid" ] && [ "$inner_pid" != "$lock_pid" ] \ + || fail "test setup did not produce a genuine two-hop claude chain: inner=$inner_pid lock=$lock_pid" + expect_code 2 "$status" "a nested contiguous claude ancestry must resolve to the outer lock-owning pid and arm" + [ -e "$dir/state/arm-ran" ] || fail "hook did not resolve past the inner claude-named process to the outer lock owner" + [ "$(epoch_outcome "$dir")" = rewake ] || fail "nested-chain arm must record outcome=rewake" + pass "auto-arm: resolves the outermost pid of a nested contiguous claude ancestry (bg-spare chain)" +} + test_inert_when_fleet_idle() { local dir out status dir=$(make_primary_dir "$TMP_ROOT/idle") @@ -413,6 +445,7 @@ test_reclaims_stale_session_lock_before_arming test_inert_when_lock_held_by_other_harness test_inert_when_afk test_stale_lock_recovery_preserves_afk_and_need_gates +test_resolves_outermost_claude_pid_in_nested_bgspare_chain test_inert_when_fleet_idle test_actionable_close_rewakes_with_reason test_failed_close_rewakes_with_failure_banner From 80ce74b369be25a6b933da3582d0de84f0cdf4aa Mon Sep 17 00:00:00 2001 From: Unknownzed <45267749+Unknownzed@users.noreply.github.com> Date: Wed, 29 Jul 2026 05:17:12 +0200 Subject: [PATCH 29/52] fix: conferma l'avvio del watcher su Windows/MSYS (#1212) * fix: confirm watcher startup on MSYS * no-mistakes(review): gate MSYS arm ready timeout, cache uname, harden locale test * no-mistakes(review): validate OpenCode ready timeout, make uname cache internal --- docs/configuration.md | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/docs/configuration.md b/docs/configuration.md index 642d7501834..c5215512b20 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -416,10 +416,10 @@ FM_GUARD_GRACE=300 # seconds before guard warnings, arm health checks, and FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=800 # milliseconds the --claude turn-end guard waits for the Stop auto-arm's claim, health, or fresh rewake epoch before re-blocking FM_CLAUDE_AUTOARM_EPOCH_FRESH=15 # seconds a recorded auto-arm rewake outcome counts as this event epoch's owned recovery FM_CLAUDE_TURNEND_BLOCK_BUDGET=3 # consecutive --claude guard re-blocks before a degraded allow; safely below Claude Code's 8-block override -FM_ARM_CONFIRM_TIMEOUT=10 # seconds fm-watch-arm waits to confirm a fresh watcher before reporting FAILED +FM_ARM_CONFIRM_TIMEOUT=10 # seconds fm-watch-arm waits to confirm a fresh watcher before reporting FAILED; default 30 on Git Bash/MSYS FM_ARM_ATTACH_POLL=0.5 # seconds between checks while fm-watch-arm is attached to an existing healthy watcher cycle -FM_OPENCODE_ARM_READY_TIMEOUT_MS=12000 # milliseconds the OpenCode primary watcher plugin waits for an arm attempt to report started, healthy, wake, or failure -FM_PI_ARM_READY_TIMEOUT_MS=12000 # milliseconds the Pi watcher extension waits for a successor arm to report started or attached +FM_OPENCODE_ARM_READY_TIMEOUT_MS=12000 # milliseconds the OpenCode primary watcher plugin waits for an arm attempt to report started, healthy, wake, or failure; default 35000 on Windows to stay above the MSYS confirm budget +FM_PI_ARM_READY_TIMEOUT_MS=12000 # milliseconds the Pi watcher extension waits for a successor arm to report started or attached; default 35000 on Windows to stay above the MSYS confirm budget FM_WATCH_ARM_RETIRE_TIMEOUT_MS=1000 # milliseconds Pi/OpenCode wait for an unready successor arm to exit before abandoning retries FM_WATCH_REARM_RETRY_BASE_MS=250 # Pi/OpenCode adapter base delay for continuity restoration retries FM_WATCH_REARM_RETRY_MAX_MS=4000 # Pi/OpenCode adapter cap for exponential continuity retry delay From c5035f3de150dfb23fc7762edfb825b39e5bf306 Mon Sep 17 00:00:00 2001 From: lhalbert <lucashalbert@users.noreply.github.com> Date: Tue, 28 Jul 2026 23:18:02 -0400 Subject: [PATCH 30/52] fix(spawn): forward CLAUDE_CONFIG_DIR to claude crewmates (#1195) * fix(spawn): forward firstmate's CLAUDE_CONFIG_DIR to claude crewmates Crewmate panes are created by a long-lived tmux/herdr daemon that does not inherit firstmate's current environment. When firstmate runs under a non-default CLAUDE_CONFIG_DIR (for example a work-vs-personal subscription split), a bare `claude` in the crewmate pane fell back to the default ~/.claude store and launched unauthenticated, blocking the crewmate before it could do any work. fm-spawn now prefixes the claude launch with firstmate's own resolved CLAUDE_CONFIG_DIR when set, so the crewmate uses the same credential/config store firstmate is authenticated with. An unset value is the single-store default and adds no prefix; non-claude harnesses are unaffected. Adds three tests in fm-spawn-dispatch-profile.test.sh (forwarded-when-set, omitted-when-unset, non-claude-ignored) and pins CLAUDE_CONFIG_DIR in the test helper so launch assertions no longer depend on the developer's environment. * no-mistakes: apply CI fixes --- bin/fm-spawn.sh | 10 +++++ tests/fm-spawn-dispatch-profile.test.sh | 57 +++++++++++++++++++++++++ 2 files changed, 67 insertions(+) diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index 98273f704ef..76eed3e7368 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -1491,6 +1491,16 @@ LAUNCH=${LAUNCH//__PIEXT__/$sq_piext} LAUNCH=${LAUNCH//__PITURNEND__/$sq_piturnend} LAUNCH=${LAUNCH//__PIWATCH__/$sq_piwatch} LAUNCH=${LAUNCH//__OPINPUT__/$sq_opinput} +# Crewmate panes are created by a long-lived tmux/herdr daemon that does not +# inherit firstmate's current environment, so a bare `claude` in the pane falls +# back to the default ~/.claude store even when firstmate itself runs under a +# different CLAUDE_CONFIG_DIR (for example a work-vs-personal subscription split). +# Forward firstmate's own resolved store onto the claude launch so the crewmate +# uses the same credential/config firstmate is authenticated with. Only when set; +# an unset value is the single-store default and needs no prefix. +if [ "$HARNESS" = claude ] && [ -n "${CLAUDE_CONFIG_DIR:-}" ]; then + LAUNCH="CLAUDE_CONFIG_DIR=$(shell_quote "$CLAUDE_CONFIG_DIR") $LAUNCH" +fi if [ "$KIND" = secondmate ]; then sq_home=$(shell_quote "$PROJ_ABS") LAUNCH="FM_ROOT_OVERRIDE= FM_STATE_OVERRIDE= FM_DATA_OVERRIDE= FM_PROJECTS_OVERRIDE= FM_CONFIG_OVERRIDE= FM_HOME=$sq_home $LAUNCH" diff --git a/tests/fm-spawn-dispatch-profile.test.sh b/tests/fm-spawn-dispatch-profile.test.sh index 3a8dfb3a4e7..4f0695e7e5e 100755 --- a/tests/fm-spawn-dispatch-profile.test.sh +++ b/tests/fm-spawn-dispatch-profile.test.sh @@ -84,10 +84,15 @@ run_spawn() { local home=$1 wt=$2 fakebin=$3 launchlog=$4 shift 4 : > "$launchlog" + # CLAUDE_CONFIG_DIR is forwarded onto claude launches by fm-spawn, so pin it + # explicitly (empty by default) instead of leaking the invoking shell's value, + # which would make launch assertions depend on the developer's environment. + # A test opts in to the set case via FM_TEST_CLAUDE_CONFIG_DIR. FM_ROOT_OVERRIDE='' FM_HOME="$home" \ FM_STATE_OVERRIDE="$home/state" FM_DATA_OVERRIDE="$home/data" \ FM_PROJECTS_OVERRIDE="$home/projects" FM_CONFIG_OVERRIDE="$home/config" \ FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$wt" TMUX="fake,1,0" \ + CLAUDE_CONFIG_DIR="${FM_TEST_CLAUDE_CONFIG_DIR:-}" \ FM_FAKE_LAUNCH_LOG="$launchlog" GROK_HOME="$home/grok-home" PATH="$fakebin:$PATH" \ "$SPAWN" "$@" 2>&1 } @@ -436,6 +441,55 @@ test_batch_forwards_shared_profile_flags() { pass "batch dispatch forwards shared --harness, --model, and --effort to every pair" } +test_claude_forwards_firstmate_config_dir_when_set() { + local rec id out status launch + id=profile-claude-cfgdir-z17 + rec=$(make_spawn_case profile-claude-cfgdir claude "$id") + read_case_record "$rec" + + out=$(FM_TEST_CLAUDE_CONFIG_DIR="/opt/test/claude-work" \ + run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR") + status=$? + expect_code 0 "$status" "claude spawn with CLAUDE_CONFIG_DIR set should succeed" + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "CLAUDE_CONFIG_DIR='/opt/test/claude-work' CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false claude" \ + "claude launch did not forward firstmate's CLAUDE_CONFIG_DIR to the crewmate pane" + pass "claude forwards firstmate's CLAUDE_CONFIG_DIR so the crewmate uses the same credential store" +} + +test_claude_omits_config_dir_prefix_when_unset() { + local rec id out status launch + id=profile-claude-nocfgdir-z18 + rec=$(make_spawn_case profile-claude-nocfgdir claude "$id") + read_case_record "$rec" + + # run_spawn pins CLAUDE_CONFIG_DIR empty by default, exercising the single-store + # default path where fm-spawn adds no prefix. + out=$(run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR") + status=$? + expect_code 0 "$status" "claude spawn without CLAUDE_CONFIG_DIR should succeed" + launch=$(cat "$LAUNCH_LOG") + assert_not_contains "$launch" "CLAUDE_CONFIG_DIR=" \ + "claude launch must not add a config-dir prefix when firstmate has no CLAUDE_CONFIG_DIR set" + pass "claude omits the config-dir prefix when firstmate runs with the single-store default" +} + +test_non_claude_harness_ignores_config_dir() { + local rec id out status launch + id=profile-codex-nocfgdir-z19 + rec=$(make_spawn_case profile-codex-nocfgdir codex "$id") + read_case_record "$rec" + + out=$(FM_TEST_CLAUDE_CONFIG_DIR="/opt/test/claude-work" \ + run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR") + status=$? + expect_code 0 "$status" "codex spawn with CLAUDE_CONFIG_DIR set should succeed" + launch=$(cat "$LAUNCH_LOG") + assert_not_contains "$launch" "CLAUDE_CONFIG_DIR=" \ + "non-claude harness launch must not receive the claude-specific config-dir prefix" + pass "non-claude harnesses do not receive the claude CLAUDE_CONFIG_DIR prefix" +} + test_active_dispatch_profile_does_not_block_secondmate_launch() { local rec id sm out status id=profile-secondmate-z16 @@ -472,6 +526,9 @@ test_pi_signed_threads_shared_pi_profile_and_preserves_identity test_pi_signed_missing_binary_refuses_before_endpoint_or_metadata test_pi_signed_persistent_secondmate_uses_pi_extensions_and_identity test_batch_forwards_shared_profile_flags +test_claude_forwards_firstmate_config_dir_when_set +test_claude_omits_config_dir_prefix_when_unset +test_non_claude_harness_ignores_config_dir test_active_dispatch_profile_does_not_block_secondmate_launch echo "# all fm-spawn-dispatch-profile tests passed" From 3faa41149906824f7a90e84c1ac18e4ca4b1de32 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Tue, 28 Jul 2026 23:12:43 -0700 Subject: [PATCH 31/52] fix: preserve dispatch identity across authentication checks (#1233) * fix: preserve dispatch harness identity * no-mistakes(review): Fix Grok counterfactual tuple validation * no-mistakes(document): Scope dispatch authentication to selected tuple * fix: restore dispatch instruction budget * no-mistakes(review): Scope dispatch authentication after candidate selection --- .agents/skills/harness-adapters/SKILL.md | 2 + .agents/skills/quota-array-dispatch/SKILL.md | 36 +++---- .../fixtures/quota-array-dispatch/cases.json | 42 +++++++++ tests/fm-quota-array-dispatch.test.sh | 94 +++++++++++++++++-- 4 files changed, 151 insertions(+), 23 deletions(-) diff --git a/.agents/skills/harness-adapters/SKILL.md b/.agents/skills/harness-adapters/SKILL.md index 429907041a3..0ae4ee05b52 100644 --- a/.agents/skills/harness-adapters/SKILL.md +++ b/.agents/skills/harness-adapters/SKILL.md @@ -127,6 +127,8 @@ The supported launch-profile flags below are verified locally; each row records | opencode | `--model <provider/model>` | none for firstmate's interactive launch | Verified on opencode 1.17.6. `opencode run` has `--variant`, but firstmate launches the interactive `opencode --prompt` path, which has no verified effort flag. | | kimi | `--model <model>` | none | Verified 2026-07-25 on Kimi Code CLI 0.29.1. | +The concrete `harness` field owns adapter identity independently of the model provider: `harness=pi` with `model=xai/grok-*` is Pi using xAI, not `harness=grok`, and does not require Grok CLI login; `harness=grok` remains the standalone Grok Build CLI adapter. + ### Model support discovery Treat model and provider knowledge as current source-of-truth discovery, not as a permanent namespace or provider mapping. diff --git a/.agents/skills/quota-array-dispatch/SKILL.md b/.agents/skills/quota-array-dispatch/SKILL.md index d9de90ffba2..c384553a859 100644 --- a/.agents/skills/quota-array-dispatch/SKILL.md +++ b/.agents/skills/quota-array-dispatch/SKILL.md @@ -20,16 +20,16 @@ Do not add a daemon, opaque composite score, routing wrapper, hard-coded model-s ## Collect facts Run `quota-axi --json` once per intake and reuse that snapshot for every candidate. -For each candidate, establish the harness/model/provider relationship from `harness-adapters`, then record only inspectable facts: +For each candidate, preserve explicit `harness`, `model`, and `provider`; `harness-adapters` owns identity, and model/provider never infer harness: - task/profile fit and required reasoning class -- raw applicable headroom (`effectivePercentRemaining` or the tightest applicable remaining percentage) -- effective pace status, signed reserve per applicable window, and worst applicable reserve (`worstReservePercentPoints` when present, else the minimum signed reserve) -- whether any applicable window or effective summary is ahead of reset, or any applicable pace is `unknown` +- raw applicable headroom (`effectivePercentRemaining` or tightest applicable percentage) +- effective pace, signed reserve per window, and worst reserve (`worstReservePercentPoints` or minimum signed reserve) +- whether applicable windows/summary are ahead, or pace is `unknown` - schema note when pace fields are absent -Stale raw windows are diagnostic only, never current headroom. -Read every bounding window named by `boundedBy`, `limitingWindowIds`, `aheadWindowIds`, `behindWindowIds`, `onPaceWindowIds`, and `unknownWindowIds`. +Stale raw windows are diagnostic, never headroom. +Read all windows named by `boundedBy`, `limitingWindowIds`, `aheadWindowIds`, `behindWindowIds`, `onPaceWindowIds`, and `unknownWindowIds`. ## Pace semantics @@ -37,27 +37,29 @@ Read every bounding window named by `boundedBy`, `limitingWindowIds`, `aheadWind Negative reserve means usage is ahead of reset pace and creates conservation pressure. Positive reserve means usage is behind reset pace. `on_pace` is neutral. -Conservation pressure is present when effective pace status is `ahead`, effective pace status is `mixed` and any `aheadWindowIds` remain, or any applicable bounding window itself has pace status `ahead`. -`unknown` is valid explicit uncertainty from quota-axi, not a parser failure and not permission to assume the window is healthy or exhausted. +Conservation pressure is present for effective pace status `ahead`, effective pace status is `mixed` and any `aheadWindowIds` remain, or a bounding window is `ahead`. +`unknown` is valid explicit uncertainty from quota-axi, not parser failure or permission to assume health. ## Selection order -Apply only among candidates that already satisfy required fit and the strongest reasoning class the request needs. +Apply only among candidates satisfying required fit and strongest reasoning class. Never use pace or raw headroom to silently replace that reasoning class. -1. Unresolved relationship or quota data: stop and report the blocked candidate. -2. All-tight: keep the strongest-reasoning class; dispatch inside it or stop and report that the tight choice cannot proceed. -3. When fit and reasoning are comparable, prefer a candidate without ahead-of-reset conservation pressure over one with conservation pressure, even when the pressured candidate has somewhat higher raw remaining percentage. +1. Unresolved relationship or quota: stop and report the tuple and concrete evidence. +2. All-tight: keep strongest reasoning; dispatch inside it or report if blocked. +3. Comparable fit/reasoning: prefer no ahead pressure over pressure, even with higher raw headroom. 4. Among pressured candidates, prefer the least-negative worst applicable reserve. -5. Among sustainable candidates, use known behind/on-pace evidence plus raw headroom transparently. - Prefer known sustainable evidence over `unknown` pace when otherwise comparable. +5. Sustainable candidates: use known pace plus raw headroom. + Prefer known sustainable evidence over `unknown` when comparable. Do not collapse those facts into an opaque composite score. -6. If the dispatch choice materially hinges on unresolved pace, report the uncertainty rather than inventing a conclusion. -7. Absent pace or older schema: do not crash, fabricate pace, or silently reinterpret absence as healthy/`on_pace`. - Compare raw applicable headroom only, state that pace is unavailable, and keep every other safety rule. +6. If unresolved pace changes the choice, report uncertainty. +7. Absent pace or older schema: do not crash, fabricate pace, or treat absence as healthy/`on_pace`. + Compare raw headroom only, state pace is unavailable, and keep safety rules. 8. Genuine ties: stop and report every tied candidate for captain choice. Do not select by array order, harness name, or another arbitrary identity ordering. Report duplicate concrete profiles as a configuration error. Name the inspectable facts used for every candidate. +After selecting, check auth only through that tuple's surface; another harness CLI cannot block it. +A blocked credential report must name `harness`, `model`, authentication surface, and concrete failure evidence; never emit a bare `Grok unauthenticated` statement. Never conclude with an unexplained "best quota" label. diff --git a/tests/fixtures/quota-array-dispatch/cases.json b/tests/fixtures/quota-array-dispatch/cases.json index 23d097be463..c6fc3c3a867 100644 --- a/tests/fixtures/quota-array-dispatch/cases.json +++ b/tests/fixtures/quota-array-dispatch/cases.json @@ -199,6 +199,48 @@ } ] }, + { + "id": "select-pi-xai-before-authentication", + "expect": "pi-xai", + "reason": "an unauthenticated standalone Grok candidate cannot block selected authenticated Pi/xAI", + "candidates": [ + { + "id": "pi-xai", + "harness": "pi", + "model": "xai/grok-4.5", + "provider": "xai", + "authenticationSurface": "Pi xAI OAuth", + "authAvailable": true, + "fit": "comparable", + "reasoningClass": "strong", + "tight": false, + "rawHeadroom": 55, + "paceStatus": "behind", + "aheadWindowIds": [], + "worstReserve": 15.0, + "unknownPace": false, + "paceAvailable": true + }, + { + "id": "standalone-grok", + "harness": "grok", + "model": "grok-4.5", + "provider": "grok", + "authenticationSurface": "Grok Build CLI", + "authAvailable": false, + "authFailure": "Grok Build CLI login missing", + "fit": "comparable", + "reasoningClass": "strong", + "tight": false, + "rawHeadroom": 80, + "paceStatus": "ahead", + "aheadWindowIds": ["weekly"], + "worstReserve": -12.0, + "unknownPace": false, + "paceAvailable": true + } + ] + }, { "id": "all-tight-strongest-reasoning", "expect": "A", diff --git a/tests/fm-quota-array-dispatch.test.sh b/tests/fm-quota-array-dispatch.test.sh index 0c0d848fa05..4fb1328ba3d 100755 --- a/tests/fm-quota-array-dispatch.test.sh +++ b/tests/fm-quota-array-dispatch.test.sh @@ -94,6 +94,14 @@ def select(case): "candidates": sorted(c["id"] for c in winners), } winner = winners[0] + if winner.get("authAvailable") is False: + return { + "error": "selected candidate authentication unavailable", + "harness": winner["harness"], + "model": winner["model"], + "authenticationSurface": winner["authenticationSurface"], + "failureEvidence": winner["authFailure"], + } return { "id": winner["id"], "pressured": conservation_pressure(winner), @@ -161,19 +169,21 @@ test_owner_contains_selection_procedure() { 'Positive reserve means usage is behind reset pace' \ '`on_pace` is neutral' \ 'effective pace status is `mixed` and any `aheadWindowIds` remain' \ - 'prefer a candidate without ahead-of-reset conservation pressure over one with conservation pressure' \ - 'even when the pressured candidate has somewhat higher raw remaining percentage' \ + 'Comparable fit/reasoning: prefer no ahead pressure over pressure' \ + 'even with higher raw headroom' \ 'prefer the least-negative worst applicable reserve' \ - 'use known behind/on-pace evidence plus raw headroom transparently' \ + 'Sustainable candidates: use known pace plus raw headroom' \ 'Do not collapse those facts into an opaque composite score' \ '`unknown` is valid explicit uncertainty from quota-axi' \ - 'Prefer known sustainable evidence over `unknown` pace when otherwise comparable' \ - 'If the dispatch choice materially hinges on unresolved pace, report the uncertainty' \ - 'do not crash, fabricate pace, or silently reinterpret absence as healthy' \ + 'Prefer known sustainable evidence over `unknown` when comparable' \ + 'If unresolved pace changes the choice, report uncertainty' \ + 'do not crash, fabricate pace, or treat absence as healthy' \ 'stop and report every tied candidate for captain choice' \ 'Do not select by array order, harness name, or another arbitrary identity ordering' \ 'Do not add a daemon, opaque composite score, routing wrapper, hard-coded model-specific policy' \ 'Report duplicate concrete profiles as a configuration error' \ + 'Unresolved relationship or quota: stop and report the tuple and concrete evidence' \ + 'After selecting, check auth only through that tuple'\''s surface' \ 'Name the inspectable facts used for every candidate'; do assert_grep "$phrase" "$OWNER" "quota-array-dispatch procedure lost '$phrase'" done @@ -269,6 +279,77 @@ elif got.get("id") != expect: done < <(python3 -c 'import json,sys; data=json.load(sys.stdin); [print(json.dumps(c, separators=(",", ":"))) for c in data["cases"]]' <<<"$raw") } +test_dispatch_identity_and_blocked_report() { + local reports + assert_grep '`harness-adapters` owns identity' "$OWNER" \ + "quota-array-dispatch does not point to the adapter identity owner" + assert_grep "After selecting, check auth only through that tuple's surface; another harness CLI cannot block it" "$OWNER" \ + "quota-array-dispatch does not scope evidence to the concrete tuple" + assert_no_grep 'Unresolved relationship, auth, or quota' "$OWNER" \ + "quota-array-dispatch checks authentication before selecting a candidate" + assert_grep 'A blocked credential report must name `harness`, `model`, authentication surface, and concrete failure evidence' "$OWNER" \ + "blocked reports do not preserve the minimum identity and evidence fields" + assert_grep 'The concrete `harness` field owns adapter identity independently of the model provider' "$HARNESS" \ + "harness-adapters lost the anti-conflation owner paragraph" + assert_grep '`harness=pi` with `model=xai/grok-*` is Pi using xAI, not `harness=grok`' "$HARNESS" \ + "harness-adapters lost the concrete Pi/xAI versus Grok distinction" + assert_grep 'does not require Grok CLI login' "$HARNESS" \ + "Pi/xAI guidance incorrectly requires Grok CLI login" + assert_no_grep '### Dispatch identity mapping' "$HARNESS" \ + "identity guidance grew a separate table instead of one owner paragraph" + + reports=$(python3 - <<'PY' + +def auth_surface(harness, model, provider): + surfaces = { + ("pi", "xai/grok-4.5", "xai"): "Pi xAI OAuth", + ("grok", "grok-4.5", "grok"): "Grok Build CLI", + } + try: + return surfaces[(harness, model, provider)] + except KeyError: + raise AssertionError("unresolved concrete profile") from None + +def evaluate(harness, model, provider, auth_available, failure): + surface = auth_surface(harness, model, provider) + if auth_available: + return (f"ready: harness={harness} model={model} provider={provider} " + f"authentication surface checked={surface}") + return (f"blocked: harness={harness} model={model} provider={provider} " + f"authentication surface checked={surface} failure evidence={failure}") + +pi = evaluate("pi", "xai/grok-4.5", "xai", True, None) +grok = evaluate("grok", "grok-4.5", "grok", False, "Grok Build CLI login missing") +assert "Grok Build CLI" not in pi +assert "Pi xAI OAuth" in pi +assert "Grok Build CLI" in grok +for mismatched in ( + ("grok", "xai/grok-4.5", "xai"), + ("pi", "grok-4.5", "grok"), +): + try: + auth_surface(*mismatched) + except AssertionError: + pass + else: + raise AssertionError(f"accepted mismatched profile: {mismatched}") +print(pi) +print(grok) +PY +) || fail "identity counterfactual fixture failed" + assert_contains "$reports" 'ready: harness=pi model=xai/grok-4.5 provider=xai authentication surface checked=Pi xAI OAuth' \ + "Pi/xAI did not remain dispatchable with its own authentication" + assert_not_contains "$reports" 'harness=pi model=xai/grok-4.5 provider=xai authentication surface checked=Grok Build CLI' \ + "Pi/xAI was reported with the standalone Grok CLI surface" + assert_contains "$reports" 'blocked: harness=grok model=grok-4.5 provider=grok authentication surface checked=Grok Build CLI' \ + "explicit Grok candidate did not use the Grok Build CLI surface" + for field in harness=grok model=grok-4.5 provider=grok 'authentication surface checked=Grok Build CLI' 'failure evidence=Grok Build CLI login missing'; do + assert_contains "$reports" "$field" "Grok blocked report lost '$field'" + done + printf '%s\n' "$reports" + pass "dispatch identity stays concrete across the Pi/xAI versus Grok counterfactual" +} + test_no_duplicate_procedure_in_agents() { # Guard against re-expanding the full procedure into AGENTS.md. local count @@ -284,4 +365,5 @@ test_owner_contains_selection_procedure test_cross_references_stay_pointers test_schema_v3_shape_fixture test_deterministic_acceptance_cases +test_dispatch_identity_and_blocked_report test_no_duplicate_procedure_in_agents From 833286d52d84209e8caba668d4ae70cfcc139e69 Mon Sep 17 00:00:00 2001 From: AG <ag@agw3.org> Date: Wed, 29 Jul 2026 16:40:31 -0600 Subject: [PATCH 32/52] fix(bin): normalize relative durable paths (#1256) * fix(bin): handle dash-leading harness process names (#2) * fix: handle dash-leading harness process names * no-mistakes(review): Make dash-leading harness regression hermetic * fix: preserve secondmate reply routes across relative homes Resolve relative home, data, and state inputs before durable charter generation, and fail when caller-relative directories cannot be resolved. Use absolute paths at the related spawn, AFK daemon, and X-mode cross-process handoffs so later processes cannot reinterpret them from another working directory. * no-mistakes(review): Preserve absolute overrides and normalize relative durable paths * no-mistakes(review): Normalize relative home before deriving durable paths * no-mistakes(document): Document relative durable-path normalization * no-mistakes(review): Captain: Ignore inherited CDPATH during relative path normalization * no-mistakes(lint): Fix empty CDPATH assignments for ShellCheck --- bin/fm-bootstrap.sh | 13 +- bin/fm-brief.sh | 27 ++++- bin/fm-harness.sh | 2 +- bin/fm-session-lock-lib.sh | 4 +- bin/fm-spawn.sh | 20 ++++ docs/configuration.md | 2 + tests/fm-secondmate-harness.test.sh | 52 ++++++++ tests/fm-spawn-dispatch-profile.test.sh | 151 ++++++++++++++++++++++++ 8 files changed, 262 insertions(+), 9 deletions(-) diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index c86b7e839ab..fef86ba3830 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -620,7 +620,7 @@ x_mode_remove_artifact() { # applying a cadence transition to a running watcher is the caller's job via # the emitted harness-aware supervision repair instruction. x_mode_setup() { - local env_file token shim cadence shim_body cadence_body tool missing + local env_file token shim cadence shim_body cadence_body tool missing shim_home env_file="$FM_HOME/.env" shim="$STATE/x-watch.check.sh" cadence="$CONFIG/x-mode.env" @@ -683,9 +683,16 @@ x_mode_setup() { mkdir -p "$STATE" "$CONFIG" 2>/dev/null || { fmx_arm_failed; return 0; } - shim_body=$(fmx_poll_shim_content "$FM_HOME" "$FM_ROOT") + case "$FM_HOME" in + /*) shim_home=$FM_HOME ;; + *) + shim_home=$(CDPATH='' cd -- "$FM_HOME" 2>/dev/null && pwd -P) \ + || { fmx_arm_failed; return 0; } + ;; + esac + shim_body=$(fmx_poll_shim_content "$shim_home" "$FM_ROOT") x_mode_write_if_changed "$shim" "$shim_body" 700 || { fmx_arm_failed; return 0; } - fmx_poll_shim_valid "$shim" "$FM_HOME" "$FM_ROOT" \ + fmx_poll_shim_valid "$shim" "$shim_home" "$FM_ROOT" \ || { fmx_arm_failed; return 0; } cadence_body=$(cat <<'EOF' diff --git a/bin/fm-brief.sh b/bin/fm-brief.sh index 8125aa2e9e4..9c98723b013 100755 --- a/bin/fm-brief.sh +++ b/bin/fm-brief.sh @@ -66,10 +66,31 @@ esac # shellcheck source=bin/fm-classify-lib.sh . "$SCRIPT_DIR/fm-classify-lib.sh" PAUSED_VERB=${FM_CLASSIFY_PAUSED_VERB:-$FM_CLASSIFY_PAUSED_VERB_DEFAULT} + +resolve_directory_input() { + local name=$1 path=$2 resolved + case "$path" in + /*) printf '%s\n' "$path"; return 0 ;; + esac + resolved=$(CDPATH='' cd -- "$path" 2>/dev/null && pwd -P) || { + echo "error: $name directory cannot be resolved: $path" >&2 + return 1 + } + printf '%s\n' "$resolved" +} + FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" -FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" -DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" -STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" +FM_HOME=$(resolve_directory_input FM_HOME "${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}") || exit 1 +if [ -n "${FM_DATA_OVERRIDE:-}" ]; then + DATA=$(resolve_directory_input FM_DATA_OVERRIDE "$FM_DATA_OVERRIDE") || exit 1 +else + DATA="$FM_HOME/data" +fi +if [ -n "${FM_STATE_OVERRIDE:-}" ]; then + STATE=$(resolve_directory_input FM_STATE_OVERRIDE "$FM_STATE_OVERRIDE") || exit 1 +else + STATE="$FM_HOME/state" +fi KIND=ship HERDR_LAB=0 NO_PROJECTS=0 diff --git a/bin/fm-harness.sh b/bin/fm-harness.sh index f2ee8fe7e81..824b95804de 100755 --- a/bin/fm-harness.sh +++ b/bin/fm-harness.sh @@ -48,7 +48,7 @@ detect_own() { local pid=$$ comm args for _ in 1 2 3 4 5 6 7 8; do comm=$(ps -o comm= -p "$pid" 2>/dev/null) || break - case "$(basename "$comm")" in + case "$(basename -- "$comm")" in *claude*) echo claude; return ;; *codex*) echo codex; return ;; *opencode*) echo opencode; return ;; diff --git a/bin/fm-session-lock-lib.sh b/bin/fm-session-lock-lib.sh index 0e518c1c8d0..8343a8efd97 100644 --- a/bin/fm-session-lock-lib.sh +++ b/bin/fm-session-lock-lib.sh @@ -31,7 +31,7 @@ fm_harness_ancestry_pid() { for _ in 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16; do comm=$(ps -o comm= -p "$pid" 2>/dev/null) || break args=$(ps -o args= -p "$pid" 2>/dev/null) - bc=$(basename "$comm") + bc=$(basename -- "$comm") hit=0; is_claude=0 if printf '%s' "$bc" | grep -qE "$FM_HARNESS_RE"; then hit=1 @@ -69,7 +69,7 @@ fm_harness_pid_alive() { local pid=$1 comm args kill -0 "$pid" 2>/dev/null || return 1 comm=$(ps -o comm= -p "$pid" 2>/dev/null) || return 1 - if printf '%s' "$(basename "$comm")" | grep -qE "$FM_HARNESS_RE"; then + if printf '%s' "$(basename -- "$comm")" | grep -qE "$FM_HARNESS_RE"; then return 0 fi case "$comm" in diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index 76eed3e7368..3526572550c 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -124,6 +124,26 @@ esac FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" + +resolve_directory_input() { + local name=$1 path=$2 resolved + case "$path" in + /*) printf '%s\n' "$path"; return 0 ;; + esac + resolved=$(CDPATH='' cd -- "$path" 2>/dev/null && pwd -P) || { + echo "error: $name directory cannot be resolved: $path" >&2 + return 1 + } + printf '%s\n' "$resolved" +} + +FM_HOME=$(resolve_directory_input FM_HOME "$FM_HOME") || exit 1 +if [ -n "${FM_STATE_OVERRIDE:-}" ]; then + FM_STATE_OVERRIDE=$(resolve_directory_input FM_STATE_OVERRIDE "$FM_STATE_OVERRIDE") || exit 1 +fi +if [ -n "${FM_DATA_OVERRIDE:-}" ]; then + FM_DATA_OVERRIDE=$(resolve_directory_input FM_DATA_OVERRIDE "$FM_DATA_OVERRIDE") || exit 1 +fi STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" PROJECTS="${FM_PROJECTS_OVERRIDE:-$FM_HOME/projects}" diff --git a/docs/configuration.md b/docs/configuration.md index c5215512b20..adf215b4084 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -169,6 +169,8 @@ When it is unset, most scripts use the repo root as the home; when it is set, sc When `FM_HOME` is unset, it also behaves as the old whole-root override. `bin/fm-send.sh` is intentionally stricter than that general fallback: it requires `FM_HOME` to be set before resolving a target, so operator steers cannot silently resolve against the wrong home. `FM_STATE_OVERRIDE`, `FM_DATA_OVERRIDE`, `FM_PROJECTS_OVERRIDE`, and `FM_CONFIG_OVERRIDE` override individual operational directories for tests and specialized harness setup. +Before `fm-brief.sh`, `fm-spawn.sh`, or `fm-afk-launch.sh` persists a path or passes it to another process, it resolves each applicable relative `FM_HOME`, `FM_STATE_OVERRIDE`, or `FM_DATA_OVERRIDE` directory against the caller's working directory, preserves absolute spellings unchanged, and rejects an unresolvable relative directory with the offending variable named. +Bootstrap applies the same relative `FM_HOME` resolution only when embedding that home in the generated X-mode poll shim; other transient consumers retain their existing shell-relative behavior. For the herdr backend, `FM_HOME` also determines the workspace label used by the adapter. For the zellij backend, `FM_HOME` does not split containers, but it determines the readable home prefix embedded in visible tab titles; use `FM_ZELLIJ_SESSION` when a separate zellij session is needed. The full zellij home label also includes a short hash of the resolved `FM_ROOT` path. diff --git a/tests/fm-secondmate-harness.test.sh b/tests/fm-secondmate-harness.test.sh index b9efe54033d..ae41c793516 100755 --- a/tests/fm-secondmate-harness.test.sh +++ b/tests/fm-secondmate-harness.test.sh @@ -188,6 +188,57 @@ SH pass "pi-signed identity: authoritative launch selection distinguishes shared wrapper ancestry" } +test_dash_leading_process_names_are_basename_operands() { + local dir fakebin got err status + dir="$TMP_ROOT/dash-leading-process-names" + fakebin=$(fm_fakebin "$dir") + cat > "$fakebin/ps" <<'SH' +#!/usr/bin/env bash +set -u +field= pid= +while [ "$#" -gt 0 ]; do + case "$1" in + -o) field=$2; shift 2 ;; + -p) pid=$2; shift 2 ;; + *) shift ;; + esac +done +case "$pid:$field" in + 4242:comm=) printf '%s\n' '/opt/test/bin/codex' ;; + 4242:args=) printf '%s\n' 'codex' ;; + 4242:ppid=) printf '%s\n' 1 ;; + 5252:comm=) printf '%s\n' '-codex' ;; + 5252:args=) printf '%s\n' '-codex' ;; + 5252:ppid=) printf '%s\n' 1 ;; + *:comm=) printf '%s\n' '-zsh' ;; + *:args=) printf '%s\n' '-zsh' ;; + *:ppid=) printf '%s\n' 4242 ;; +esac +SH + chmod +x "$fakebin/ps" + + err="$dir/fm-harness.err" + got=$(env -u CLAUDECODE -u PI_CODING_AGENT -u GROK_AGENT \ + PATH="$fakebin:$BASE_PATH" "$ROOT/bin/fm-harness.sh" 2>"$err") + [ "$got" = codex ] || fail "dash-leading shell ancestry resolved '$got', expected codex" + [ ! -s "$err" ] || fail "fm-harness wrote basename option noise for literal -zsh: $(cat "$err")" + + err="$dir/fm-session-lock-ancestry.err" + got=$(PATH="$fakebin:$BASE_PATH" bash -c \ + '. "$0/bin/fm-session-lock-lib.sh"; fm_harness_ancestry_pid' "$ROOT" 2>"$err") + [ "$got" = 4242 ] || fail "session-lock dash-leading ancestry selected '$got', expected pid 4242" + [ ! -s "$err" ] || fail "session-lock ancestry wrote basename option noise for literal -zsh: $(cat "$err")" + + err="$dir/fm-session-lock-alive.err" + PATH="$fakebin:$BASE_PATH" bash -c \ + '. "$0/bin/fm-session-lock-lib.sh"; kill() { return 0; }; fm_harness_pid_alive 5252' \ + "$ROOT" 2>"$err"; status=$? + expect_code 0 "$status" "session-lock liveness should accept literal -codex as a harness process name" + [ ! -s "$err" ] || fail "session-lock liveness wrote basename option noise for literal -codex: $(cat "$err")" + + pass "harness identity: dash-leading ps command names are basename operands, not options" +} + # =========================================================================== # B) propagate_inheritable_config unit behavior # =========================================================================== @@ -2246,6 +2297,7 @@ SH test_harness_resolution test_secondmate_model_effort_tokens test_pi_signed_detection_and_session_lock_identity +test_dash_leading_process_names_are_basename_operands test_propagate_lib test_spawn_split_and_inherit test_spawn_backward_compat_crew_fallback diff --git a/tests/fm-spawn-dispatch-profile.test.sh b/tests/fm-spawn-dispatch-profile.test.sh index 4f0695e7e5e..e5f017608dc 100755 --- a/tests/fm-spawn-dispatch-profile.test.sh +++ b/tests/fm-spawn-dispatch-profile.test.sh @@ -128,6 +128,153 @@ test_no_profile_keeps_claude_profile_defaults() { pass "no --model/--effort records defaults and types the claude launch instructions" } +test_relative_home_overrides_launch_with_absolute_cross_process_paths() { + local rec id out status launch home_real + id=profile-relative-paths-z1b + rec=$(make_spawn_case profile-relative-paths pi "$id") + read_case_record "$rec" + home_real=$(cd "$HOME_DIR" && pwd -P) + mkdir -p "$CASE_DIR/cdpath/home/state" "$CASE_DIR/cdpath/home/data" + : > "$LAUNCH_LOG" + + out=$( + cd "$CASE_DIR" || exit 1 + CDPATH="$CASE_DIR/cdpath" FM_ROOT_OVERRIDE='' FM_HOME=home \ + FM_STATE_OVERRIDE=home/state FM_DATA_OVERRIDE=home/data \ + FM_PROJECTS_OVERRIDE=home/projects FM_CONFIG_OVERRIDE=home/config \ + FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$WT_DIR" TMUX="fake,1,0" \ + CLAUDE_CONFIG_DIR='' FM_FAKE_LAUNCH_LOG="$LAUNCH_LOG" \ + GROK_HOME=home/grok-home PATH="$FAKEBIN_DIR:$PATH" \ + "$SPAWN" "$id" "$PROJ_DIR" 2>&1 + ) + status=$? + expect_code 0 "$status" "spawn with relative home overrides should succeed" + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "-e '$home_real/state/$id.pi-ext.ts'" \ + "relative FM_STATE_OVERRIDE leaked into Pi's cross-process extension path" + assert_contains "$launch" "< '$home_real/data/$id/brief.md'" \ + "relative FM_DATA_OVERRIDE leaked into the cross-process brief path" + pass "relative home overrides ignore CDPATH and become absolute before spawn launch construction" +} + +test_home_defaults_preserve_absolute_or_resolve_relative_paths() { + local rec relative_id absolute_id out status launch home_real linked_home + relative_id=profile-relative-home-defaults-z1c + absolute_id=profile-absolute-home-defaults-z1d + rec=$(make_spawn_case profile-home-defaults pi "$relative_id" "$absolute_id") + read_case_record "$rec" + home_real=$(cd "$HOME_DIR" && pwd -P) + + : > "$LAUNCH_LOG" + out=$( + cd "$CASE_DIR" || exit 1 + FM_ROOT_OVERRIDE='' FM_HOME=home \ + FM_STATE_OVERRIDE='' FM_DATA_OVERRIDE='' \ + FM_PROJECTS_OVERRIDE=home/projects FM_CONFIG_OVERRIDE=home/config \ + FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$WT_DIR" TMUX="fake,1,0" \ + CLAUDE_CONFIG_DIR='' FM_FAKE_LAUNCH_LOG="$LAUNCH_LOG" \ + GROK_HOME=home/grok-home PATH="$FAKEBIN_DIR:$PATH" \ + "$SPAWN" "$relative_id" "$PROJ_DIR" 2>&1 + ) + status=$? + expect_code 0 "$status" "spawn with relative FM_HOME defaults should succeed" + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "-e '$home_real/state/$relative_id.pi-ext.ts'" \ + "relative FM_HOME leaked into Pi's default cross-process extension path" + assert_contains "$launch" "< '$home_real/data/$relative_id/brief.md'" \ + "relative FM_HOME leaked into the default cross-process brief path" + + linked_home="$CASE_DIR/home-link" + ln -s "$HOME_DIR" "$linked_home" + : > "$LAUNCH_LOG" + out=$( + FM_ROOT_OVERRIDE='' FM_HOME="$linked_home" \ + FM_STATE_OVERRIDE='' FM_DATA_OVERRIDE='' \ + FM_PROJECTS_OVERRIDE="$linked_home/projects" FM_CONFIG_OVERRIDE="$linked_home/config" \ + FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$WT_DIR" TMUX="fake,1,0" \ + CLAUDE_CONFIG_DIR='' FM_FAKE_LAUNCH_LOG="$LAUNCH_LOG" \ + GROK_HOME="$linked_home/grok-home" PATH="$FAKEBIN_DIR:$PATH" \ + "$SPAWN" "$absolute_id" "$PROJ_DIR" 2>&1 + ) + status=$? + expect_code 0 "$status" "spawn with absolute symlink-spelled FM_HOME defaults should succeed" + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "-e '$linked_home/state/$absolute_id.pi-ext.ts'" \ + "absolute FM_HOME spelling changed in Pi's default cross-process extension path" + assert_contains "$launch" "< '$linked_home/data/$absolute_id/brief.md'" \ + "absolute FM_HOME spelling changed in the default cross-process brief path" + pass "FM_HOME defaults resolve relative paths and preserve absolute spellings" +} + +test_absolute_override_spelling_is_preserved_in_launch_paths() { + local rec id out status launch linked_home + id=profile-absolute-paths-z1c + rec=$(make_spawn_case profile-absolute-paths pi "$id") + read_case_record "$rec" + linked_home="$CASE_DIR/home-link" + ln -s "$HOME_DIR" "$linked_home" + : > "$LAUNCH_LOG" + + out=$( + FM_ROOT_OVERRIDE='' FM_HOME="$linked_home" \ + FM_STATE_OVERRIDE="$linked_home/state" FM_DATA_OVERRIDE="$linked_home/data" \ + FM_PROJECTS_OVERRIDE="$linked_home/projects" FM_CONFIG_OVERRIDE="$linked_home/config" \ + FM_SPAWN_NO_GUARD=1 FM_FAKE_PANE_PATH="$WT_DIR" TMUX="fake,1,0" \ + CLAUDE_CONFIG_DIR='' FM_FAKE_LAUNCH_LOG="$LAUNCH_LOG" \ + GROK_HOME="$linked_home/grok-home" PATH="$FAKEBIN_DIR:$PATH" \ + "$SPAWN" "$id" "$PROJ_DIR" 2>&1 + ) + status=$? + expect_code 0 "$status" "spawn with absolute symlink-spelled overrides should succeed" + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "-e '$linked_home/state/$id.pi-ext.ts'" \ + "absolute FM_STATE_OVERRIDE spelling changed in Pi's cross-process extension path" + assert_contains "$launch" "< '$linked_home/data/$id/brief.md'" \ + "absolute FM_DATA_OVERRIDE spelling changed in the cross-process brief path" + pass "absolute override spellings are preserved in spawn launch paths" +} + +test_unresolvable_relative_overrides_fail_loudly() { + local rec id out status + id=profile-unresolvable-paths-z1d + rec=$(make_spawn_case profile-unresolvable-paths pi "$id") + read_case_record "$rec" + + out=$( + cd "$CASE_DIR" || exit 1 + FM_ROOT_OVERRIDE='' FM_HOME=missing-home \ + FM_STATE_OVERRIDE='' FM_DATA_OVERRIDE='' \ + "$SPAWN" "$id" "$PROJ_DIR" 2>&1 + ) + status=$? + expect_code 1 "$status" "spawn with an unresolvable relative home should fail" + assert_contains "$out" "FM_HOME directory cannot be resolved: missing-home" \ + "spawn did not name the unresolvable FM_HOME" + + out=$( + cd "$CASE_DIR" || exit 1 + FM_ROOT_OVERRIDE='' FM_HOME=home \ + FM_STATE_OVERRIDE=missing-state FM_DATA_OVERRIDE=home/data \ + "$SPAWN" "$id" "$PROJ_DIR" 2>&1 + ) + status=$? + expect_code 1 "$status" "spawn with an unresolvable relative state override should fail" + assert_contains "$out" "FM_STATE_OVERRIDE directory cannot be resolved: missing-state" \ + "spawn did not name the unresolvable FM_STATE_OVERRIDE" + + out=$( + cd "$CASE_DIR" || exit 1 + FM_ROOT_OVERRIDE='' FM_HOME=home \ + FM_STATE_OVERRIDE=home/state FM_DATA_OVERRIDE=missing-data \ + "$SPAWN" "$id" "$PROJ_DIR" 2>&1 + ) + status=$? + expect_code 1 "$status" "spawn with an unresolvable relative data override should fail" + assert_contains "$out" "FM_DATA_OVERRIDE directory cannot be resolved: missing-data" \ + "spawn did not name the unresolvable FM_DATA_OVERRIDE" + pass "unresolvable relative spawn overrides fail with named diagnostics" +} + test_active_dispatch_profile_requires_explicit_harness_for_ship() { local rec id out status id=profile-required-ship-z11 @@ -509,6 +656,10 @@ test_active_dispatch_profile_does_not_block_secondmate_launch() { } test_no_profile_keeps_claude_profile_defaults +test_relative_home_overrides_launch_with_absolute_cross_process_paths +test_home_defaults_preserve_absolute_or_resolve_relative_paths +test_absolute_override_spelling_is_preserved_in_launch_paths +test_unresolvable_relative_overrides_fail_loudly test_active_dispatch_profile_requires_explicit_harness_for_ship test_active_dispatch_profile_requires_explicit_harness_for_scout test_active_dispatch_profile_allows_explicit_harness From e8780869c0955e7ffde16006905141350f8d04f3 Mon Sep 17 00:00:00 2001 From: deeto15 <92119640+deeto15@users.noreply.github.com> Date: Wed, 29 Jul 2026 18:58:34 -0400 Subject: [PATCH 33/52] refactor(skills): make Bearings chat-only by default (#1136) * Add internal status skill * no-mistakes(document): register /status skill in documentation-audiences inventory * no-mistakes(lint): replace grep|wc -l with grep -c in status skill test * test: silence literal status skill patterns * Refactor bearings default to chat-only --------- Co-authored-by: Kun Chen <3233006+kunchenguid@users.noreply.github.com> --- README.md | 9 +- tests/fm-bearings-skill.test.sh | 129 +++++++++++++++++++++++++++++ tests/fm-bearings-snapshot.test.sh | 30 +++++++ 3 files changed, 167 insertions(+), 1 deletion(-) create mode 100755 tests/fm-bearings-skill.test.sh diff --git a/README.md b/README.md index a7f69e39c23..6647ba2c802 100644 --- a/README.md +++ b/README.md @@ -171,10 +171,17 @@ Claude and grok use the slash form shown here; codex uses the same names with `$ | ------------------ | -------------------------------------------------------------------------------------------------------------------------------------------- | | `/afk` | Enter away-mode supervision: the sub-supervisor self-handles routine notifications in bash, escalates captain-relevant events and bounded declared-external-wait rechecks as batched digests, and actively alerts if delivery gets stuck while you step away | | `/ahoy` | Recap visible session events since the prior real captain message plus visibly unanswered captain decisions, falling back to Bearings when invoked as the session's first real captain message | -| `/bearings` | Generate a standalone current-status report from bounded local fleet and registered-secondmate state, with live PR enrichment only when requested, written to a dated file in `data/` and surfaced concisely in chat; read-mostly, mutates no task state | +| `/bearings` | Generate a concise four-section chat digest from bounded local fleet and registered-secondmate state; use `/bearings file` to also replace today's dated report in `data/`, and add `include PRs` when live PR enrichment is wanted | | `/updatefirstmate` | Self-update the running firstmate and its secondmates to the latest from origin with fast-forward-only pulls, then re-read instructions and nudge secondmates | | `/stow` | Sweep the session for uncaptured durable knowledge, route each finding to its disk home per AGENTS.md, file undone next steps to the backlog, and report what is now safe to reset | +Bearings invocation examples: + +- `/bearings` returns the fresh four-section digest in chat only. +- `/bearings include PRs` keeps chat-only mode and opts into live PR enrichment. +- `/bearings file` replaces today's `data/status-report-<YYYY-MM-DD>.md` from scratch and links it from the four-section chat digest. +- `/bearings file include PRs` combines the dated report with live PR enrichment. + Agent-only reference skills live under `.agents/skills/` and are loaded by firstmate at the trigger points named in [`AGENTS.md`](AGENTS.md). ### Two-tier skill layout diff --git a/tests/fm-bearings-skill.test.sh b/tests/fm-bearings-skill.test.sh new file mode 100755 index 00000000000..4fd9d485d3a --- /dev/null +++ b/tests/fm-bearings-skill.test.sh @@ -0,0 +1,129 @@ +#!/usr/bin/env bash +# Static regression tests for the internal /bearings skill contract. +# shellcheck disable=SC2016 +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +BEARINGS_SKILL="$ROOT/.agents/skills/bearings/SKILL.md" +README="$ROOT/README.md" +AUDIENCES="$ROOT/docs/documentation-audiences.json" + +skill_body() { + awk 'BEGIN { seen = 0 } /^---$/ { seen += 1; next } seen >= 2 { print }' "$BEARINGS_SKILL" +} + +chat_contract() { + awk '/^## Chat-response contract$/{capture=1; next} capture && /^## /{exit} capture' "$BEARINGS_SKILL" +} + +test_status_skill_is_absent() { + assert_absent "$ROOT/.agents/skills/status" "internal status skill directory must be removed" + assert_absent "$ROOT/.agents/skills/status/SKILL.md" "internal status skill file must be removed" + assert_absent "$ROOT/skills/status" "public status skill directory must not exist" + assert_no_grep '.agents/skills/status/SKILL.md' "$AUDIENCES" "documentation audience inventory still lists status" + assert_no_grep '| `/status`' "$README" "README still lists /status" + pass "/status is absent from the skill and documentation surfaces" +} + +test_plain_bearings_is_chat_only_by_default() { + assert_grep 'Plain `/bearings` returns only the concise four-section chat digest.' "$BEARINGS_SKILL" \ + "plain bearings default is not chat-only" + assert_grep 'Plain `/bearings` gathers a fresh bounded snapshot and renders the four-section chat digest without creating, deleting, reading, or replacing `data/status-report-<YYYY-MM-DD>.md`.' "$BEARINGS_SKILL" \ + "plain bearings does not forbid dated report writes" + assert_grep 'Plain mode stops here and writes no report artifact.' "$BEARINGS_SKILL" \ + "plain bearings can continue into report writing" + pass "plain /bearings is chat-only and forbids report artifacts" +} + +test_file_mode_owns_prior_report_artifact_behavior() { + assert_grep 'Only `/bearings file` writes the dated markdown report artifact and then returns the concise four-section chat digest linked to that report.' "$BEARINGS_SKILL" \ + "file mode is not the only report-writing mode" + assert_grep 'Write the full report to `data/status-report-<YYYY-MM-DD>.md` using today'"'"'s date.' "$BEARINGS_SKILL" \ + "file mode does not write the dated report" + assert_grep 'If today'"'"'s file already exists, delete it first, then create a new file from scratch.' "$BEARINGS_SKILL" \ + "file mode does not replace today's report from scratch" + assert_grep 'This is the only write allowed by the skill.' "$BEARINGS_SKILL" \ + "file mode write boundary is missing" + assert_grep 'After writing the file, return the concise four-section chat digest and include the report path or link without adding a fifth section.' "$BEARINGS_SKILL" \ + "file mode does not return the linked four-section digest" + pass "/bearings file owns the prior dated-report behavior" +} + +test_file_option_is_explicit_and_prs_compose() { + assert_grep 'Treat `file` only as an explicit invocation option in the slash command.' "$BEARINGS_SKILL" \ + "file is not pinned to an explicit slash option" + assert_grep 'Do not treat natural-language requests such as "write a report", "save this", "persist it", or "make a file" as file mode unless the invocation explicitly includes the standalone `file` option.' "$BEARINGS_SKILL" \ + "file mode can be triggered fuzzily" + assert_grep 'When the captain asks to include PRs, pass the snapshot command'"'"'s live-PR opt-in.' "$BEARINGS_SKILL" \ + "live PR opt-in is missing" + assert_grep '`/bearings include PRs` remains chat-only and makes the live-PR opt-in.' "$BEARINGS_SKILL" \ + "include PRs does not compose with chat-only mode" + assert_grep '`/bearings file include PRs` writes the dated report and makes the live-PR opt-in.' "$BEARINGS_SKILL" \ + "include PRs does not compose with file mode" + pass "file is explicit and live PR enrichment composes with both modes" +} + +test_single_fresh_snapshot_source_and_authoritative_provenance() { + local body count + body=$(skill_body) + count=$(grep -cF 'bin/fm-bearings-snapshot.sh' "$BEARINGS_SKILL") + [ "$count" = 1 ] || fail "bearings should reference the snapshot owner exactly once, found $count" + assert_contains "$body" 'Run `bin/fm-bearings-snapshot.sh` at invocation time and read its compact output.' \ + "bearings does not gather a fresh snapshot at invocation time" + assert_contains "$body" 'It is the single bounded, deterministic fleet-state source for Bearings and renders TOON by default.' \ + "bearings does not name the single bounded source" + assert_contains "$body" 'Do not create or consult a second fleet-state reader, parser contract, status-event-tail interpretation, visible-session recap, ad-hoc project probe, or ad-hoc `gh-axi`/`gh` query.' \ + "bearings allows a second reader or ad-hoc probe" + assert_contains "$body" 'For registered secondmates, use the snapshot'"'"'s structured-home classification and provenance.' \ + "bearings does not use structured secondmate provenance" + assert_contains "$body" 'Structured captain-held decisions come from `decision-hold-lifecycle` and appear under `decisions_open`.' \ + "bearings does not preserve structured captain-held decisions" + assert_not_contains "$body" 'fm-fleet-snapshot.sh' "bearings creates a second canonical snapshot path" + assert_not_contains "$body" 'fm-crew-state.sh' "bearings creates an extra current-state reader" + pass "bearings keeps the fresh structured snapshot as the single source" +} + +test_chat_contract_four_sections_for_both_modes() { + local body headings expected report_headings + body=$(chat_contract) + headings=$(printf '%s\n' "$body" | sed -nE "s/^[0-9]+\. \*\*([^*]+)\*\*.*/\1/p") + expected=$(printf '%s\n' "Captain's Call" "Recently Landed" "Underway" "Charted Next") + [ "$headings" = "$expected" ] || fail "chat contract must contain exactly four numbered sections in fixed order, got: $headings" + assert_contains "$body" "Nothing needs your action right now" "Captain's Call empty-state sentence" + assert_contains "$body" "No recent completions are in the current baseline" "Recently Landed empty-state sentence" + assert_contains "$body" "Nothing is underway" "Underway empty-state sentence" + assert_contains "$body" "Nothing is queued" "Charted Next empty-state sentence" + report_headings=$(sed -nE 's/^ - \*\*(Captain.s Call|Recently Landed|Underway|Charted Next)\*\*.*/\1/p' "$BEARINGS_SKILL") + [ "$report_headings" = "$expected" ] || fail "detailed report contract must contain the same four complete sections, got: $report_headings" + assert_contains "$body" "no At Anchor section" "the At Anchor exclusion must be documented" + assert_contains "$body" "Every chat digest and file-mode report is a complete current snapshot" "both modes must be complete current snapshots" + assert_contains "$body" "Detailed decisions, plans, full gate reasons, and evidence belong in the file only when file mode is explicit" \ + "plain chat must not depend on a detailed report file" + assert_contains "$body" "In file mode, include the report path or link inside the four-section digest without adding another heading." \ + "file mode must link the report without a fifth section" + pass "both Bearings modes keep the exact four-section chat contract" +} + +test_readme_describes_bearings_modes() { + assert_grep '| `/bearings` | Generate a concise four-section chat digest from bounded local fleet and registered-secondmate state; use `/bearings file` to also replace today'"'"'s dated report in `data/`, and add `include PRs` when live PR enrichment is wanted |' "$README" \ + "README skill table does not describe chat-only default and file option" + assert_grep '- `/bearings` returns the fresh four-section digest in chat only.' "$README" \ + "README lacks plain bearings example" + assert_grep '- `/bearings include PRs` keeps chat-only mode and opts into live PR enrichment.' "$README" \ + "README lacks chat-only live PR example" + assert_grep '- `/bearings file` replaces today'"'"'s `data/status-report-<YYYY-MM-DD>.md` from scratch and links it from the four-section chat digest.' "$README" \ + "README lacks file mode example" + assert_grep '- `/bearings file include PRs` combines the dated report with live PR enrichment.' "$README" \ + "README lacks file mode live PR example" + pass "README documents Bearings default and file mode" +} + +test_status_skill_is_absent +test_plain_bearings_is_chat_only_by_default +test_file_mode_owns_prior_report_artifact_behavior +test_file_option_is_explicit_and_prs_compose +test_single_fresh_snapshot_source_and_authoritative_provenance +test_chat_contract_four_sections_for_both_modes +test_readme_describes_bearings_modes diff --git a/tests/fm-bearings-snapshot.test.sh b/tests/fm-bearings-snapshot.test.sh index 31c27a677ae..32c1e1b1e9b 100755 --- a/tests/fm-bearings-snapshot.test.sh +++ b/tests/fm-bearings-snapshot.test.sh @@ -1861,6 +1861,35 @@ EOF pass "main and secondmate captain actionability use the same blocker readiness" } +# The /bearings skill is the one owner of the four-section chat-response contract. +# Assert it states exactly the four fixed sections in order, each with its explicit +# empty-state sentence, documents the At Anchor exclusion, and keeps file-mode links +# inside the four-section digest. +test_chat_contract_four_sections() { + local skill body headings report_headings expected + skill="$ROOT/.agents/skills/bearings/SKILL.md" + [ -f "$skill" ] || fail "bearings SKILL.md missing at $skill" + body=$(awk '/^## Chat-response contract$/{capture=1; next} capture && /^## /{exit} capture' "$skill") + headings=$(printf '%s\n' "$body" | sed -nE "s/^[0-9]+\. \*\*([^*]+)\*\*.*/\1/p") + expected=$(printf '%s\n' "Captain's Call" "Recently Landed" "Underway" "Charted Next") + [ "$headings" = "$expected" ] || fail "chat contract must contain exactly four numbered sections in fixed order, got: $headings" + assert_contains "$body" "Nothing needs your action right now" "Captain's Call empty-state sentence" + assert_contains "$body" "No recent completions are in the current baseline" "Recently Landed empty-state sentence" + assert_contains "$body" "Nothing is underway" "Underway empty-state sentence" + assert_contains "$body" "Nothing is queued" "Charted Next empty-state sentence" + report_headings=$(sed -nE 's/^ - \*\*(Captain.s Call|Recently Landed|Underway|Charted Next)\*\*.*/\1/p' "$skill") + [ "$report_headings" = "$expected" ] || fail "detailed report contract must contain the same four complete sections, got: $report_headings" + grep -Eq 'since the (prior|last) report|Nothing has landed since|unchanged delta' "$skill" \ + && fail "bearings contract still contains prior-report delta wording" + # shellcheck disable=SC2016 # Backticks are literal Markdown in the expected text. + assert_contains "$(cat "$skill")" 'Never read an earlier `data/status-report-*.md`' "prior reports must not influence current output" + assert_contains "$(cat "$skill")" "bounded current recent-completions baseline" "Recently Landed must be a current baseline" + assert_contains "$body" "no At Anchor section" "the At Anchor exclusion must be documented" + assert_contains "$body" "materially shorter" "the file-mode chat must be materially shorter than the report file" + assert_contains "$body" "report path or link" "file mode must link the report from inside the digest" + pass "the /bearings skill states the four-section chat contract in order, with empty-states and the At Anchor exclusion" +} + test_domain_alpha_stale_parent_event_does_not_become_current_work test_gnu_stat_uses_file_formats_without_bsd_fallback_pollution test_parent_activity_evidence_is_bounded_and_disclosed @@ -1891,6 +1920,7 @@ test_main_unstructured_current_is_disclosed_with_structured_sibling test_main_orphan_counterfactual_meta_clears_inventory_warning test_mixed_secondmate_roles_partial_state_and_captain_readiness test_main_captain_readiness_matches_secondmate_projection +test_chat_contract_four_sections test_completed_scout_report_not_pending test_open_decision_surfaces_end_to_end test_report_pointers_surface From 32a588b4884f9e6387c1d55b021d33219671da60 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Wed, 29 Jul 2026 17:06:04 -0700 Subject: [PATCH 34/52] Clarify follow-up routing during validation (#1277) --- AGENTS.md | 1 + 1 file changed, 1 insertion(+) diff --git a/AGENTS.md b/AGENTS.md index 82bcbe8e55a..a558ad3bf37 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -293,6 +293,7 @@ After an autonomous merge, give the captain a one-line full-URL or local-main ou For a no-mistakes ship, trigger validation on the same worker after its implementation commit, using the harness invocation owned by `harness-adapters`. The task worker that starts a no-mistakes run drives the pipeline and owns every `no-mistakes axi run` and `no-mistakes axi respond` call through the next gate or outcome. Firstmate never invokes `no-mistakes axi respond` for a crew-owned run. +Once validation starts, prefer routing new requirements to follow-up work rather than expanding the current task, unless a new requirement completely invalidates the work being validated; corrections required to satisfy already accepted intent are not new requirements. An ask-user finding returns as `needs-decision`; firstmate decides only when the configured authority permits, otherwise escalates to the captain. Send the same worker one exact decision naming the decision key, step, action, affected finding IDs, instructions where needed, and exact response command. From 0a89f136bce6b66225cb319f2962515ea18190fe Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Wed, 29 Jul 2026 17:33:12 -0700 Subject: [PATCH 35/52] fix: honor concrete approval for project operations (#1272) * docs: add captain-approved project operation exception to hard rule 1 Firstmate stays read-only over projects by default, but when the captain clearly approves a concrete project operation and scope in the moment, firstmate may perform exactly that approved operation with its own tools. The approval is never inferred, broadened, or standing, and it does not relax the existing force, discard, unlanded-work, or merge-authority boundaries. * no-mistakes(review): Clarify captain-approved project operation boundaries * no-mistakes(document): Clarify captain-approved project operation scope * docs: cover directories and preserve the operation-or-scope alternative Widen the captain-approved project operation exception in AGENTS.md to files or directories, and restore the explicit operation-or-scope alternative that a prior pipeline auto-fix had collapsed into "and". Rework project-management SKILL.md's Remove section, which previously told firstmate to refuse project removal until a guarded helper existed; that helper was never built, so the text directly contradicted the new instruction-only exception. It now points at the exception plus the existing removal preflight it still requires unchanged. Update the one instruction-owners test assertion that hard-coded the sentence removed above, so the suite tracks current, not obsolete, text. * docs: add captain-approved project operation exception to hard rule 1 Firstmate stays read-only over projects by default, but when the captain clearly approves a concrete project operation and scope in the moment, firstmate may perform exactly that approved operation with its own tools. The approval is never inferred, broadened, or standing, and it does not relax the existing force, discard, unlanded-work, or merge-authority boundaries. * no-mistakes(review): Clarify captain-approved project operation boundaries * no-mistakes(document): Clarify captain-approved project operation scope * docs: cover directories and preserve the operation-or-scope alternative Widen the captain-approved project operation exception in AGENTS.md to files or directories, and restore the explicit operation-or-scope alternative that a prior pipeline auto-fix had collapsed into "and". Rework project-management SKILL.md's Remove section, which previously told firstmate to refuse project removal until a guarded helper existed; that helper was never built, so the text directly contradicted the new instruction-only exception. It now points at the exception plus the existing removal preflight it still requires unchanged. Update the one instruction-owners test assertion that hard-coded the sentence removed above, so the suite tracks current, not obsolete, text. * no-mistakes(review): Align project removal preflight with approved exception * no-mistakes(document): Align project removal documentation with approved exception * fix: restore removal test byte-for-byte and preserve the default sentence tests/fm-instruction-owners.test.sh had been changed to assert different text; restore it byte-for-byte to origin/main. project-management SKILL.md's Remove section now keeps the exact default "Never issue a raw removal command from Firstmate." sentence that test still asserts, immediately followed by the already-approved captain-operation-or-scope exception, so the default and the exception both stay explicit and consistent. * no-mistakes(document): Align project-write boundary documentation --- AGENTS.md | 15 ++++++++------- README.md | 2 +- docs/configuration.md | 2 +- 3 files changed, 10 insertions(+), 9 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index a558ad3bf37..26dffecd5da 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -14,16 +14,17 @@ For captain-facing escalation style and outcome phrasing, see section 9. ## 1. Identity and prime directives You are the captain's only point of contact for all software work across all of their projects. -You do not do project-specific work yourself. -Delegate coding, investigation, planning, bug reproduction, and audits to a crewmate you spawn and supervise, or to a secondmate whose registered scope fits. +Outside hard rule 1's concrete captain-approved project operation exception, you do not do project-specific work yourself. +For all other project-specific work, delegate coding, investigation, planning, bug reproduction, and audits to a crewmate you spawn and supervise, or to a secondmate whose registered scope fits. A secondmate is a crewmate with an isolated firstmate home and a charter, not a second architecture. Hard rules, in priority order: 1. **Never write to a project.** Do not edit, commit, or run state-changing commands under `projects/` or in any project worktree; firstmate reads projects and crewmates change them. - The only exceptions are the guarded project initialization, fleet sync, secondmate sync and inherited local-material propagation, self-update, and approved `local-only` merge paths owned by their referenced skills and scripts. + The only exceptions are the guarded project initialization, fleet sync, secondmate sync and inherited local-material propagation, self-update, and approved `local-only` merge paths, each owned by its referenced skill or script, plus a concrete captain-approved project operation governed directly by this rule. Those paths never authorize forcing, stashing, discarding unlanded work, or hand-writing a project's `AGENTS.md`. + Firstmate may directly edit, create, move, or delete project files or directories only when the captain clearly and concretely approves, in the moment, for a specific project, either a specific operation or a concrete scope whose authorized action needs no inference; firstmate performs exactly that approval with its own file tools, never infers or broadens it, and gains no standing authority, while the force, discard, unlanded-work, merge-authority, destructive, irreversible, and security-sensitive boundaries remain independently in force. 2. **Never merge a PR without the captain's explicit word.** A project's captain-approved `yolo` posture is the only standing relaxation for routine decisions; section 7 owns its exceptions and preserves the stronger destructive, irreversible, and security-sensitive captain boundaries. 3. **Never tear down unlanded work.** @@ -50,7 +51,7 @@ Never add an agent name as a commit co-author. Each secondmate has a persistent isolated `FM_HOME`, including its own state, backlog, projects, and session lock. `bin/fm-send.sh` fails closed unless `FM_HOME` is explicit, so a steer cannot silently resolve against another home. -Tracked files hold shared instructions and tooling; `data/` holds durable private fleet records; `state/` holds volatile runtime records and append-only status events; `config/` holds local operating choices; and `projects/` contains clones that are read-only to firstmate. +Tracked files hold shared instructions and tooling; `data/` holds durable private fleet records; `state/` holds volatile runtime records and append-only status events; `config/` holds local operating choices; and `projects/` contains clones that are read-only to firstmate except under hard rule 1's concrete captain-approved project operation exception. ``` AGENTS.md this file (CLAUDE.md is a symlink to it) @@ -82,7 +83,7 @@ data/ personal fleet records; LOCAL, gitignored as a whole secondmates.md secondmate routing table; firstmate-private, maintained by fm-home-seed.sh (section 6) <id>/brief.md per-task crewmate brief, or per-secondmate charter brief when kind=secondmate <id>/report.md scout task deliverable, written by the crewmate; survives teardown -projects/ cloned repos; gitignored; READ-ONLY for you +projects/ cloned repos; gitignored; read-only except under hard rule 1's concrete captain-approved project operation exception state/ volatile runtime signals; gitignored <id>.status appended by crewmates: "<state>: <note>" wake-event lines, not current-state truth <id>.turn-ended touched by turn-end hooks @@ -196,8 +197,8 @@ A restart must be a non-event because durable state and live backend inventory, ## 6. Project and knowledge management Load `project-management` before adding, creating, removing, or initializing a project. -That skill owns registry syntax, delivery-mode selection, outward-facing consent, clone and initialization procedure, safe rollback, and removal refusal. -Project creation never authorizes an unmentioned remote, and project removal never bypasses the project-write boundary or unlanded-work checks. +That skill owns registry syntax, delivery-mode selection, outward-facing consent, clone and initialization procedure, safe rollback, and removal preflight. +Project creation never authorizes an unmentioned remote, and project removal never bypasses that preflight or unlanded-work checks; hard rule 1's concrete captain-approved project operation exception remains available when its exact conditions are met. Load `secondmate-provisioning` before creating, seeding, validating, launching, handing backlog to, recovering, pushing inherited local material into, or retiring a secondmate home, and before editing `data/secondmates.md`. Its scope field drives routing and its project list is non-exclusive provisioning data, not ownership. diff --git a/README.md b/README.md index 6647ba2c802..5a03c4b9b1b 100644 --- a/README.md +++ b/README.md @@ -49,7 +49,7 @@ Launching a supported harness inside it instantiates your first mate - and makes - **Optional secondmates** - opt in to persistent second mates that run from isolated firstmate homes with their own `FM_HOME`, state, projects, and session lock, supervising project clones or a project-less firstmate-repo domain, kept on the primary firstmate version by guarded local fast-forwards and checked for live agent processes at session start. - **Event-driven, zero-token supervision** - a bash watcher sleeps on the fleet and wakes the first mate only when something needs you; verified primary harnesses also get a turn-end backstop that blocks or follows up on a blind stop when work is under way and supervision is not live. - **Optional X mode** - opt in with one local `.env` token so firstmate can answer your public `@myfirstmate` mentions, act on normal reversible mention requests through the same lifecycle as chat requests, acknowledge spawned work, and post up to three public-safe completion follow-ups within seven days for genuine milestones and the final outcome without changing non-X behavior; dry-run preview records would-be replies and dismissals locally before go-live. -- **Guarded by construction** - the first mate is read-only over your projects except for the guarded paths authorized by [hard rule 1](AGENTS.md#1-identity-and-prime-directives), with fleet sync's safe branch pruning remaining part of the fleet-sync exception; crewmates make every project change behind the configured merge authority. +- **Strict project boundary** - the first mate is read-only over your projects except for the narrow guarded and captain-approved operations authorized by [hard rule 1](AGENTS.md#1-identity-and-prime-directives), including fleet sync's guarded safe branch pruning; crewmates make every other project change behind the configured merge authority. - **Restart-proof** - all state lives on disk and in the active session backend (tmux by hard default, herdr or cmux when selected or auto-detected, zellij/orca when explicitly selected); kill the session anytime and the next one reconciles, including confirmed-dead secondmate agents, and carries on. Full detail on every feature lives in [docs/architecture.md](docs/architecture.md). diff --git a/docs/configuration.md b/docs/configuration.md index adf215b4084..6bf78b7900f 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -12,7 +12,7 @@ This section is the single owner of the top-level operational-home layout; produ The tracked code root contains the shared instruction, skill, documentation, workflow, and `bin/` surfaces, while each effective `FM_HOME` contains private operational directories. `data/` holds durable private fleet records such as the project and secondmate registries, captain preferences, optional shared captain preferences, learnings, backlog, briefs, and scout reports. `state/` holds volatile runtime records such as task metadata, append-only status events, endpoint signals, watcher and wake-queue coordination, away-mode state, generated X-mode artifacts, private secondmate config-reread generations with their retry and quarantine state, and parent-owned secondmate pending-reply records under `state/pending-replies/` (`bin/fm-pending-reply-lib.sh`). -`config/` holds local gitignored operating choices, and `projects/` holds the local project clones that Firstmate reads but changes only through the guarded exceptions in `AGENTS.md`. +`config/` holds local gitignored operating choices, and `projects/` holds the local project clones that Firstmate reads but changes only through the narrow guarded and concrete captain-approved exceptions in `AGENTS.md`. `bin/fm-spawn.sh` owns the base task-metadata fields it emits, while the runtime-backend section below owns backend-specific fields and selector interpretation. The producing PR and X helpers own the fields they append, `bin/fm-classify-lib.sh` owns status-event vocabulary, and `bin/fm-crew-state.sh` owns current-state reconciliation. From 5d939e0317da231b01f1ed45ee46c0e4cb307be7 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Wed, 29 Jul 2026 17:58:01 -0700 Subject: [PATCH 36/52] fix(skills): route new project intake through secondmate scopes (#1275) * Route project intake through secondmate scopes * no-mistakes(test): Guard all main-home project registry mutations * no-mistakes(document): Consolidate secondmate routing documentation * no-mistakes: apply CI fixes * Restore new-project routing scope * no-mistakes(document): Clarify secondmate routing for new-project intake * no-mistakes: apply CI fixes --- .agents/skills/project-management/SKILL.md | 8 +------- AGENTS.md | 2 ++ 2 files changed, 3 insertions(+), 7 deletions(-) diff --git a/.agents/skills/project-management/SKILL.md b/.agents/skills/project-management/SKILL.md index b6211f56b95..f4c62a57901 100644 --- a/.agents/skills/project-management/SKILL.md +++ b/.agents/skills/project-management/SKILL.md @@ -4,7 +4,7 @@ description: >- Agent-only procedure for Firstmate project management. Use before adding, creating, removing, or initializing a project. Cloning or registering a project is add intake and uses the same trigger. - Owns project add, create, clone, remove, initialization, registry, delivery-mode, autonomy, outward-consent decisions, and the secrets-intake handoff. + Owns project add, create, clone, remove, initialization, registry, delivery-mode, autonomy, and outward-consent decisions. user-invocable: false metadata: internal: true @@ -75,12 +75,6 @@ Initialization configures the local gate and does not vendor a no-mistakes skill Do not create a commit merely because initialization ran. If doctor reports an environment, authentication, or daemon problem, resolve that blocker before dispatching work and never restart the shared daemon from a project operation. -Load `secrets-management` during every project intake or initialization. -From the Firstmate root, run `bin/fm-secrets-check.sh inventory projects/<name>` after the gate check. -The inventory is read-only and value-safe, and it is not a substitute for the declared project classification. -If the project lacks `docs/secrets-policy.json`, assign its first ship task to copy and complete `docs/examples/project-secrets-policy.json` before any secret-bearing workflow or deployment change. -Firstmate never hand-writes that project file. - ## Remove Project removal is destructive. diff --git a/AGENTS.md b/AGENTS.md index 26dffecd5da..f61336a5ebe 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -197,6 +197,7 @@ A restart must be a non-event because durable state and live backend inventory, ## 6. Project and knowledge management Load `project-management` before adding, creating, removing, or initializing a project. +Cloning or registering a project is add intake and uses the same trigger. That skill owns registry syntax, delivery-mode selection, outward-facing consent, clone and initialization procedure, safe rollback, and removal preflight. Project creation never authorizes an unmentioned remote, and project removal never bypasses that preflight or unlanded-work checks; hard rule 1's concrete captain-approved project operation exception remains available when its exact conditions are met. @@ -479,6 +480,7 @@ These skills are not captain-invocable; load them only at their precise triggers - `harness-adapters` - load before spawning or recovering a crewmate or secondmate, handling a trust dialog, sending a harness-specific skill invocation, interrupting or exiting an agent, resuming an exited agent, or verifying a new harness adapter. - `firstmate-orca` - load before switching to Orca, spawning or supervising Orca-backed work, smoke-testing Orca backend behavior, debugging Orca task state, or reconciling Orca-backed task metadata. - `project-management` - load before adding, creating, removing, or initializing a project. + Cloning or registering a project is add intake and uses the same trigger. - `stuck-crewmate-recovery` - load when the session-start digest reports an ordinary direct report's endpoint dead or its metadata has no window, or after a stale wake, looping pane, repeated confusion, an answered-by-brief question, an unresponsive crewmate, or a failed steer. - `secondmate-provisioning` - load before creating, seeding, validating, launching, handing backlog to, recovering, pushing inherited local material into, or retiring a secondmate home, and before editing `data/secondmates.md`. - `decision-hold-lifecycle` - load before treating an investigation or visual review as complete, before ending a visual review that exposed a decision, and when recording or routing the captain's answer. From 38c84db04881c596496532666af5363c379f69c0 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Wed, 29 Jul 2026 18:06:41 -0700 Subject: [PATCH 37/52] fix: scope validation corrections by accepted behavior (#1281) * fix: scope validation corrections by accepted behavior * no-mistakes(review): Classify stale delivery evidence as an autonomous correction --- AGENTS.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/AGENTS.md b/AGENTS.md index f61336a5ebe..ee23c840dc9 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -295,7 +295,7 @@ After an autonomous merge, give the captain a one-line full-URL or local-main ou For a no-mistakes ship, trigger validation on the same worker after its implementation commit, using the harness invocation owned by `harness-adapters`. The task worker that starts a no-mistakes run drives the pipeline and owns every `no-mistakes axi run` and `no-mistakes axi respond` call through the next gate or outcome. Firstmate never invokes `no-mistakes axi respond` for a crew-owned run. -Once validation starts, prefer routing new requirements to follow-up work rather than expanding the current task, unless a new requirement completely invalidates the work being validated; corrections required to satisfy already accepted intent are not new requirements. +Once validation starts, prefer routing new requirements to follow-up work rather than expanding the current task, unless a new requirement completely invalidates the work being validated; however, the smallest downstream changes needed to keep already accepted product or engineering behavior correct, add behavioral tests where an executable contract exists, or keep documentation accurate remain within the current task even when they touch files not named at intake, and corrections required to satisfy already accepted intent are not new requirements. An ask-user finding returns as `needs-decision`; firstmate decides only when the configured authority permits, otherwise escalates to the captain. Send the same worker one exact decision naming the decision key, step, action, affected finding IDs, instructions where needed, and exact response command. From da2d2cad4354ba7861a724d28435e5e1c9df9da0 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Wed, 29 Jul 2026 19:17:06 -0700 Subject: [PATCH 38/52] test: replace source assertions with behavioral coverage (#1282) * test: remove source-content assertions * no-mistakes(review): Replace source assertions with runtime behavior coverage * no-mistakes(review): Isolate Kimi task temp runtime coverage * no-mistakes(document): Refresh test cleanup documentation * no-mistakes: apply CI fixes --- .../firstmate-coding-guidelines/SKILL.md | 1 + bin/fm-test-run.sh | 58 ++- docs/fm-test-isolation-proof.json | 210 ++-------- docs/fm-test-isolation-proof.md | 179 +++------ docs/fm-test-portable-shards.md | 133 +++---- docs/sessionstart-nudge.md | 2 - docs/verification/supervision.md | 1 - tests/fm-backend.test.sh | 32 -- tests/fm-bearings-skill.test.sh | 129 ------ tests/fm-bearings-snapshot.test.sh | 30 -- tests/fm-calm-pi-extension.test.sh | 62 --- tests/fm-captain-translation-contract.test.sh | 27 -- tests/fm-claude-stop-autoarm.test.sh | 23 -- tests/fm-documentation-audiences.test.sh | 19 - tests/fm-instruction-owners.test.sh | 305 --------------- tests/fm-kimi-harness.test.sh | 58 +-- tests/fm-lint.test.sh | 84 +--- tests/fm-pi-watch-extension.test.sh | 81 ---- tests/fm-quota-array-dispatch.test.sh | 369 ------------------ tests/fm-subagent-pretool-check.test.sh | 35 +- tests/fm-test-isolation-proof.test.sh | 177 +-------- tests/fm-test-run.test.sh | 124 +----- tests/fm-turnend-guard.test.sh | 89 ----- 23 files changed, 191 insertions(+), 2037 deletions(-) delete mode 100755 tests/fm-bearings-skill.test.sh delete mode 100755 tests/fm-captain-translation-contract.test.sh delete mode 100755 tests/fm-instruction-owners.test.sh delete mode 100755 tests/fm-quota-array-dispatch.test.sh diff --git a/.agents/skills/firstmate-coding-guidelines/SKILL.md b/.agents/skills/firstmate-coding-guidelines/SKILL.md index c7126ff3583..8bbb275dae8 100644 --- a/.agents/skills/firstmate-coding-guidelines/SKILL.md +++ b/.agents/skills/firstmate-coding-guidelines/SKILL.md @@ -97,6 +97,7 @@ Run `bin/fm-doc-audience-check.sh`; it enforces classification, README setup rou - `bin/*.sh` and `bin/backends/*.sh` must pass `shellcheck`. - Run `bin/fm-lint.sh` before treating a script change as done; it is the single owner of the lint definition (file set, config, and pinned shellcheck version) that CI and the no-mistakes pre-push gate both invoke, and it refuses to run under any other shellcheck version. - Colocate tests with the existing pattern in `tests/`, name them `<subject>.test.sh`, and extend an existing script rather than inventing a new runner. +- Tests must exercise behavior through an executable or public interface and must never assert implementation-source bytes, including through parsers, regexes, snapshots, or indirect wrappers. - A maintainer-verification record under `docs/verification/` records active empirical facts, not assumptions or task chronology. - Include the date, version, exact commands run, and exact output needed to support the current guarantee. - Keep incident chronology and delivery evidence in private task reports or PR evidence unless a concise rationale is required to maintain a current safety boundary. diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 255c1cdc31d..314232ae88b 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -38,7 +38,7 @@ # silently pass as a gate skip. # --jobs N run the selected scripts with up to N concurrent workers. # Default is 1 (serial). N>1 is allowed only when every -# selected script is in the Phase 2 proven-isolated set +# selected script is in the proven-isolated set # (bin/fm-test-isolation-proof.sh --list). Cap is 8. Stateful # families never schedule under --jobs. # -h, --help print this header @@ -118,14 +118,13 @@ now_ms() { family_for_basename() { case "$1" in fm-arm-pretool-check.test.sh|fm-ask-user-authority.test.sh|fm-brief.test.sh|\ - fm-calm-pi-extension.test.sh|fm-captain-translation-contract.test.sh|fm-cd-pretool-check.test.sh|\ + fm-calm-pi-extension.test.sh|fm-cd-pretool-check.test.sh|\ fm-composer-ghost.test.sh|fm-composer-lib.test.sh|\ fm-crew-state.test.sh|fm-decision-hold-lifecycle.test.sh|\ fm-documentation-audiences.test.sh|fm-ensure-agents-md.test.sh|fm-grok-harness.test.sh|\ - fm-kimi-harness.test.sh|fm-herdr-lab.test.sh|fm-instruction-owners.test.sh|fm-lint.test.sh|\ - fm-install-herdr.test.sh|fm-nm-test-contract.test.sh|fm-no-mistakes-ownership.test.sh|\ + fm-kimi-harness.test.sh|fm-herdr-lab.test.sh|fm-lint.test.sh|\ fm-operational-input.test.sh|fm-pi-primary-types.test.sh|\ - fm-send-popup-settle.test.sh|fm-send-settle.test.sh|fm-stow-contract.test.sh|\ + fm-send-popup-settle.test.sh|fm-send-settle.test.sh|\ fm-subagent-pretool-check.test.sh|\ fm-supervision-instructions.test.sh|fm-tmux-submit-busy.test.sh|fm-transition-lib.test.sh|\ fm-test-run.test.sh|fm-test-isolation-proof.test.sh) @@ -229,7 +228,7 @@ real-herdr-gated EOF } -# Exact Phase 2 proven-isolated candidate set (same paths as +# Exact proven-isolated candidate set (same paths as # bin/fm-test-isolation-proof.sh --list). Do not expand without a new concurrent # isolation proof archive. list_proven_isolated() { @@ -237,7 +236,6 @@ list_proven_isolated() { tests/fm-arm-pretool-check.test.sh tests/fm-backend-herdr.test.sh tests/fm-brief.test.sh -tests/fm-captain-translation-contract.test.sh tests/fm-cd-pretool-check.test.sh tests/fm-composer-ghost.test.sh tests/fm-composer-lib.test.sh @@ -246,10 +244,7 @@ tests/fm-decision-hold-lifecycle.test.sh tests/fm-ensure-agents-md.test.sh tests/fm-grok-harness.test.sh tests/fm-herdr-lab.test.sh -tests/fm-instruction-owners.test.sh tests/fm-lint.test.sh -tests/fm-nm-test-contract.test.sh -tests/fm-no-mistakes-ownership.test.sh tests/fm-pi-primary-types.test.sh tests/fm-pr-merge.test.sh tests/fm-review-diff.test.sh @@ -257,7 +252,6 @@ tests/fm-send-popup-settle.test.sh tests/fm-send-settle.test.sh tests/fm-send-strict.test.sh tests/fm-spawn-batch.test.sh -tests/fm-stow-contract.test.sh tests/fm-supervision-instructions.test.sh tests/fm-test-run.test.sh tests/fm-tmux-submit-busy.test.sh @@ -266,47 +260,41 @@ tests/fm-x-mode.test.sh EOF } -# Portable parallel shard 1: LPT balance of the proven-isolated set using -# Phase 1 serial duration averages from CI timing artifacts on main after -# #825/#832/#834 (docs/fm-test-portable-shards.md). Execution order is longest -# first so wall-clock stays near the balanced sum. +# Portable parallel shard 1: LPT balance of the proven-isolated set using the +# current concurrent-proof durations in docs/fm-test-isolation-proof.json. +# Execution order is longest first so wall-clock stays near the balanced sum. list_portable_parallel_1() { cat <<'EOF' -tests/fm-arm-pretool-check.test.sh +tests/fm-x-mode.test.sh tests/fm-cd-pretool-check.test.sh -tests/fm-backend-herdr.test.sh -tests/fm-pr-merge.test.sh +tests/fm-decision-hold-lifecycle.test.sh tests/fm-test-run.test.sh -tests/fm-send-popup-settle.test.sh +tests/fm-composer-ghost.test.sh +tests/fm-grok-harness.test.sh +tests/fm-lint.test.sh +tests/fm-pi-primary-types.test.sh tests/fm-review-diff.test.sh tests/fm-brief.test.sh -tests/fm-ensure-agents-md.test.sh -tests/fm-instruction-owners.test.sh -tests/fm-pi-primary-types.test.sh tests/fm-transition-lib.test.sh -tests/fm-composer-lib.test.sh -tests/fm-stow-contract.test.sh EOF } # Portable parallel shard 2: the complementary LPT half of the proven set. list_portable_parallel_2() { cat <<'EOF' -tests/fm-decision-hold-lifecycle.test.sh -tests/fm-x-mode.test.sh -tests/fm-herdr-lab.test.sh +tests/fm-backend-herdr.test.sh +tests/fm-arm-pretool-check.test.sh tests/fm-crew-state.test.sh -tests/fm-grok-harness.test.sh -tests/fm-spawn-batch.test.sh -tests/fm-send-strict.test.sh +tests/fm-herdr-lab.test.sh +tests/fm-pr-merge.test.sh +tests/fm-send-popup-settle.test.sh tests/fm-tmux-submit-busy.test.sh -tests/fm-composer-ghost.test.sh tests/fm-send-settle.test.sh +tests/fm-send-strict.test.sh +tests/fm-spawn-batch.test.sh tests/fm-supervision-instructions.test.sh -tests/fm-lint.test.sh -tests/fm-nm-test-contract.test.sh -tests/fm-captain-translation-contract.test.sh -tests/fm-no-mistakes-ownership.test.sh +tests/fm-ensure-agents-md.test.sh +tests/fm-composer-lib.test.sh EOF } diff --git a/docs/fm-test-isolation-proof.json b/docs/fm-test-isolation-proof.json index 92e227c075f..ec605bf10f2 100644 --- a/docs/fm-test-isolation-proof.json +++ b/docs/fm-test-isolation-proof.json @@ -1,190 +1,36 @@ { "concurrency": 4, - "finished_at": "2026-07-25T08:44:54Z", + "finished_at": "2026-07-29T23:21:46Z", "fm_test_run_jobs_enabled": false, "kind": "isolation-proof", "production_sharding_enabled": false, - "run_id": "fm-isolation-1784968984050-13742", + "run_id": "fm-isolation-1785367157179-18165", "scripts": [ - { - "duration_ms": 26535, - "exit": 0, - "path": "tests/fm-arm-pretool-check.test.sh", - "worker": 1 - }, - { - "duration_ms": 29446, - "exit": 0, - "path": "tests/fm-backend-herdr.test.sh", - "worker": 2 - }, - { - "duration_ms": 973, - "exit": 0, - "path": "tests/fm-brief.test.sh", - "worker": 3 - }, - { - "duration_ms": 181, - "exit": 0, - "path": "tests/fm-captain-translation-contract.test.sh", - "worker": 4 - }, - { - "duration_ms": 17218, - "exit": 0, - "path": "tests/fm-cd-pretool-check.test.sh", - "worker": 5 - }, - { - "duration_ms": 1810, - "exit": 0, - "path": "tests/fm-composer-ghost.test.sh", - "worker": 6 - }, - { - "duration_ms": 66, - "exit": 0, - "path": "tests/fm-composer-lib.test.sh", - "worker": 7 - }, - { - "duration_ms": 15250, - "exit": 0, - "path": "tests/fm-crew-state.test.sh", - "worker": 8 - }, - { - "duration_ms": 18509, - "exit": 0, - "path": "tests/fm-decision-hold-lifecycle.test.sh", - "worker": 9 - }, - { - "duration_ms": 358, - "exit": 0, - "path": "tests/fm-ensure-agents-md.test.sh", - "worker": 10 - }, - { - "duration_ms": 5276, - "exit": 0, - "path": "tests/fm-grok-harness.test.sh", - "worker": 11 - }, - { - "duration_ms": 11199, - "exit": 0, - "path": "tests/fm-herdr-lab.test.sh", - "worker": 12 - }, - { - "duration_ms": 297, - "exit": 0, - "path": "tests/fm-instruction-owners.test.sh", - "worker": 13 - }, - { - "duration_ms": 4882, - "exit": 0, - "path": "tests/fm-lint.test.sh", - "worker": 14 - }, - { - "duration_ms": 180, - "exit": 0, - "path": "tests/fm-nm-test-contract.test.sh", - "worker": 15 - }, - { - "duration_ms": 35, - "exit": 0, - "path": "tests/fm-no-mistakes-ownership.test.sh", - "worker": 16 - }, - { - "duration_ms": 1842, - "exit": 0, - "path": "tests/fm-pi-primary-types.test.sh", - "worker": 17 - }, - { - "duration_ms": 6630, - "exit": 0, - "path": "tests/fm-pr-merge.test.sh", - "worker": 18 - }, - { - "duration_ms": 2410, - "exit": 0, - "path": "tests/fm-review-diff.test.sh", - "worker": 19 - }, - { - "duration_ms": 4496, - "exit": 0, - "path": "tests/fm-send-popup-settle.test.sh", - "worker": 20 - }, - { - "duration_ms": 2179, - "exit": 0, - "path": "tests/fm-send-settle.test.sh", - "worker": 21 - }, - { - "duration_ms": 1390, - "exit": 0, - "path": "tests/fm-send-strict.test.sh", - "worker": 22 - }, - { - "duration_ms": 626, - "exit": 0, - "path": "tests/fm-spawn-batch.test.sh", - "worker": 23 - }, - { - "duration_ms": 52, - "exit": 0, - "path": "tests/fm-stow-contract.test.sh", - "worker": 24 - }, - { - "duration_ms": 336, - "exit": 0, - "path": "tests/fm-supervision-instructions.test.sh", - "worker": 25 - }, - { - "duration_ms": 8900, - "exit": 0, - "path": "tests/fm-test-run.test.sh", - "worker": 26 - }, - { - "duration_ms": 1845, - "exit": 0, - "path": "tests/fm-tmux-submit-busy.test.sh", - "worker": 27 - }, - { - "duration_ms": 96, - "exit": 0, - "path": "tests/fm-transition-lib.test.sh", - "worker": 28 - }, - { - "duration_ms": 34920, - "exit": 0, - "path": "tests/fm-x-mode.test.sh", - "worker": 29 - } + {"duration_ms": 46788, "exit": 0, "path": "tests/fm-arm-pretool-check.test.sh", "worker": 1}, + {"duration_ms": 48294, "exit": 0, "path": "tests/fm-backend-herdr.test.sh", "worker": 2}, + {"duration_ms": 2224, "exit": 0, "path": "tests/fm-brief.test.sh", "worker": 3}, + {"duration_ms": 34207, "exit": 0, "path": "tests/fm-cd-pretool-check.test.sh", "worker": 4}, + {"duration_ms": 9065, "exit": 0, "path": "tests/fm-composer-ghost.test.sh", "worker": 5}, + {"duration_ms": 64, "exit": 0, "path": "tests/fm-composer-lib.test.sh", "worker": 6}, + {"duration_ms": 25365, "exit": 0, "path": "tests/fm-crew-state.test.sh", "worker": 7}, + {"duration_ms": 30771, "exit": 0, "path": "tests/fm-decision-hold-lifecycle.test.sh", "worker": 8}, + {"duration_ms": 581, "exit": 0, "path": "tests/fm-ensure-agents-md.test.sh", "worker": 9}, + {"duration_ms": 6251, "exit": 0, "path": "tests/fm-grok-harness.test.sh", "worker": 10}, + {"duration_ms": 15422, "exit": 0, "path": "tests/fm-herdr-lab.test.sh", "worker": 11}, + {"duration_ms": 5237, "exit": 0, "path": "tests/fm-lint.test.sh", "worker": 12}, + {"duration_ms": 2945, "exit": 0, "path": "tests/fm-pi-primary-types.test.sh", "worker": 13}, + {"duration_ms": 8564, "exit": 0, "path": "tests/fm-pr-merge.test.sh", "worker": 14}, + {"duration_ms": 2875, "exit": 0, "path": "tests/fm-review-diff.test.sh", "worker": 15}, + {"duration_ms": 5644, "exit": 0, "path": "tests/fm-send-popup-settle.test.sh", "worker": 16}, + {"duration_ms": 2911, "exit": 0, "path": "tests/fm-send-settle.test.sh", "worker": 17}, + {"duration_ms": 2747, "exit": 0, "path": "tests/fm-send-strict.test.sh", "worker": 18}, + {"duration_ms": 855, "exit": 0, "path": "tests/fm-spawn-batch.test.sh", "worker": 19}, + {"duration_ms": 703, "exit": 0, "path": "tests/fm-supervision-instructions.test.sh", "worker": 20}, + {"duration_ms": 15674, "exit": 0, "path": "tests/fm-test-run.test.sh", "worker": 21}, + {"duration_ms": 4816, "exit": 0, "path": "tests/fm-tmux-submit-busy.test.sh", "worker": 22}, + {"duration_ms": 248, "exit": 0, "path": "tests/fm-transition-lib.test.sh", "worker": 23}, + {"duration_ms": 52939, "exit": 0, "path": "tests/fm-x-mode.test.sh", "worker": 24} ], - "started_at": "2026-07-25T08:43:04Z", - "summary": { - "duration_ms": 110623, - "failed": 0, - "total": 29 - } + "started_at": "2026-07-29T23:19:17Z", + "summary": {"duration_ms": 149010, "failed": 0, "total": 24} } diff --git a/docs/fm-test-isolation-proof.md b/docs/fm-test-isolation-proof.md index 19e4b6a516c..716dca73a56 100644 --- a/docs/fm-test-isolation-proof.md +++ b/docs/fm-test-isolation-proof.md @@ -1,48 +1,30 @@ -# Firstmate test isolation proof (Phase 2) +# Firstmate test isolation proof -This document is the archived concurrent isolation proof for the portable parallel candidate set. -It is the human-readable companion to `bin/fm-test-isolation-proof.sh`. -Phase 4 production portable shards and bounded local `fm-test-run.sh --jobs` for this exact set are owned by `bin/fm-test-run.sh` and documented in [fm-test-portable-shards.md](fm-test-portable-shards.md). -The archived proof JSON below still records the Phase 2 proof-time flags (`production_sharding_enabled` / `fm_test_run_jobs_enabled` false at proof time). +This record is the concurrent isolation proof for the portable parallel candidate set. +`bin/fm-test-isolation-proof.sh` is the authoritative harness and `docs/fm-test-isolation-proof.json` is the machine-readable result. +`bin/fm-test-run.sh` owns the production lane partition. -## Owner +## Verification -- Harness: `bin/fm-test-isolation-proof.sh` -- Contract tests: `tests/fm-test-isolation-proof.test.sh` -- Family labels (Phase 1): `bin/fm-test-run.sh` -- Timing evidence used for planning: CI artifact `fm-test-timing` from Phase 1 PR #825 - -## Proof posture +- Date: 2026-07-29 +- Command: `bin/fm-test-isolation-proof.sh --jobs 4 --json /tmp/fm-source-content-test-cleanup-r1-isolation.json` +- Result: `FM_ISOLATION_SUMMARY total=24 failed=0 concurrency=4 duration_ms=149010` | Field | Value | |---|---| -| `run_id` | `fm-isolation-1784968984050-13742` | -| `started_at` | `2026-07-25T08:43:04Z` | -| `finished_at` | `2026-07-25T08:44:54Z` | -| concurrency | **4** | -| candidates | **29** | -| failed | **0** | -| wall duration_ms | **110623** (~110.6s) | -| `production_sharding_enabled` | `False` | -| `fm_test_run_jobs_enabled` | `False` | -| host proof date | 2026-07-25 (UTC day of archive write) | - -Isolation checks that passed with this run: - -- Distinct mode-`0700` temporary roots per worker under a proof-owned parent -- Per-worker `TMPDIR`/`TMP` so `mktemp` / `fm_test_tmproot` stay private -- Ambient `FM_HOME` / `FM_*_OVERRIDE` cleared for each worker -- `git config --global` snapshot unchanged before/after the matrix -- Aggregate failure reporting (any non-zero candidate fails the harness; no retry-until-green) +| `run_id` | `fm-isolation-1785367157179-18165` | +| `started_at` | `2026-07-29T23:19:17Z` | +| `finished_at` | `2026-07-29T23:21:46Z` | +| concurrency | 4 | +| candidates | 24 | +| failed | 0 | +| wall duration | 149010 ms | -## Exact candidate set - -Sorted paths as selected by `bin/fm-test-isolation-proof.sh --list` at proof time: +## Candidate set - `tests/fm-arm-pretool-check.test.sh` - `tests/fm-backend-herdr.test.sh` - `tests/fm-brief.test.sh` -- `tests/fm-captain-translation-contract.test.sh` - `tests/fm-cd-pretool-check.test.sh` - `tests/fm-composer-ghost.test.sh` - `tests/fm-composer-lib.test.sh` @@ -51,10 +33,7 @@ Sorted paths as selected by `bin/fm-test-isolation-proof.sh --list` at proof tim - `tests/fm-ensure-agents-md.test.sh` - `tests/fm-grok-harness.test.sh` - `tests/fm-herdr-lab.test.sh` -- `tests/fm-instruction-owners.test.sh` - `tests/fm-lint.test.sh` -- `tests/fm-nm-test-contract.test.sh` -- `tests/fm-no-mistakes-ownership.test.sh` - `tests/fm-pi-primary-types.test.sh` - `tests/fm-pr-merge.test.sh` - `tests/fm-review-diff.test.sh` @@ -62,109 +41,51 @@ Sorted paths as selected by `bin/fm-test-isolation-proof.sh --list` at proof tim - `tests/fm-send-settle.test.sh` - `tests/fm-send-strict.test.sh` - `tests/fm-spawn-batch.test.sh` -- `tests/fm-stow-contract.test.sh` - `tests/fm-supervision-instructions.test.sh` - `tests/fm-test-run.test.sh` - `tests/fm-tmux-submit-busy.test.sh` - `tests/fm-transition-lib.test.sh` - `tests/fm-x-mode.test.sh` -## Per-candidate durations (concurrent run) +## Durations | duration_ms | exit | worker | script | |---:|---:|---:|---| -| 34920 | 0 | 29 | `tests/fm-x-mode.test.sh` | -| 29446 | 0 | 2 | `tests/fm-backend-herdr.test.sh` | -| 26535 | 0 | 1 | `tests/fm-arm-pretool-check.test.sh` | -| 18509 | 0 | 9 | `tests/fm-decision-hold-lifecycle.test.sh` | -| 17218 | 0 | 5 | `tests/fm-cd-pretool-check.test.sh` | -| 15250 | 0 | 8 | `tests/fm-crew-state.test.sh` | -| 11199 | 0 | 12 | `tests/fm-herdr-lab.test.sh` | -| 8900 | 0 | 26 | `tests/fm-test-run.test.sh` | -| 6630 | 0 | 18 | `tests/fm-pr-merge.test.sh` | -| 5276 | 0 | 11 | `tests/fm-grok-harness.test.sh` | -| 4882 | 0 | 14 | `tests/fm-lint.test.sh` | -| 4496 | 0 | 20 | `tests/fm-send-popup-settle.test.sh` | -| 2410 | 0 | 19 | `tests/fm-review-diff.test.sh` | -| 2179 | 0 | 21 | `tests/fm-send-settle.test.sh` | -| 1845 | 0 | 27 | `tests/fm-tmux-submit-busy.test.sh` | -| 1842 | 0 | 17 | `tests/fm-pi-primary-types.test.sh` | -| 1810 | 0 | 6 | `tests/fm-composer-ghost.test.sh` | -| 1390 | 0 | 22 | `tests/fm-send-strict.test.sh` | -| 973 | 0 | 3 | `tests/fm-brief.test.sh` | -| 626 | 0 | 23 | `tests/fm-spawn-batch.test.sh` | -| 358 | 0 | 10 | `tests/fm-ensure-agents-md.test.sh` | -| 336 | 0 | 25 | `tests/fm-supervision-instructions.test.sh` | -| 297 | 0 | 13 | `tests/fm-instruction-owners.test.sh` | -| 181 | 0 | 4 | `tests/fm-captain-translation-contract.test.sh` | -| 180 | 0 | 15 | `tests/fm-nm-test-contract.test.sh` | -| 96 | 0 | 28 | `tests/fm-transition-lib.test.sh` | -| 66 | 0 | 7 | `tests/fm-composer-lib.test.sh` | -| 52 | 0 | 24 | `tests/fm-stow-contract.test.sh` | -| 35 | 0 | 16 | `tests/fm-no-mistakes-ownership.test.sh` | - -## Audit notes (why this set) - -Source families from the Phase 1 manifest and scout report §3.1: - -1. **pure-contract-unit** candidates audited from the Phase 1 family manifest, minus deliberate serial exclusions -2. **Extra hermetic candidates** after static audit: fake backend, private git fixtures, stubbed network - -The harness pins this exact archived set and does not automatically admit later family additions. -A candidate-set change requires a new audit and concurrent proof archive. - -### Included extras (beyond pure-contract-unit) - -| Script | Why included | -|---|---| -| `tests/fm-backend-herdr.test.sh` | Fake Herdr CLI + private temps; no real Herdr binary | -| `tests/fm-send-strict.test.sh` | Fake tmux PATH shim; private `FM_HOME` | -| `tests/fm-spawn-batch.test.sh` | Argument routing only; no real windows/worktrees | -| `tests/fm-pr-merge.test.sh` | Fake `gh`/`gh-axi`; private state | -| `tests/fm-review-diff.test.sh` | Local git fixtures via `fm_git_*`; no live forge | -| `tests/fm-x-mode.test.sh` | Fake `curl`; inert without token | - -### Deliberately serial (kept out of this pool) - -Run `bin/fm-test-isolation-proof.sh --list-exclusions` for the machine-readable list. -High-signal classes: - -| Class | Examples | Reason | -|---|---|---| -| Watcher / wake / locks | `fm-watcher-lock`, `fm-wake-queue`, ... | Intentional process locks and daemon races | -| AFK | `fm-afk-inject-e2e`, ... | Daemon lifecycle and inject path | -| Real Herdr | `fm-backend-herdr-smoke`, presentation e2e, ... | Named labs, session-global locks; Herdr lane is Phase 3+ | -| Real tmux smoke | `fm-backend-tmux-smoke` | Real multiplexer server (even on private socket) | -| Live harness opt-in | `fm-*-live-e2e` | Real interactive agents | -| GUI backends | cmux smoke | Shared GUI app | -| Gray-zone git/spawn | `fm-backend`, spawn settle/profile, teardown | Heavier worktree or lock-race matrices | -| Watcher-adjacent forge security | `fm-pr-check-security` | `.watch.lock` / poll security surface | -| Self | `fm-test-isolation-proof.test.sh` | Must not re-enter the concurrent matrix | - -### Small isolation fix landed with this phase - -`tests/fm-arm-pretool-check.test.sh` no longer writes Claude deny stderr to a fixed `/tmp/fm-arm-pretool-check-claude-stderr.$$` path. -It uses `mktemp` under `TMPDIR` so concurrent workers cannot collide on a global temp name pattern. - -## Failures - -None. -Every candidate exited 0 under concurrency=4. - -Policy: a script that fails only under concurrency is **removed** from the candidate set and investigated. -It is never retried into green, skipped more broadly, or weakened in assertions. - -## What this phase did not do (Phase 2 scope) - -- Did not land production CI Behavior matrix / shard jobs (Phase 4) -- Did not add general `bin/fm-test-run.sh --jobs` (Phase 4 enables it only for this proven set) -- Did not land the Herdr install lane (Phase 3) -- Did not re-run the complete local suite as part of this proof (focused matrix only) - -## How to re-run +| 52939 | 0 | 24 | `tests/fm-x-mode.test.sh` | +| 48294 | 0 | 2 | `tests/fm-backend-herdr.test.sh` | +| 46788 | 0 | 1 | `tests/fm-arm-pretool-check.test.sh` | +| 34207 | 0 | 4 | `tests/fm-cd-pretool-check.test.sh` | +| 30771 | 0 | 8 | `tests/fm-decision-hold-lifecycle.test.sh` | +| 25365 | 0 | 7 | `tests/fm-crew-state.test.sh` | +| 15674 | 0 | 21 | `tests/fm-test-run.test.sh` | +| 15422 | 0 | 11 | `tests/fm-herdr-lab.test.sh` | +| 9065 | 0 | 5 | `tests/fm-composer-ghost.test.sh` | +| 8564 | 0 | 14 | `tests/fm-pr-merge.test.sh` | +| 6251 | 0 | 10 | `tests/fm-grok-harness.test.sh` | +| 5644 | 0 | 16 | `tests/fm-send-popup-settle.test.sh` | +| 5237 | 0 | 12 | `tests/fm-lint.test.sh` | +| 4816 | 0 | 22 | `tests/fm-tmux-submit-busy.test.sh` | +| 2945 | 0 | 13 | `tests/fm-pi-primary-types.test.sh` | +| 2911 | 0 | 17 | `tests/fm-send-settle.test.sh` | +| 2875 | 0 | 15 | `tests/fm-review-diff.test.sh` | +| 2747 | 0 | 18 | `tests/fm-send-strict.test.sh` | +| 2224 | 0 | 3 | `tests/fm-brief.test.sh` | +| 855 | 0 | 19 | `tests/fm-spawn-batch.test.sh` | +| 703 | 0 | 20 | `tests/fm-supervision-instructions.test.sh` | +| 581 | 0 | 9 | `tests/fm-ensure-agents-md.test.sh` | +| 248 | 0 | 23 | `tests/fm-transition-lib.test.sh` | +| 64 | 0 | 6 | `tests/fm-composer-lib.test.sh` | + +## Scope + +Each worker used a separate mode-`0700` temporary root and private `TMPDIR` and `TMP`. +The harness cleared ambient `FM_HOME` and `FM_*_OVERRIDE` values for every worker and verified that global Git configuration was unchanged. +A candidate failure fails the aggregate run and requires investigation rather than a retry. + +## Re-run ```sh bin/fm-test-isolation-proof.sh --list bin/fm-test-isolation-proof.sh --jobs 4 --json /tmp/fm-isolation-proof.json -bash tests/fm-test-isolation-proof.test.sh +bin/fm-test-run.sh --check-coverage ``` diff --git a/docs/fm-test-portable-shards.md b/docs/fm-test-portable-shards.md index ce153cbd74e..0bfa5e6bee4 100644 --- a/docs/fm-test-portable-shards.md +++ b/docs/fm-test-portable-shards.md @@ -1,89 +1,67 @@ -# Firstmate portable test shards (Phase 4) +# Firstmate portable test shards -This document records how the two portable parallel CI shards were balanced from measured evidence. -Composition and execution are owned by `bin/fm-test-run.sh` (`--lane portable-parallel-1` / `portable-parallel-2` / `portable-serial`). -The proven-isolated candidate set remains owned by `bin/fm-test-isolation-proof.sh`. +`bin/fm-test-run.sh` owns portable lane composition and execution. +`bin/fm-test-isolation-proof.sh` owns the proven-isolated candidate set. -## Inputs +## Verification inputs -| Input | Owner / source | -|---|---| -| Proven-isolated set (29 scripts) | `bin/fm-test-isolation-proof.sh --list` and `docs/fm-test-isolation-proof.md` | -| Phase 1 serial durations | CI timing artifacts `fm-test-timing` from main after #825 / #832 / #834 | -| Real-Herdr family | `bin/fm-test-run.sh --family real-herdr-gated` (dedicated required CI lane) | +The current candidate timings came from the 2026-07-29 concurrent proof recorded in [fm-test-isolation-proof.md](fm-test-isolation-proof.md). +The proof ran 24 candidates with four workers and no failures. -Phase 1 averages used for balance (mean of available serial `duration_ms` across those artifacts): - -| duration_ms (avg) | script | +| duration_ms | script | |---:|---| -| 29639 | `tests/fm-arm-pretool-check.test.sh` | -| 25402 | `tests/fm-decision-hold-lifecycle.test.sh` | -| 19428 | `tests/fm-x-mode.test.sh` | -| 14979 | `tests/fm-cd-pretool-check.test.sh` | -| 9339 | `tests/fm-backend-herdr.test.sh` | -| 6885 | `tests/fm-herdr-lab.test.sh` | -| 5127 | `tests/fm-crew-state.test.sh` | -| 4044 | `tests/fm-pr-merge.test.sh` | -| 3922 | `tests/fm-grok-harness.test.sh` | -| 2492 | `tests/fm-test-run.test.sh` | -| 1901 | `tests/fm-send-popup-settle.test.sh` | -| 1234 | `tests/fm-spawn-batch.test.sh` | -| 851 | `tests/fm-send-strict.test.sh` | -| 791 | `tests/fm-review-diff.test.sh` | -| 627 | `tests/fm-tmux-submit-busy.test.sh` | -| 525 | `tests/fm-brief.test.sh` | -| 321 | `tests/fm-composer-ghost.test.sh` | -| 276 | `tests/fm-send-settle.test.sh` | -| 189 | `tests/fm-ensure-agents-md.test.sh` | -| 175 | `tests/fm-supervision-instructions.test.sh` | -| 138 | `tests/fm-instruction-owners.test.sh` | -| 133 | `tests/fm-lint.test.sh` | -| 108 | `tests/fm-pi-primary-types.test.sh` | -| 106 | `tests/fm-nm-test-contract.test.sh` | -| 67 | `tests/fm-transition-lib.test.sh` | -| 64 | `tests/fm-captain-translation-contract.test.sh` | -| 48 | `tests/fm-composer-lib.test.sh` | -| 36 | `tests/fm-stow-contract.test.sh` | -| 28 | `tests/fm-no-mistakes-ownership.test.sh` | - -## Balancing history - -The original 30-script set used longest-processing-time (LPT) assignment onto two workers with the Phase 1 averages above. -The current 29-script lanes retain that assignment after one 283 ms candidate was removed from `portable-parallel-1`. -The current totals are therefore intentionally not a fresh LPT balance of the 29-script set. -Do not rebalance alphabetically or by family intuition. -Shard execution order remains longest-first within each retained lane. - -| Lane | Script count | Sum of Phase 1 averages | +| 52939 | `tests/fm-x-mode.test.sh` | +| 48294 | `tests/fm-backend-herdr.test.sh` | +| 46788 | `tests/fm-arm-pretool-check.test.sh` | +| 34207 | `tests/fm-cd-pretool-check.test.sh` | +| 30771 | `tests/fm-decision-hold-lifecycle.test.sh` | +| 25365 | `tests/fm-crew-state.test.sh` | +| 15674 | `tests/fm-test-run.test.sh` | +| 15422 | `tests/fm-herdr-lab.test.sh` | +| 9065 | `tests/fm-composer-ghost.test.sh` | +| 8564 | `tests/fm-pr-merge.test.sh` | +| 6251 | `tests/fm-grok-harness.test.sh` | +| 5644 | `tests/fm-send-popup-settle.test.sh` | +| 5237 | `tests/fm-lint.test.sh` | +| 4816 | `tests/fm-tmux-submit-busy.test.sh` | +| 2945 | `tests/fm-pi-primary-types.test.sh` | +| 2911 | `tests/fm-send-settle.test.sh` | +| 2875 | `tests/fm-review-diff.test.sh` | +| 2747 | `tests/fm-send-strict.test.sh` | +| 2224 | `tests/fm-brief.test.sh` | +| 855 | `tests/fm-spawn-batch.test.sh` | +| 703 | `tests/fm-supervision-instructions.test.sh` | +| 581 | `tests/fm-ensure-agents-md.test.sh` | +| 248 | `tests/fm-transition-lib.test.sh` | +| 64 | `tests/fm-composer-lib.test.sh` | + +## Parallel lanes + +The two parallel lanes use longest-processing-time assignment from those measured durations. + +| Lane | Script count | Estimated duration | |---|---:|---:| -| `portable-parallel-1` | 14 | 64296 ms (~64.3 s) | -| `portable-parallel-2` | 15 | 64579 ms (~64.6 s) | -| imbalance | | 283 ms | +| `portable-parallel-1` | 11 | 162436 ms (~162.4 s) | +| `portable-parallel-2` | 13 | 162754 ms (~162.8 s) | +| imbalance | | 318 ms | -Exact ordered membership is the heredoc lists in `bin/fm-test-run.sh` (`list_portable_parallel_1` / `list_portable_parallel_2`). +`bin/fm-test-run.sh` contains the exact ordered memberships in `list_portable_parallel_1` and `list_portable_parallel_2`. ## Portable serial remainder -`portable-serial` is every `tests/*.test.sh` that is neither proven-isolated nor `real-herdr-gated`. -That keeps watcher, lock, AFK, real tmux, daemon, secondmate lifecycle, bootstrap, live-harness opt-in (default skip), GUI backends, and other stateful or unproven work serial. -Measured serial remainder wall (from the same Phase 1 artifacts, excluding Herdr) is about **13 minutes**. +`portable-serial` includes every `tests/*.test.sh` that is neither proven-isolated nor `real-herdr-gated`. +It keeps watcher, lock, AFK, real tmux, daemon, secondmate lifecycle, bootstrap, live-harness opt-in, GUI-backend, and other unproven work serial. ## Coverage guard -`bin/fm-test-run.sh --check-coverage` proves: - -1. The two portable parallel shards are a partition of the proven-isolated set. -2. Proven-isolated embeds match `bin/fm-test-isolation-proof.sh --list`. -3. Union of portable parallel shards + portable serial + real-Herdr family equals the complete `tests/*.test.sh` inventory. -4. Those four partitions are pairwise disjoint (no missing scripts, no duplicates). - -CI runs that guard as a required job (`test-coverage`). +`bin/fm-test-run.sh --check-coverage` verifies that both parallel lanes partition the proven-isolated set. +It also verifies that the parallel lanes, portable serial lane, and real-Herdr family are disjoint and cover every `tests/*.test.sh` script. ## Timing artifacts -Every portable shard, the portable serial lane, and the Herdr lane upload their runner-generated timing JSON even when the behavior run reports failures. -The dependent aggregate job runs after all four lanes, combines every available lane JSON through `bin/fm-test-run.sh --aggregate-json`, and uploads one summary artifact for critical-path review. -The workflow in `.github/workflows/ci.yml` owns the exact artifact names and aggregation wiring. +Portable shards, the portable serial lane, and the Herdr lane upload runner-generated timing JSON. +`bin/fm-test-run.sh --aggregate-json` creates the combined summary artifact. +`.github/workflows/ci.yml` owns the exact artifact names and aggregation wiring. ## Local entry points @@ -94,15 +72,8 @@ The workflow in `.github/workflows/ci.yml` owns the exact artifact names and agg | Job | timeout-minutes | Rationale | |---|---:|---| -| portable parallel 1/2 | 10 | Measured shard sum ~1 min; hang tripwire with margin | -| portable serial | 20 | Measured ~13 min remainder; reduced from interim 25m full-portable slack after sharding | -| Herdr | 40 | Unchanged hang tripwire for the real-Herdr lane | - -Timeouts remain hang tripwires, not expected healthy ends of green suites. -Do not raise them as a substitute for green results, retries, or weaker assertions. - -## What this phase does not do +| portable parallel 1/2 | 10 | The measured shard sums are about three minutes and the timeout is a hang tripwire. | +| portable serial | 20 | The serial remainder needs a larger hang tripwire. | +| Herdr | 40 | The real-Herdr lane keeps its dedicated timeout. | -- Does not expand the proven-isolated set without a new concurrent isolation proof. -- Does not parallelize watcher, AFK, real Herdr, real tmux, or other stateful families. -- Does not start rollout verification; that waits until this PR is green and merged. +Timeouts are hang tripwires rather than expected healthy durations. diff --git a/docs/sessionstart-nudge.md b/docs/sessionstart-nudge.md index 7830dfcb3b5..c39c8149259 100644 --- a/docs/sessionstart-nudge.md +++ b/docs/sessionstart-nudge.md @@ -36,8 +36,6 @@ That alternative expands trust and writes outside this repository, so Firstmate `tests/fm-sessionstart-nudge.test.sh` proves wrapper silence for both gate signals, an unmarked linked worktree, a missing state directory, and an already-owned lock. It proves exact U+2063 `FIRSTMATE_OP:`-prefixed, `session-start`-typed one-line output for a plain primary and a marked linked secondmate primary. -It also verifies every tracked transport registration listed above. -`tests/fm-captain-translation-contract.test.sh` proves Ahoy's current marker rule, narrow legacy compatibility exclusions, genuine captain-message near misses, and the shared marker on supported user-role operational injections. `tests/fm-pi-primary-live-e2e.test.sh` and `tests/fm-opencode-primary-live-e2e.test.sh` exercise native startup paths with first-message and later-message Ahoy regressions. `tests/fm-turnend-guard.test.sh`, `tests/fm-pi-watch-extension.test.sh`, and `tests/fm-daemon.test.sh` cover marked guard, monitoring, and away-mode delivery. diff --git a/docs/verification/supervision.md b/docs/verification/supervision.md index a364f8db042..20f415e4564 100644 --- a/docs/verification/supervision.md +++ b/docs/verification/supervision.md @@ -52,7 +52,6 @@ Current deterministic and live entry points: ```sh tests/fm-sessionstart-nudge.test.sh -tests/fm-captain-translation-contract.test.sh FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh FM_OPENCODE_LIVE_E2E=1 tests/fm-opencode-primary-live-e2e.test.sh ``` diff --git a/tests/fm-backend.test.sh b/tests/fm-backend.test.sh index 622bb12839a..7ac873f0878 100755 --- a/tests/fm-backend.test.sh +++ b/tests/fm-backend.test.sh @@ -959,37 +959,6 @@ run_teardown_case() { "$script" "$id" } -test_permissive_tmux_kill_ref_stays_historical() { - local ref body_hist body_head head - head=$(git -C "$ROOT" rev-parse HEAD) - ref=$(resolve_permissive_tmux_kill_ref) \ - || fail "unable to locate a historical bin/backends/tmux.sh with permissive kill-window selectors" - body_hist=$(git -C "$ROOT" show "$ref:bin/backends/tmux.sh") \ - || fail "could not read historical tmux adapter at $ref" - body_head=$(cat "$ROOT/bin/backends/tmux.sh") - - # shellcheck disable=SC2016 - case "$body_hist" in - *'tmux kill-window -t "=$session:=$window"'*) - fail "resolve_permissive_tmux_kill_ref returned exact selectors at $ref" - ;; - esac - # shellcheck disable=SC2016 - case "$body_hist" in - *'tmux kill-window -t "$1"'*|*'tmux kill-window -t "$target"'*) ;; - *) fail "historical tmux adapter at $ref lacks a permissive kill-window target" ;; - esac - # shellcheck disable=SC2016 - case "$body_head" in - *'tmux kill-window -t "=$session:=$window"'*) ;; - *) fail "current tmux adapter lost exact kill-window selectors" ;; - esac - [ "$ref" != "$head" ] \ - || fail "permissive tmux baseline collapsed to HEAD; fixture is no longer historical" - - pass "historical permissive tmux kill baseline stays distinct from current exact selectors" -} - test_teardown_conformance_old_vs_new() { local old_bin fb proj wt id old_tmux_ref saved_base_ref local state_old state_new config_old config_new data log_old log_new out_old out_new rc_old rc_new @@ -1185,7 +1154,6 @@ test_backend_of_selector_matches_explicit_target_meta test_send_conformance_old_vs_new test_peek_conformance_old_vs_new test_spawn_symlinked_project_prefix_avoids_false_refusal -test_permissive_tmux_kill_ref_stays_historical test_teardown_conformance_old_vs_new test_spawn_refuses_unknown_backend_flag test_spawn_refuses_codex_app_backend_flag diff --git a/tests/fm-bearings-skill.test.sh b/tests/fm-bearings-skill.test.sh deleted file mode 100755 index 4fd9d485d3a..00000000000 --- a/tests/fm-bearings-skill.test.sh +++ /dev/null @@ -1,129 +0,0 @@ -#!/usr/bin/env bash -# Static regression tests for the internal /bearings skill contract. -# shellcheck disable=SC2016 -set -u - -# shellcheck source=tests/lib.sh -. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" - -BEARINGS_SKILL="$ROOT/.agents/skills/bearings/SKILL.md" -README="$ROOT/README.md" -AUDIENCES="$ROOT/docs/documentation-audiences.json" - -skill_body() { - awk 'BEGIN { seen = 0 } /^---$/ { seen += 1; next } seen >= 2 { print }' "$BEARINGS_SKILL" -} - -chat_contract() { - awk '/^## Chat-response contract$/{capture=1; next} capture && /^## /{exit} capture' "$BEARINGS_SKILL" -} - -test_status_skill_is_absent() { - assert_absent "$ROOT/.agents/skills/status" "internal status skill directory must be removed" - assert_absent "$ROOT/.agents/skills/status/SKILL.md" "internal status skill file must be removed" - assert_absent "$ROOT/skills/status" "public status skill directory must not exist" - assert_no_grep '.agents/skills/status/SKILL.md' "$AUDIENCES" "documentation audience inventory still lists status" - assert_no_grep '| `/status`' "$README" "README still lists /status" - pass "/status is absent from the skill and documentation surfaces" -} - -test_plain_bearings_is_chat_only_by_default() { - assert_grep 'Plain `/bearings` returns only the concise four-section chat digest.' "$BEARINGS_SKILL" \ - "plain bearings default is not chat-only" - assert_grep 'Plain `/bearings` gathers a fresh bounded snapshot and renders the four-section chat digest without creating, deleting, reading, or replacing `data/status-report-<YYYY-MM-DD>.md`.' "$BEARINGS_SKILL" \ - "plain bearings does not forbid dated report writes" - assert_grep 'Plain mode stops here and writes no report artifact.' "$BEARINGS_SKILL" \ - "plain bearings can continue into report writing" - pass "plain /bearings is chat-only and forbids report artifacts" -} - -test_file_mode_owns_prior_report_artifact_behavior() { - assert_grep 'Only `/bearings file` writes the dated markdown report artifact and then returns the concise four-section chat digest linked to that report.' "$BEARINGS_SKILL" \ - "file mode is not the only report-writing mode" - assert_grep 'Write the full report to `data/status-report-<YYYY-MM-DD>.md` using today'"'"'s date.' "$BEARINGS_SKILL" \ - "file mode does not write the dated report" - assert_grep 'If today'"'"'s file already exists, delete it first, then create a new file from scratch.' "$BEARINGS_SKILL" \ - "file mode does not replace today's report from scratch" - assert_grep 'This is the only write allowed by the skill.' "$BEARINGS_SKILL" \ - "file mode write boundary is missing" - assert_grep 'After writing the file, return the concise four-section chat digest and include the report path or link without adding a fifth section.' "$BEARINGS_SKILL" \ - "file mode does not return the linked four-section digest" - pass "/bearings file owns the prior dated-report behavior" -} - -test_file_option_is_explicit_and_prs_compose() { - assert_grep 'Treat `file` only as an explicit invocation option in the slash command.' "$BEARINGS_SKILL" \ - "file is not pinned to an explicit slash option" - assert_grep 'Do not treat natural-language requests such as "write a report", "save this", "persist it", or "make a file" as file mode unless the invocation explicitly includes the standalone `file` option.' "$BEARINGS_SKILL" \ - "file mode can be triggered fuzzily" - assert_grep 'When the captain asks to include PRs, pass the snapshot command'"'"'s live-PR opt-in.' "$BEARINGS_SKILL" \ - "live PR opt-in is missing" - assert_grep '`/bearings include PRs` remains chat-only and makes the live-PR opt-in.' "$BEARINGS_SKILL" \ - "include PRs does not compose with chat-only mode" - assert_grep '`/bearings file include PRs` writes the dated report and makes the live-PR opt-in.' "$BEARINGS_SKILL" \ - "include PRs does not compose with file mode" - pass "file is explicit and live PR enrichment composes with both modes" -} - -test_single_fresh_snapshot_source_and_authoritative_provenance() { - local body count - body=$(skill_body) - count=$(grep -cF 'bin/fm-bearings-snapshot.sh' "$BEARINGS_SKILL") - [ "$count" = 1 ] || fail "bearings should reference the snapshot owner exactly once, found $count" - assert_contains "$body" 'Run `bin/fm-bearings-snapshot.sh` at invocation time and read its compact output.' \ - "bearings does not gather a fresh snapshot at invocation time" - assert_contains "$body" 'It is the single bounded, deterministic fleet-state source for Bearings and renders TOON by default.' \ - "bearings does not name the single bounded source" - assert_contains "$body" 'Do not create or consult a second fleet-state reader, parser contract, status-event-tail interpretation, visible-session recap, ad-hoc project probe, or ad-hoc `gh-axi`/`gh` query.' \ - "bearings allows a second reader or ad-hoc probe" - assert_contains "$body" 'For registered secondmates, use the snapshot'"'"'s structured-home classification and provenance.' \ - "bearings does not use structured secondmate provenance" - assert_contains "$body" 'Structured captain-held decisions come from `decision-hold-lifecycle` and appear under `decisions_open`.' \ - "bearings does not preserve structured captain-held decisions" - assert_not_contains "$body" 'fm-fleet-snapshot.sh' "bearings creates a second canonical snapshot path" - assert_not_contains "$body" 'fm-crew-state.sh' "bearings creates an extra current-state reader" - pass "bearings keeps the fresh structured snapshot as the single source" -} - -test_chat_contract_four_sections_for_both_modes() { - local body headings expected report_headings - body=$(chat_contract) - headings=$(printf '%s\n' "$body" | sed -nE "s/^[0-9]+\. \*\*([^*]+)\*\*.*/\1/p") - expected=$(printf '%s\n' "Captain's Call" "Recently Landed" "Underway" "Charted Next") - [ "$headings" = "$expected" ] || fail "chat contract must contain exactly four numbered sections in fixed order, got: $headings" - assert_contains "$body" "Nothing needs your action right now" "Captain's Call empty-state sentence" - assert_contains "$body" "No recent completions are in the current baseline" "Recently Landed empty-state sentence" - assert_contains "$body" "Nothing is underway" "Underway empty-state sentence" - assert_contains "$body" "Nothing is queued" "Charted Next empty-state sentence" - report_headings=$(sed -nE 's/^ - \*\*(Captain.s Call|Recently Landed|Underway|Charted Next)\*\*.*/\1/p' "$BEARINGS_SKILL") - [ "$report_headings" = "$expected" ] || fail "detailed report contract must contain the same four complete sections, got: $report_headings" - assert_contains "$body" "no At Anchor section" "the At Anchor exclusion must be documented" - assert_contains "$body" "Every chat digest and file-mode report is a complete current snapshot" "both modes must be complete current snapshots" - assert_contains "$body" "Detailed decisions, plans, full gate reasons, and evidence belong in the file only when file mode is explicit" \ - "plain chat must not depend on a detailed report file" - assert_contains "$body" "In file mode, include the report path or link inside the four-section digest without adding another heading." \ - "file mode must link the report without a fifth section" - pass "both Bearings modes keep the exact four-section chat contract" -} - -test_readme_describes_bearings_modes() { - assert_grep '| `/bearings` | Generate a concise four-section chat digest from bounded local fleet and registered-secondmate state; use `/bearings file` to also replace today'"'"'s dated report in `data/`, and add `include PRs` when live PR enrichment is wanted |' "$README" \ - "README skill table does not describe chat-only default and file option" - assert_grep '- `/bearings` returns the fresh four-section digest in chat only.' "$README" \ - "README lacks plain bearings example" - assert_grep '- `/bearings include PRs` keeps chat-only mode and opts into live PR enrichment.' "$README" \ - "README lacks chat-only live PR example" - assert_grep '- `/bearings file` replaces today'"'"'s `data/status-report-<YYYY-MM-DD>.md` from scratch and links it from the four-section chat digest.' "$README" \ - "README lacks file mode example" - assert_grep '- `/bearings file include PRs` combines the dated report with live PR enrichment.' "$README" \ - "README lacks file mode live PR example" - pass "README documents Bearings default and file mode" -} - -test_status_skill_is_absent -test_plain_bearings_is_chat_only_by_default -test_file_mode_owns_prior_report_artifact_behavior -test_file_option_is_explicit_and_prs_compose -test_single_fresh_snapshot_source_and_authoritative_provenance -test_chat_contract_four_sections_for_both_modes -test_readme_describes_bearings_modes diff --git a/tests/fm-bearings-snapshot.test.sh b/tests/fm-bearings-snapshot.test.sh index 32c1e1b1e9b..31c27a677ae 100755 --- a/tests/fm-bearings-snapshot.test.sh +++ b/tests/fm-bearings-snapshot.test.sh @@ -1861,35 +1861,6 @@ EOF pass "main and secondmate captain actionability use the same blocker readiness" } -# The /bearings skill is the one owner of the four-section chat-response contract. -# Assert it states exactly the four fixed sections in order, each with its explicit -# empty-state sentence, documents the At Anchor exclusion, and keeps file-mode links -# inside the four-section digest. -test_chat_contract_four_sections() { - local skill body headings report_headings expected - skill="$ROOT/.agents/skills/bearings/SKILL.md" - [ -f "$skill" ] || fail "bearings SKILL.md missing at $skill" - body=$(awk '/^## Chat-response contract$/{capture=1; next} capture && /^## /{exit} capture' "$skill") - headings=$(printf '%s\n' "$body" | sed -nE "s/^[0-9]+\. \*\*([^*]+)\*\*.*/\1/p") - expected=$(printf '%s\n' "Captain's Call" "Recently Landed" "Underway" "Charted Next") - [ "$headings" = "$expected" ] || fail "chat contract must contain exactly four numbered sections in fixed order, got: $headings" - assert_contains "$body" "Nothing needs your action right now" "Captain's Call empty-state sentence" - assert_contains "$body" "No recent completions are in the current baseline" "Recently Landed empty-state sentence" - assert_contains "$body" "Nothing is underway" "Underway empty-state sentence" - assert_contains "$body" "Nothing is queued" "Charted Next empty-state sentence" - report_headings=$(sed -nE 's/^ - \*\*(Captain.s Call|Recently Landed|Underway|Charted Next)\*\*.*/\1/p' "$skill") - [ "$report_headings" = "$expected" ] || fail "detailed report contract must contain the same four complete sections, got: $report_headings" - grep -Eq 'since the (prior|last) report|Nothing has landed since|unchanged delta' "$skill" \ - && fail "bearings contract still contains prior-report delta wording" - # shellcheck disable=SC2016 # Backticks are literal Markdown in the expected text. - assert_contains "$(cat "$skill")" 'Never read an earlier `data/status-report-*.md`' "prior reports must not influence current output" - assert_contains "$(cat "$skill")" "bounded current recent-completions baseline" "Recently Landed must be a current baseline" - assert_contains "$body" "no At Anchor section" "the At Anchor exclusion must be documented" - assert_contains "$body" "materially shorter" "the file-mode chat must be materially shorter than the report file" - assert_contains "$body" "report path or link" "file mode must link the report from inside the digest" - pass "the /bearings skill states the four-section chat contract in order, with empty-states and the At Anchor exclusion" -} - test_domain_alpha_stale_parent_event_does_not_become_current_work test_gnu_stat_uses_file_formats_without_bsd_fallback_pollution test_parent_activity_evidence_is_bounded_and_disclosed @@ -1920,7 +1891,6 @@ test_main_unstructured_current_is_disclosed_with_structured_sibling test_main_orphan_counterfactual_meta_clears_inventory_warning test_mixed_secondmate_roles_partial_state_and_captain_readiness test_main_captain_readiness_matches_secondmate_projection -test_chat_contract_four_sections test_completed_scout_report_not_pending test_open_decision_surfaces_end_to_end test_report_pointers_surface diff --git a/tests/fm-calm-pi-extension.test.sh b/tests/fm-calm-pi-extension.test.sh index 46b945e23f7..a8e09be1456 100755 --- a/tests/fm-calm-pi-extension.test.sh +++ b/tests/fm-calm-pi-extension.test.sh @@ -67,67 +67,6 @@ find_chrome() { return 1 } -test_static_contract() { - local text assistant_layout operational_user_layout visibility watch operational - assert_present "$EXT" "tracked Pi calm extension is missing" - assert_present "$ASSISTANT_LAYOUT" "tracked Pi Calm assistant-layout adapter is missing" - assert_present "$OPERATIONAL_USER_LAYOUT" "tracked Pi Calm operational-user layout adapter is missing" - assert_present "$VISIBILITY" "tracked Pi calm visibility policy is missing" - text=$(cat "$EXT") - assistant_layout=$(cat "$ASSISTANT_LAYOUT") - operational_user_layout=$(cat "$OPERATIONAL_USER_LAYOUT") - visibility=$(cat "$VISIBILITY") - watch=$(cat "$WATCH_EXT") - operational=$(cat "$PI_OPERATIONAL_INPUT") - assert_contains "$text" 'pi.registerCommand("calm"' "Pi calm extension does not register /calm" - assert_contains "$text" 'pi.on("session_start"' "Pi calm extension does not restore presentation on every session start" - assert_contains "$text" 'loadCalmPreference()' "Pi calm extension does not restore the home-persistent toggle choice" - assert_contains "$text" 'persistCalmPreference(active)' "Pi calm extension does not persist the captain's toggle choice" - assert_not_contains "$text" 'setCalmPresentation(false)' "Pi calm extension still resets the toggle on session start" - assert_contains "$text" 'ctx.ui.setToolsExpanded(!expanded)' "Pi calm extension does not redraw existing custom entries" - assert_contains "$text" 'ctx.ui.setToolsExpanded(expanded)' "Pi calm extension does not restore Ctrl+O state after redraw" - assert_not_contains "$text" 'ctx.navigateTree' "Pi calm extension reconstructs the transcript and drops transient diagnostics" - assert_not_contains "$visibility" 'deliverFirstmateSyntheticInput' "Pi calm visibility policy can still replace operational input semantics" - assert_not_contains "$visibility" 'classifyFirstmateSyntheticInput' "Pi calm visibility policy still classifies operational input for interception" - assert_contains "$text" 'ctx.ui.setWorkingVisible(true)' "Pi calm extension does not preserve Pi's live working row" - assert_not_contains "$text" 'ctx.ui.setWorkingVisible(!active)' "Pi calm extension still hides Pi's live working row" - assert_contains "$text" 'ctx.ui.setHiddenThinkingLabel(active ? "" : undefined)' "Pi calm extension does not hide collapsed thinking labels" - assert_contains "$text" 'installCalmPresentationAdapter("collapsed-thinking", installCalmAssistantLayout)' "Pi Calm extension does not install its zero-height assistant layout" - assert_contains "$text" 'installCalmPresentationAdapter("operational-user-row", installCalmOperationalUserLayout)' "Pi Calm extension does not install its operational-user layout" - assert_contains "$text" 'function installCalmPresentationAdapter' "Pi Calm extension does not degrade a missing presentation adapter independently with a diagnostic" - assert_contains "$assistant_layout" 'import * as PiCodingAgent' "Pi Calm assistant layout still requires its optional runtime class as a named import" - assert_contains "$assistant_layout" 'AssistantMessageComponent.prototype.updateContent' "Pi Calm assistant layout does not control the exported component presentation path" - assert_contains "$assistant_layout" 'block.type !== "thinking"' "Pi Calm assistant layout does not remove thinking from its presentation copy" - assert_contains "$operational_user_layout" 'import * as PiCodingAgent' "Pi Calm operational-user layout still requires its optional runtime class as a named import" - assert_contains "$operational_user_layout" 'InteractiveMode.prototype' "Pi Calm operational-user layout does not control the transcript owner" - assert_contains "$operational_user_layout" 'classifyFirstmateCurrentOperationalText(text)' "Pi Calm operational-user layout bypasses canonical current classification" - assert_contains "$operational_user_layout" 'text.includes("\u2063")' "Pi Calm operational-user layout spawns its classifier for ordinary captain rows" - assert_contains "$operational_user_layout" '"\u2063Supervisor escalate ("' "Pi Calm operational-user layout lost the narrow legacy marker" - assert_contains "$operational_user_layout" 'hidesOperationalInput()' "Pi Calm operational-user row does not use presentation-only hiding" - assert_not_contains "$operational_user_layout" 'FIRSTMATE_OP: ' "Pi Calm operational-user layout duplicates the canonical marker grammar" - assert_not_contains "$text" 'calm transcript' "Pi calm extension still adds a persistent Calm status row" - assert_not_contains "$text" 'pi.on("input"' "Pi calm extension still intercepts semantic input" - assert_not_contains "$text" 'sendMessage' "Pi calm extension still replaces user-role input with custom context" - assert_contains "$text" 'ctx.ui.onTerminalInput' "Pi calm extension does not scope export rendering to terminal submissions" - assert_contains "$text" 'getKeybindings().matches(data, "tui.input.submit")' "Pi calm export boundary ignores the active submit keybinding" - assert_contains "$text" 'input !== "/share"' "Pi calm export boundary does not cover /share" - assert_not_contains "$text" 'FIRSTMATE_PI_LAUNCH_BRIEF_ENV' "Pi calm presentation still depends on launch-input provenance" - assert_contains "$text" 'renderShell: "self"' "Pi calm extension cannot remove complete built-in tool shells" - assert_contains "$visibility" 'CALM_VISIBLE_CLASSES' "Pi calm policy does not centralize its visibility allowlist" - assert_contains "$operational" 'fm-operational-input.sh' "Pi adapter does not delegate to the canonical cross-language owner" - assert_not_contains "$visibility" 'FIRSTMATE WATCHER WAKE:' "current Calm classification still matches watcher payload prose" - assert_not_contains "$visibility" 'TURN WOULD END BLIND' "current Calm classification still matches turn-end payload prose" - # shellcheck disable=SC2016 # Backticks are literal prompt markup. - assert_not_contains "$visibility" 'Run `bin/fm-session-start.sh`' "current Calm classification still matches session-start payload prose" - assert_not_contains "$visibility" 'FIRSTMATE_OP: ' "current Calm classification duplicates the canonical marker grammar" - assert_contains "$watch" 'calmHides("assistant-tool-call")' "Firstmate watcher tool does not participate in Calm presentation" - assert_contains "$watch" 'renderShell: "self"' "Firstmate watcher tool cannot remove its complete shell" - for name in Read Bash Edit Write Grep Find Ls; do - assert_contains "$text" "create${name}ToolDefinition" "Pi calm extension does not wrap the $name built-in" - done - pass "Pi calm extension is presentation-only with one persisted visibility choice, no Calm status row, native working visibility, supported redraw controls, and the Firstmate watcher-tool integration" -} - test_home_resolution() { local fixture out status version if ! command -v node >/dev/null 2>&1 || ! command -v npm >/dev/null 2>&1; then @@ -2146,7 +2085,6 @@ JS pass "Pi calm native E2E keeps Working and captain turns visible, hides exact operational user rows without changing persistence, restores them Calm-off, survives restart, and preserves export plus Ctrl+O behavior" } -test_static_contract test_home_resolution test_pi_compat_no_upper_bound test_pi_compat_degraded_adapter diff --git a/tests/fm-captain-translation-contract.test.sh b/tests/fm-captain-translation-contract.test.sh deleted file mode 100755 index 142b160dc29..00000000000 --- a/tests/fm-captain-translation-contract.test.sh +++ /dev/null @@ -1,27 +0,0 @@ -#!/usr/bin/env bash -set -u - -# shellcheck source=tests/lib.sh -. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" - -BRIEF="$ROOT/bin/fm-brief.sh" -TMP_ROOT=$(fm_test_tmproot fm-captain-translation) - -test_captain_facing_scout_path_preserves_evidence_and_action() { - local home id report - home="$TMP_ROOT/home" - id="captain-translation" - - FM_HOME="$home" "$BRIEF" "$id" firstmate --scout >/dev/null 2>&1 \ - || fail "scout brief generation failed" - report="$home/data/$id/brief.md" - assert_present "$report" "scout brief was not generated" - assert_grep '# Definition of done' "$report" "scout brief lacks a completion contract" - assert_grep 'what you did, what you found, the evidence' "$report" \ - "scout brief does not require concrete evidence in its report" - assert_grep 'what you recommend' "$report" \ - "scout brief does not require a concrete recommendation" - pass "captain-facing scout work preserves evidence and recommendation handoff" -} - -test_captain_facing_scout_path_preserves_evidence_and_action diff --git a/tests/fm-claude-stop-autoarm.test.sh b/tests/fm-claude-stop-autoarm.test.sh index bcc4fceefb0..6be8bc15333 100755 --- a/tests/fm-claude-stop-autoarm.test.sh +++ b/tests/fm-claude-stop-autoarm.test.sh @@ -150,28 +150,6 @@ epoch_outcome() { # --- registration contract ---------------------------------------------------- -test_settings_registers_autoarm_with_multi_hour_timeout() { - local settings - settings="$ROOT/.claude/settings.json" - jq -e ' - [.hooks.Stop[].hooks[] | select(.command | contains("fm-claude-stop-autoarm.sh"))] - | length == 1 - ' "$settings" >/dev/null || fail "settings must register exactly one Stop auto-arm hook" - jq -e ' - [.hooks.Stop[].hooks[] | select(.command | contains("fm-claude-stop-autoarm.sh"))][0] - | .asyncRewake == true and .type == "command" and (.timeout | type == "number" and . >= 28800) - ' "$settings" >/dev/null || fail "auto-arm must be asyncRewake with an explicit timeout of at least 28800s (the 600s default is forbidden)" - jq -e ' - [.hooks.Stop[].hooks[] | select(.command | contains("fm-claude-stop-autoarm.sh"))][0].command - | contains("&") | not - ' "$settings" >/dev/null || fail "auto-arm registration must not use shell fire-and-forget" - grep -q '"$SCRIPT_DIR/fm-watch-arm.sh" >"$OUT" 2>&1' "$ROOT/bin/fm-claude-stop-autoarm.sh" \ - || fail "auto-arm must foreground the arm wrapper inside the hook-owned process tree" - grep -q 'asyncRewake' "$ROOT/bin/fm-claude-stop-autoarm.sh" \ - || fail "auto-arm header must document its asyncRewake registration contract" - pass "settings.json registers the asyncRewake auto-arm with timeout >= 28800 and a foreground arm" -} - # --- scope and gates ---------------------------------------------------------- test_inert_in_child_worktree() { @@ -438,7 +416,6 @@ test_fm_lock_status_still_works_with_shared_lib() { pass "fm-lock: shared session-lock lib preserves the status path" } -test_settings_registers_autoarm_with_multi_hour_timeout test_inert_in_child_worktree test_inert_without_session_lock test_reclaims_stale_session_lock_before_arming diff --git a/tests/fm-documentation-audiences.test.sh b/tests/fm-documentation-audiences.test.sh index 11854594afe..90222802f6a 100755 --- a/tests/fm-documentation-audiences.test.sh +++ b/tests/fm-documentation-audiences.test.sh @@ -135,26 +135,7 @@ MD pass "local links resolve while dates, versions, commands, and incident prose remain semantically reviewed" } -test_no_mistakes_document_schema() { - local config="$ROOT/.no-mistakes.yaml" - assert_grep 'document:' "$config" "trusted Document config is missing" - assert_grep ' instructions: |' "$config" "Document instructions use an unsupported shape" - assert_grep 'docs/documentation-audiences.json' "$config" \ - "Document instructions do not point to the audience inventory" - assert_grep 'complete' "$config" \ - "Document instructions do not require a complete branch-diff review" - if command -v ruby >/dev/null 2>&1; then - ruby -e ' - require "yaml" - data = YAML.safe_load(File.read(ARGV.fetch(0))) - abort unless data.dig("document", "instructions").is_a?(String) - ' "$config" || fail ".no-mistakes.yaml did not parse document.instructions" - fi - pass "no-mistakes uses the supported trusted document.instructions schema" -} - test_repository_inventory_passes test_duplicate_and_setup_classification_fail test_required_pointer_fails test_local_links_and_no_keyword_heuristic -test_no_mistakes_document_schema diff --git a/tests/fm-instruction-owners.test.sh b/tests/fm-instruction-owners.test.sh deleted file mode 100755 index 754e00ddc83..00000000000 --- a/tests/fm-instruction-owners.test.sh +++ /dev/null @@ -1,305 +0,0 @@ -#!/usr/bin/env bash -# Static contract tests for conditional instruction owners introduced before the -# AGENTS.md reduction pass. -# shellcheck disable=SC2016 -set -u - -# shellcheck source=tests/lib.sh -. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" - -DIAG="$ROOT/.agents/skills/diagnostic-reasoning/SKILL.md" -PROJECT="$ROOT/.agents/skills/project-management/SKILL.md" -HARNESS="$ROOT/.agents/skills/harness-adapters/SKILL.md" -CODING="$ROOT/.agents/skills/firstmate-coding-guidelines/SKILL.md" -RECOVERY="$ROOT/.agents/skills/stuck-crewmate-recovery/SKILL.md" -SECONDMATE="$ROOT/.agents/skills/secondmate-provisioning/SKILL.md" -CONFIG="$ROOT/docs/configuration.md" -AGENTS="$ROOT/AGENTS.md" -BRIEF="$ROOT/bin/fm-brief.sh" -BOOTSTRAP="$ROOT/bin/fm-bootstrap.sh" - -test_new_skill_metadata_and_triggers() { - local skill name count - for pair in "diagnostic-reasoning:$DIAG" "project-management:$PROJECT"; do - name=${pair%%:*} - skill=${pair#*:} - assert_present "$skill" "$name skill is missing" - assert_grep "name: $name" "$skill" "$name skill metadata has the wrong name" - assert_grep "user-invocable: false" "$skill" "$name skill must not be user-invocable" - assert_grep " internal: true" "$skill" "$name skill must be internal" - count=$(grep -Fc -- "- \`$name\` -" "$ROOT/AGENTS.md") - [ "$count" -eq 1 ] || fail "$name must have exactly one AGENTS.md trigger entry, found $count" - done - assert_grep 'Use before scoping a reported bug and before acting on a diagnostic report.' "$DIAG" \ - "diagnostic skill metadata lost its precise load trigger" - assert_grep '`diagnostic-reasoning` - load before scoping a reported bug and before acting on a diagnostic report.' "$ROOT/AGENTS.md" \ - "AGENTS.md lost the diagnostic-reasoning trigger" - assert_grep 'Use before adding, creating, removing, or initializing a project.' "$PROJECT" \ - "project-management skill metadata lost its precise load trigger" - assert_grep '`project-management` - load before adding, creating, removing, or initializing a project.' "$ROOT/AGENTS.md" \ - "AGENTS.md lost the project-management trigger" - pass "new internal skills have one precise AGENTS.md trigger each" -} - -test_diagnostic_owner_covers_causal_procedure() { - assert_grep "single owner of Firstmate's bug-diagnosis reasoning procedure" "$DIAG" \ - "diagnostic skill does not declare ownership" - for phrase in \ - "end-to-end reproduction aligned with the real user path" \ - "initiating trigger" \ - "masking condition" \ - "visible symptom" \ - "proven path" \ - "relevant history" \ - "smallest counterfactual" \ - "disconfirming evidence"; do - assert_grep "$phrase" "$DIAG" "diagnostic owner is missing '$phrase'" - done - assert_grep "evidence, not authorization to change code" "$DIAG" \ - "diagnostic owner lost the diagnosis-only authority boundary" - pass "diagnostic-reasoning owns the approved evidence procedure" -} - -test_project_management_owner_covers_guarded_operations() { - assert_grep "single owner of Firstmate's project-management procedure" "$PROJECT" \ - "project-management skill does not declare ownership" - for phrase in \ - 'bin/fm-project-mode.sh' \ - '`no-mistakes`' \ - '`direct-PR`' \ - '`local-only`' \ - 'Default it off' \ - 'Creating a GitHub repository is outward-facing.' \ - "captain's explicit consent" \ - 'Never issue a raw removal command from Firstmate.' \ - 'no-mistakes init && no-mistakes doctor'; do - assert_grep "$phrase" "$PROJECT" "project-management owner is missing '$phrase'" - done - pass "project-management owns registry, delivery posture, consent, initialization, and removal safety" -} - -test_generic_effort_fallback_respects_precedence() { - local section - section=$(awk ' - /^Effort precedence is / { found = 1 } - found && /^The supported launch-profile flags / { exit } - found { print } - ' "$HARNESS") - assert_contains "$section" "explicit per-task captain instruction first" \ - "effort rubric lost per-task captain precedence" - assert_contains "$section" "standing dispatch profile or secondmate pin" \ - "effort rubric lost standing configuration precedence" - assert_contains "$section" 'Use `low` for well-understood work' \ - "effort rubric lost its low fallback" - assert_contains "$section" '`xhigh` for ambiguous investigation or design' \ - "effort rubric lost its xhigh fallback" - assert_contains "$section" "Choose intermediate levels proportionally" \ - "effort rubric lost proportional intermediate levels" - assert_contains "$section" 'Never select `max` from this fallback' \ - "effort rubric permits max without an explicit captain preference" - if printf '%s\n' "$section" | grep -qi sol; then - fail "generic effort fallback must not contain Sol-specific policy" - fi - pass "generic effort fallback applies only below captain and standing configuration" -} - -test_agent_owned_quota_array_dispatch_contract() { - local phrase - for phrase in \ - 'Firstmate alone resolves a matched profile array' \ - 'run `quota-axi --json` at that intake' \ - 'evaluate every configured candidate against that current output' \ - 'inspectable real headroom including quota-window pace' \ - 'if any harness/model/provider relationship, applicable quota data, or interpretation cannot be established, stop and report that candidate' \ - 'instead of omitting it, guessing, falling back, or calling the result quota-informed' \ - 'Preserve malformed profile configuration as an actionable error' \ - "preserve the captain's strongest-reasoning class rather than silently downgrading it" \ - 'Break genuine headroom ties without array-order or harness bias' \ - '`quota-axi` owns how model or product windows relate to bounding account windows' \ - 'remains data-only' \ - 'Load `quota-array-dispatch` before choosing among a matched profile array'; do - assert_grep "$phrase" "$AGENTS" "array-dispatch contract lost '$phrase'" - done - - for phrase in \ - '| claude | Open the current interactive session' \ - '| codex | Open the current interactive session' \ - '| opencode | Run `opencode models [provider]`' \ - '| pi / pi-signed | Run the selected executable as `<executable> --list-models [search]`' \ - '| grok | Run `grok models`' \ - "For an unfamiliar harness or model namespace, establish support and provider identity from that harness's authoritative CLI help, model listing, or current documentation rather than guessing" \ - 'If those sources do not establish the relationship needed for dispatch, fail loudly and report the unresolved candidate.'; do - assert_grep "$phrase" "$HARNESS" "model discovery guidance lost '$phrase'" - done - assert_grep 'not as a permanent namespace or provider mapping' "$HARNESS" \ - "model discovery guidance permits a fixed provider table" - assert_grep 'load `quota-array-dispatch` for the pace-aware candidate choice' "$HARNESS" \ - "harness-adapters lost the quota-array-dispatch handoff" - assert_grep '`quota-array-dispatch` owns the pace-aware profile-array selection procedure' "$CONFIG" \ - "configuration docs do not point to quota-array-dispatch" - assert_grep 'quota-axi is required for the' "$BOOTSTRAP" \ - "bootstrap docs lost the quota-axi dependency pointer" - assert_grep 'agent-owned dispatch-profile array procedure in AGENTS.md section 4' "$BOOTSTRAP" \ - "bootstrap docs do not point to the agent-owned array procedure" - assert_grep 'quota-array-dispatch/SKILL.md' "$BOOTSTRAP" \ - "bootstrap docs do not point to quota-array-dispatch" - pass "firstmate directly compares every quota candidate with authoritative model discovery" -} - -test_shared_authoring_requirements_are_owned() { - assert_grep "review every affected supported primary harness and runtime backend" "$CODING" \ - "coding guidance lost the supported compatibility matrix review" - assert_grep "prefer deterministic and idempotent enforcement over relying on agent memory alone" "$CODING" \ - "coding guidance lost deterministic idempotent enforcement" - assert_grep "critical safety, routing, startup, and supervision infrastructure" "$CODING" \ - "coding guidance lost the critical infrastructure scope" - pass "firstmate-coding-guidelines owns compatibility review and deterministic enforcement" -} - -test_secondmate_registry_contract_stays_concise() { - local guidance routing_section schema_line - routing_section=$(awk ' - /^## Routing table$/ { found = 1 } - found && /^## Charter and seed$/ { exit } - found { print } - ' "$SECONDMATE") - guidance=$(awk ' - /^## Routing table$/ { found = 1 } - found && /^## Backlog handoff$/ { exit } - found { print } - ' "$SECONDMATE") - schema_line="- <id> - <one-sentence charter summary> (home: <absolute-home-path>; scope: <natural-language responsibility>; projects: <project-a>, <project-b>; added <date>)" - assert_contains "$routing_section" "$schema_line" \ - "secondmate routing table lost the parser-compatible single-line schema" - assert_contains "$routing_section" "Each registry entry stays concise and single-line" \ - "secondmate routing table no longer requires concise single-line entries" - assert_contains "$routing_section" "genuinely domain-specific hard rules" \ - "secondmate routing table no longer limits extra prose to domain-specific hard rules" - assert_contains "$routing_section" "The home-seeded \`data/charter.md\` is the sole owner of boilerplate idle-by-default behavior, the normal delegation lifecycle, and standard escalation contracts" \ - "secondmate routing table lost the explicit charter ownership pointer" - assert_contains "$routing_section" "no extra registry pointer field is needed" \ - "secondmate routing table no longer explains why the existing home field is the charter pointer" - for phrase in \ - "go idle and wait silently" \ - "Act only on tasks" \ - "never spawn a survey" \ - "run normal firstmate bootstrap" \ - "escalation back to the main firstmate status file" \ - "requests-from-main-firstmate contract" \ - "waits for routed tasks, never self-initiating a survey or audit" \ - "marked supervisor requests return through status" \ - "unmarked captain messages stay conversational"; do - if printf '%s\n' "$guidance" | grep -F "$phrase" >/dev/null; then - fail "secondmate provisioning guidance restated charter boilerplate: $phrase" - fi - done - pass "secondmate registry guidance keeps concise routes and points to the charter" -} - -test_state_startup_and_ordinary_recovery_placement() { - assert_grep "single owner of the top-level operational-home layout" "$CONFIG" \ - "configuration docs do not own the operational state layout" - assert_grep "header is the single owner of session-start ordering" "$CONFIG" \ - "session-start mechanism is not assigned to the script header" - assert_grep "Ordinary dead-direct-report recovery is owned by \`stuck-crewmate-recovery\`" "$CONFIG" \ - "D05 ordinary recovery placement is missing" - assert_grep "## Session-start reconciliation for a dead ordinary direct report" "$RECOVERY" \ - "stuck-crewmate-recovery lacks the dead ordinary direct-report procedure" - assert_grep "treehouse status" "$RECOVERY" \ - "ordinary recovery lost treehouse inventory inspection" - assert_grep "recorded \`orca_worktree_id=\` and \`terminal=\`" "$RECOVERY" \ - "ordinary recovery lost Orca inventory inspection" - assert_grep "session-start digest reports an ordinary direct report's endpoint dead or its metadata has no window" "$AGENTS" \ - "AGENTS.md does not trigger ordinary dead-report recovery" - pass "state, startup, and ordinary recovery have focused owners and triggers" -} - -test_compressed_agents_owner_map() { - assert_grep '`docs/configuration.md` is the single owner of the top-level operational-home layout' "$AGENTS" \ - "AGENTS.md lost the state-layout owner pointer" - assert_grep 'header is the single owner of composed commands, ordering, and digest contents' "$AGENTS" \ - "AGENTS.md lost the session-start owner pointer" - assert_grep '`docs/configuration.md` owns dispatch-profile and runtime-backend schemas' "$AGENTS" \ - "AGENTS.md lost the dispatch-schema owner pointer" - assert_grep 'That skill owns registry syntax, delivery-mode selection' "$AGENTS" \ - "AGENTS.md lost the project-management owner pointer" - assert_grep 'The delivery lifecycle is an always-loaded operational contract' "$AGENTS" \ - "AGENTS.md no longer owns the delivery lifecycle" - assert_grep 'Fleet supervision is an always-loaded operational contract' "$AGENTS" \ - "AGENTS.md no longer owns fleet supervision" - assert_grep '`.tasks.toml`, `docs/configuration.md`, and current `tasks-axi --help` own the backlog schema' "$AGENTS" \ - "AGENTS.md lost the backlog-mechanics owner pointer" - assert_grep '`bin/fm-brief.sh` and its help own scaffold syntax' "$AGENTS" \ - "AGENTS.md lost the brief-mechanics owner pointer" - assert_grep '`docs/configuration.md` owns activation, generated state, cadence, wire protocol' "$AGENTS" \ - "AGENTS.md lost the X-mode mechanics owner pointer" - pass "compressed AGENTS.md records the approved one-owner map" -} - -test_intake_reuses_evidence_and_parallelizes_safe_work() { - for phrase in \ - 'consult existing reports and established evidence' \ - 'remaining bounded research inside it' \ - 'unresolved uncertainty could materially change whether or what to build' \ - 'relay it without a design-only scout' \ - 'ask one concise implementation question when useful' \ - 'Never both present a likely-enough solution' \ - 'overlap as a risk signal rather than an automatic reason to wait' \ - 'independently implemented and validated' \ - 'selected delivery path can reconcile ordinary rebases or conflicts' \ - 'Serialize only for a true semantic dependency' \ - 'shared mutable external state' \ - 'incompatible concurrent migration' \ - 'same-file editing alone is insufficient' \ - 'genuine blockers remain durable'; do - assert_grep "$phrase" "$AGENTS" "intake contract lost '$phrase'" - done - assert_grep 'dispatch isolated work immediately with no concurrency cap' "$AGENTS" \ - "intake contract lost unbounded safe parallel dispatch" - assert_grep 'captain explicitly requests a separate knowledge or design deliverable' "$AGENTS" \ - "intake contract lost captain-requested separate scouts" - assert_grep 'When implementation is separately authorized, promote the existing scout' "$AGENTS" \ - "intake contract lost genuine scout promotion" - pass "intake reuses evidence, reserves scouts for uncertainty, and parallelizes safe work" -} - -test_compressed_agents_retains_authority_and_supervision_safety() { - for phrase in \ - 'A lock-refused session must not spawn, steer, merge, drain the wake queue' \ - 'A diagnostic request, report, recommendation, or implementation-ready finding is evidence, not authorization to change code.' \ - 'The selected delivery path owns its own rigor.' \ - 'When no-mistakes is selected, no-mistakes alone owns review, fixes, tests, documentation, push, PR, and CI; otherwise follow the faster path without adding an independent reviewer.' \ - 'Never hold work outside no-mistakes for a manual clean verdict, stack serial manual reviews, or infer authority for one from security, architecture, or risk alone.' \ - 'A separate review or audit is allowed only when the captain explicitly requests that deliverable or the authorized task is a knowledge-only review; one named question remains scoped to that question.' \ - 'If fast-path risk needs more rigor, escalate whether to use no-mistakes instead of inventing a manual gate.' \ - '**local-only** has the worker stop with a clean ready branch, then waits for the configured merge authority' \ - 'A status line is a wake event, not current state' \ - 'keep exactly one live supervision cycle' \ - 'Never broadly kill watchers' \ - 'While `state/.afk` exists, the daemon owns supervision' \ - 'post the final completion follow-up before teardown'; do - assert_grep "$phrase" "$AGENTS" "compressed AGENTS.md lost safety phrase '$phrase'" - done - assert_no_grep 'Firstmate does not personally review code or deliverables' "$AGENTS" \ - "AGENTS.md retained the weaker duplicate review prohibition" - assert_no_grep 'firstmate reviews your branch' "$AGENTS" \ - "AGENTS.md retained a personal branch-review requirement" - assert_no_grep 'firstmate reviews, captain approves' "$BRIEF" \ - "generated brief retained a stacked personal-review requirement" - if grep -q "$(printf '\342\200\224')" "$AGENTS"; then - fail "AGENTS.md contains an em dash" - fi - pass "compressed AGENTS.md retains authority, supervision, AFK, and X safety" -} - -test_new_skill_metadata_and_triggers -test_diagnostic_owner_covers_causal_procedure -test_project_management_owner_covers_guarded_operations -test_generic_effort_fallback_respects_precedence -test_agent_owned_quota_array_dispatch_contract -test_shared_authoring_requirements_are_owned -test_secondmate_registry_contract_stays_concise -test_state_startup_and_ordinary_recovery_placement -test_compressed_agents_owner_map -test_intake_reuses_evidence_and_parallelizes_safe_work -test_compressed_agents_retains_authority_and_supervision_safety diff --git a/tests/fm-kimi-harness.test.sh b/tests/fm-kimi-harness.test.sh index 8ac5922ec51..8e27052d8ce 100755 --- a/tests/fm-kimi-harness.test.sh +++ b/tests/fm-kimi-harness.test.sh @@ -9,33 +9,17 @@ SPAWN="$ROOT/bin/fm-spawn.sh" TEARDOWN="$ROOT/bin/fm-teardown.sh" KIMI_HOOK="$ROOT/bin/fm-kimi-turnend-hook.sh" TMP_ROOT=$(fm_test_tmproot fm-kimi-harness) +KIMI_RUNTIME_TASK_TMP= PYTHON_BIN=$(command -v python3) || fail "test needs python3" PYTHON_BIN_DIR=$(dirname "$PYTHON_BIN") JQ_BIN=$(command -v jq) || fail "test needs jq" BASE_PATH=${FM_TEST_BASE_PATH:-$PYTHON_BIN_DIR:/usr/bin:/bin:/usr/sbin:/sbin} -assert_source_line() { - local line=$1 - grep -Fqx -- "$line" "$SPAWN" || fail "existing launch template changed: $line" -} - -test_existing_launch_templates_are_byte_pinned() { - assert_source_line " claude) printf '%s' 'CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false claude --dangerously-skip-permissions __MODELFLAG____EFFORTFLAG__\"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"' ;;" - assert_source_line " printf '%s' 'codex __MODELFLAG____EFFORTFLAG__--dangerously-bypass-approvals-and-sandbox \"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"'" - assert_source_line " printf '%s' 'codex __MODELFLAG____EFFORTFLAG__--dangerously-bypass-approvals-and-sandbox -c \"notify=[\\\"bash\\\",\\\"-c\\\",\\\"touch __TURNEND__\\\"]\" \"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"'" - assert_source_line " opencode) printf '%s' 'OPENCODE_CONFIG_CONTENT='\\''{\"permission\":{\"*\":\"allow\"}}'\\'' opencode __MODELFLAG__--prompt \"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"' ;;" - assert_source_line " printf '%s%s' \"\$harness\" ' __MODELFLAG____EFFORTFLAG__-e __PITURNEND__ -e __PIWATCH__ \"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"'" - assert_source_line " printf '%s%s' \"\$harness\" ' __MODELFLAG____EFFORTFLAG__-e __PIEXT__ \"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"'" - assert_source_line " grok) printf '%s' 'grok --always-approve __MODELFLAG____EFFORTFLAG__\"\$(__OPINPUT__ encode launch-brief < __BRIEF__)\"' ;;" - pass "fm-spawn: the five pre-existing adapters' launch templates stay byte-pinned" -} - -test_tracked_files_have_no_user_absolute_paths() { - local pattern="/""Users/" matches - matches=$(git -C "$ROOT" grep -n -F "$pattern" -- . || true) - [ -z "$matches" ] || fail "tracked files contain user-specific absolute paths: $matches" - pass "repository: tracked files contain no user-specific absolute paths" +cleanup_kimi_harness() { + [ -z "$KIMI_RUNTIME_TASK_TMP" ] || rm -rf "$KIMI_RUNTIME_TASK_TMP" + rm -rf "$TMP_ROOT" } +trap cleanup_kimi_harness EXIT make_spawn_fakebin() { local dir=$1 fakebin @@ -193,8 +177,11 @@ EOF } test_kimi_launch_then_send_is_verified() { - local id rec out rc launch pointer brief_real meta - id=kimi-success-z1 + local id rec out rc launch pointer brief_real meta task_tmp + id="kimi-success-z1-$$" + task_tmp="/tmp/fm-$id" + KIMI_RUNTIME_TASK_TMP=$task_tmp + rm -rf "$task_tmp" rec=$(make_spawn_case success "$id") read_spawn_record "$rec" out=$(FM_FAKE_KIMI_SWALLOW_FIRST=yes run_spawn \ @@ -218,6 +205,10 @@ test_kimi_launch_then_send_is_verified() { meta="$HOME_DIR/state/$id.meta" assert_grep 'model=kimi-code/k3' "$meta" "kimi meta lost the requested model" assert_grep 'effort=high' "$meta" "kimi meta did not retain the unsupported effort axis" + assert_grep "tasktmp=$task_tmp" "$meta" "kimi meta did not record its task temp root" + assert_present "$task_tmp/gotmp" "kimi spawn did not create its Go temp directory" + assert_grep "export GOTMPDIR=$task_tmp/gotmp" "$CASE_DIR/tmux-calls.log" \ + "kimi spawn did not export its Go temp directory into the pane" assert_grep 'BEGIN FIRSTMATE KIMI TURN-END HOOK' "$HOME_DIR/.kimi-code/config.toml" \ "kimi spawn did not install its guarded global hook region" assert_grep 'token=' "$WT_DIR/.fm-kimi-turnend" "kimi spawn did not write its token pointer" @@ -575,7 +566,7 @@ SH } test_kimi_busy_signature_is_scoped_to_spinner_lines() { - local capture phase kimi_regex_lines + local capture # shellcheck source=/dev/null . "$ROOT/bin/fm-tmux-lib.sh" unset FM_BUSY_REGEX @@ -589,10 +580,11 @@ test_kimi_busy_signature_is_scoped_to_spinner_lines() { # These fixtures reproduce the observed spinner shape rather than byte-exact # transcriptions. Leading whitespace is deliberately varied; separator whitespace # follows the captured contract. - printf ' 🌑 · Tip: ask Kimi to schedule tasks, e.g. "remind me at 5pm"\n│ > │\n' > "$capture" - fm_pane_is_busy fake kimi || fail "the first real Kimi spinner shape was not recognized as busy" - printf ' 🌗 · Tip: /plugins: manage plugins ...\n│ > │\n' > "$capture" - fm_pane_is_busy fake kimi || fail "the tool-execution Kimi spinner shape was not recognized as busy" + local phase + for phase in 🌑 🌒 🌓 🌔 🌕 🌖 🌗 🌘; do + printf ' %s · Tip: Kimi is working\n│ > │\n' "$phase" > "$capture" + fm_pane_is_busy fake kimi || fail "Kimi spinner phase $phase was not recognized as busy" + done printf 'ordinary response ending with 🌕\n│ > │\n' > "$capture" if fm_pane_is_busy fake kimi; then fail "a moon outside Kimi's spinner-line shape was misread as busy" @@ -617,14 +609,6 @@ test_kimi_busy_signature_is_scoped_to_spinner_lines() { if fm_pane_is_busy fake kimi; then fail "Kimi's idle thinking-effort status label was misread as busy" fi - kimi_regex_lines=$(grep 'KIMI_BUSY_REGEX' "$ROOT/bin/fm-tmux-lib.sh" "$ROOT/bin/fm-watch.sh") - if printf '%s\n' "$kimi_regex_lines" | grep -qi thinking; then - fail "Kimi busy regex still depends on a Thinking or thinking token" - fi - for phase in 🌑 🌒 🌓 🌔 🌕 🌖 🌗 🌘; do - grep -Fq "$phase" "$ROOT/bin/fm-tmux-lib.sh" \ - || fail "shared Kimi matcher is missing moon phase $phase" - done pass "busy detection: real Kimi moon-plus-middot captures require its harness while idle labels stay idle" } @@ -673,8 +657,6 @@ test_kimi_bordered_prompt_needs_no_override() { pass "composer classifier: kimi's existing bordered > shape is already safe without an override" } -test_tracked_files_have_no_user_absolute_paths -test_existing_launch_templates_are_byte_pinned test_kimi_hook_install_is_surgical_idempotent_and_removable test_kimi_hook_remove_preserves_owned_newline_boundary test_kimi_hook_fails_closed_on_missing_malformed_or_partial_config diff --git a/tests/fm-lint.test.sh b/tests/fm-lint.test.sh index a2b3c8fb296..17fb097f758 100755 --- a/tests/fm-lint.test.sh +++ b/tests/fm-lint.test.sh @@ -18,11 +18,7 @@ set -u . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" LINT="$ROOT/bin/fm-lint.sh" -CI="$ROOT/.github/workflows/ci.yml" -NM="$ROOT/.no-mistakes.yaml" INSTALLER="$ROOT/bin/fm-install-shellcheck.sh" -# The authoritative file set the one owner must run. -CANON='ROOTS=(bin/*.sh bin/backends/*.sh tests/*.sh)' # The pinned version, read from the single source (the one owner itself). REQUIRED=$("$LINT" --required-version) @@ -33,48 +29,13 @@ pinned_ready() { [ "$(shellcheck --version | awk '/^version:/ {print $2; exit}')" = "$REQUIRED" ] } -test_owner_exists_and_executable() { - assert_present "$LINT" "bin/fm-lint.sh is missing" - [ -x "$LINT" ] || fail "bin/fm-lint.sh must be executable so CI/gate can run it directly" - pass "one-owner lint script exists and is executable" -} - -test_owner_defines_canonical_set() { - assert_grep "$CANON" "$LINT" "fm-lint.sh must run the canonical shellcheck file set" - # It must not weaken CI: no severity downgrade and no blanket disable/exclude - # that would hide findings CI fails on. - assert_no_grep '--severity' "$LINT" "fm-lint.sh must not lower severity below the CI default" - assert_no_grep '--exclude' "$LINT" "fm-lint.sh must not blanket-exclude checks CI enforces" - assert_grep "\"\$FM_LINT_SHELLCHECK\" --norc --external-sources -- \"\${roots[@]}\"" "$LINT" "every bounded worker must ignore ambient config and preserve annotated production sources" - [ "$(grep -Fc -- '--norc --external-sources' "$LINT")" -eq 1 ] || fail "the one worker command must own ShellCheck configuration" - assert_grep "JOBS=\${FM_LINT_JOBS:-2}" "$LINT" "canonical lint must default to two bounded workers" - pass "fm-lint.sh is the sole authoritative definition at CI-default severity" -} - -test_ci_invokes_the_owner() { - grep -Eq '^ - run: bin/fm-lint\.sh$' "$CI" || fail "CI lint job must invoke the one-owner script as a run step" - # Guard against regression to an inline re-spelling of the command. - assert_no_grep 'run: shellcheck' "$CI" "CI must call fm-lint.sh, not re-spell shellcheck inline" - pass "CI lint job calls the one-owner script, not an inline command" -} - -test_stock_bash_parse_uses_owner_inventory() { +test_list_files_reports_the_shell_inventory() { local listed expected listed=$("$LINT" --list-files) expected=$(find bin bin/backends tests -maxdepth 1 -type f -name '*.sh' -print | LC_ALL=C sort) [ "$(printf '%s\n' "$listed" | LC_ALL=C sort)" = "$expected" ] \ - || fail "fm-lint.sh --list-files did not return the complete canonical shell inventory" - # shellcheck disable=SC2016 # Literal assertion must remain unexpanded. - assert_grep 'bin/fm-lint.sh --list-files > "$shell_inventory"' "$CI" \ - "stock macOS Bash parse sweep must consume fm-lint.sh's canonical inventory" - assert_no_grep 'for f in bin/*.sh bin/backends/*.sh tests/*.sh' "$CI" \ - "stock macOS Bash parse sweep must not duplicate the canonical inventory" - pass "stock macOS Bash parse sweep consumes the canonical lint inventory" -} - -test_nomistakes_invokes_the_owner() { - grep -Fqx " lint: 'bin/fm-lint.sh'" "$NM" || fail "no-mistakes commands.lint must map exactly to the one-owner script" - pass "no-mistakes pre-push lint calls the one-owner script" + || fail "fm-lint.sh --list-files did not return the complete shell inventory" + pass "fm-lint.sh --list-files reports the complete shell inventory" } test_pins_an_explicit_version() { @@ -85,17 +46,6 @@ test_pins_an_explicit_version() { pass "fm-lint.sh pins an explicit ShellCheck version ($REQUIRED)" } -test_ci_installs_and_logs_the_pinned_version() { - # CI must derive the version from the one owner (never hardcode a divergent - # number) and log the resolved version as parity evidence. - assert_grep "VERSION=\"\$(\"\$ROOT/bin/fm-lint.sh\" --required-version)\"" "$INSTALLER" "installer must read the version fm-lint.sh pins" - [ "$(grep -Fc "bin/fm-install-shellcheck.sh \"\$RUNNER_TEMP/bin\"" "$CI")" -eq 4 ] || fail "lint and all three portable behavior jobs must use the shared ShellCheck installer" - assert_grep "ACTUAL_SHA256=\$(sha256sum" "$INSTALLER" "installer must calculate the ShellCheck archive checksum" - assert_grep "[ \"\$ACTUAL_SHA256\" = \"\$SHA256\" ]" "$INSTALLER" "installer must verify the ShellCheck archive checksum" - assert_grep "\"\$DESTINATION/shellcheck\" --version" "$INSTALLER" "installer must log the resolved ShellCheck version as evidence" - pass "CI installs and logs the pinned ShellCheck version from the one owner" -} - test_installer_retries_transient_download_failure() { local tmp fakebin destination out tmp=$(fm_test_tmproot fm-shellcheck-download) @@ -252,26 +202,6 @@ SH pass "fm-lint.sh passes a clean fixture" } -test_source_graph_boundaries_keep_every_owner() { - local adapter file production_context_tests="" - [ "$(grep -Fc '# shellcheck source=/dev/null' "$ROOT/bin/fm-backend.sh")" -eq 5 ] \ - || fail "the dispatcher must stop static source following at all five dynamic adapters" - for adapter in tmux herdr zellij orca cmux; do - assert_present "$ROOT/bin/backends/$adapter.sh" "canonical adapter root is missing: $adapter" - done - assert_present "$ROOT/bin/fm-push-transition-lib.sh" "narrow push-transition owner is missing" - assert_grep '# shellcheck source=bin/fm-push-transition-lib.sh' "$ROOT/bin/fm-watch.sh" "the watcher must consume the narrow push-transition owner" - assert_grep ". \"\$ROOT/bin/fm-push-transition-lib.sh\"" "$ROOT/tests/fm-backend-herdr-eventwait-smoke.test.sh" "the Herdr event-wait smoke must consume the narrow production owner" - assert_no_grep '# shellcheck source=bin/fm-watch.sh' "$ROOT/tests/fm-backend-herdr-eventwait-smoke.test.sh" "the event-wait smoke must not re-import the whole watcher graph" - for file in "$ROOT"/tests/*.sh; do - grep -q '^[[:space:]]*# shellcheck source=bin/' "$file" || continue - production_context_tests="${production_context_tests}$(basename "$file")|" - done - [ "$production_context_tests" = 'fm-backend-herdr.test.sh|fm-daemon.test.sh|fm-pending-reply.test.sh|fm-secondmate-sync.test.sh|' ] \ - || fail "only callback/variable interop tests may retain production source context: $production_context_tests" - pass "dispatcher, adapters, production owner, and tests have explicit lint boundaries" -} - test_jobs_are_deterministic_and_complete() { if ! pinned_ready; then pass "SKIP (ShellCheck $REQUIRED not resolved): deterministic bounded jobs check" @@ -496,19 +426,13 @@ SH pass "seeded dispatcher, adapter, production-owner, and test-local diagnostics preserve parity" } -test_owner_exists_and_executable -test_owner_defines_canonical_set -test_ci_invokes_the_owner -test_stock_bash_parse_uses_owner_inventory -test_nomistakes_invokes_the_owner +test_list_files_reports_the_shell_inventory test_pins_an_explicit_version -test_ci_installs_and_logs_the_pinned_version test_installer_retries_transient_download_failure test_rejects_wrong_shellcheck_version test_catches_a_real_lint_defect test_ignores_ambient_shellcheck_opts test_clean_fixture_passes -test_source_graph_boundaries_keep_every_owner test_jobs_are_deterministic_and_complete test_worker_trees_stop_on_signal test_seeded_module_boundary_parity diff --git a/tests/fm-pi-watch-extension.test.sh b/tests/fm-pi-watch-extension.test.sh index 518df0e874e..f8883194898 100755 --- a/tests/fm-pi-watch-extension.test.sh +++ b/tests/fm-pi-watch-extension.test.sh @@ -59,64 +59,6 @@ export const Type = { JS } -test_tracked_extension_present_and_self_hashing() { - local text expected_config_source - expected_config_source="config_dir=\\\"\${FM_CONFIG_OVERRIDE:-\$FM_HOME/config}\\\"" - assert_present "$EXT" "tracked Pi primary watcher extension is missing" - text=$(cat "$EXT") - assert_contains "$text" "fm_watch_arm_pi" "tracked extension missing tool name" - assert_contains "$text" "fm-watch-arm-pi" "tracked extension missing command name" - assert_contains "$text" "fm-watch-arm.sh" "tracked extension missing watcher arm" - assert_contains "$text" "sendUserMessage" "tracked extension missing Pi wake API" - assert_contains "$text" 'encodeFirstmateOperationalInput' "tracked extension does not construct typed synthetic user-role wakes" - assert_contains "$text" "deliverAs: \"followUp\"" "tracked extension missing followUp delivery" - assert_contains "$text" ".pi-watch-extension-loaded" "tracked extension missing loaded marker" - assert_contains "$text" 'createHash("sha256").update(readFileSync(extensionFile)).digest("hex")' "tracked extension does not self-hash its own content for extensionVersion" - assert_contains "$text" 'fileURLToPath(import.meta.url)' "tracked extension does not self-locate via import.meta.url" - assert_contains "$text" 'type LockOwnership = "owned" | "missing" | "other"' "tracked extension does not distinguish missing lock from another owner" - assert_contains "$text" "readFileSync(\`\${state}/.lock\`" "tracked extension does not read the effective session lock" - assert_contains "$text" 'return pidAlive(lockPid) ? "other" : "missing"' "tracked extension does not allow a pre-lock load marker" - assert_contains "$text" 'if (lockOwnership() === "other") return' "tracked extension overwrites another live session marker" - assert_contains "$text" 'const ownership = lockOwnership()' "tracked extension arm does not inspect the distinct lock ownership state" - assert_contains "$text" 'if (ownership === "other") return { ok: false' "tracked extension arm does not preserve the live-other read-only refusal" - assert_contains "$text" 'if (ownership === "missing")' "tracked extension arm collapses a stale or absent lock into the live-other refusal" - assert_contains "$text" "no live session holds the lock" "tracked extension arm missing stale-lock recovery guidance" - assert_contains "$text" "run bin/fm-session-start.sh to reclaim it" "tracked extension arm does not direct stale-lock reclamation" - assert_contains "$text" "call fm_watch_arm_pi to re-arm" "tracked extension arm does not direct supervision re-arm" - assert_contains "$text" "writeFileSync(marker, \`\${extensionVersion}\\n\${process.pid}\\n\`)" "tracked extension does not write the content version and process marker" - assert_contains "$text" "const config = process.env.FM_CONFIG_OVERRIDE" "tracked extension missing effective config resolution" - assert_contains "$text" "FM_CONFIG_OVERRIDE: config" "tracked extension does not pass the effective config to the watcher arm" - assert_contains "$text" "FM_WATCH_ARM_SCRIPT: armScript" "tracked extension does not pass the effective watcher arm script" - assert_contains "$text" "$expected_config_source" "tracked extension does not source the effective x-mode config" - assert_contains "$text" "exec \\\"\$FM_WATCH_ARM_SCRIPT\\\" --restart" "tracked extension does not restart into a Pi-owned watcher child" - assert_contains "$text" 'label: "Arm firstmate watcher"' "tracked extension tool is missing its human-readable label" - assert_not_contains "$text" "Always use this tool" "tracked extension kept broad tool-selection guidance" - assert_contains "$text" "only for the first required cycle or after a notification says the cycle is missing, failed, or unhealthy" "tracked extension tool metadata is missing the Pi first-cycle or explicit-repair rule" - assert_contains "$text" "Do not call it after ordinary work, turn completion, or ordinary signal, stale, check, or heartbeat handling" "tracked extension prompt guidance does not prevent redundant ordinary-notification calls" - assert_contains "$text" 'parameters: Type.Object({})' "tracked extension tool is not using Pi's canonical TypeBox schema" - assert_contains "$text" 'content: [{ type: "text", text: result.message }]' "tracked extension tool is missing Pi text content" - assert_contains "$text" 'details: result' "tracked extension tool is missing structured result details" - assert_contains "$text" 'ctx.ui.notify' "tracked extension command does not notify through Pi's UI" - assert_contains "$text" 'process.once("exit", cleanupOnProcessExit)' "tracked extension lacks clean-process-exit cleanup" - assert_contains "$text" "type SessionGeneration" "tracked extension lacks an explicit session-generation owner" - assert_contains "$text" "function activateGeneration" "tracked extension does not activate a live generation for replacement sessions" - assert_contains "$text" "function generationIsLive" "tracked extension does not gate arm mutations on the live generation" - assert_contains "$text" "watcher: not armed - Pi session is shutting down" "tracked extension missing the terminal shutdown refusal" - assert_not_contains "$text" "[ -f config/x-mode.env ]" "tracked extension kept a repo-relative x-mode config path" - pass "Pi primary watcher extension is tracked, self-hashing, and self-locating" -} - -test_spawn_template_mentions_pi_watch_placeholder() { - local text - text=$(cat "$ROOT/bin/fm-spawn.sh") - assert_contains "$text" "-e __PITURNEND__ -e __PIWATCH__" "Pi secondmate launch template does not include both primary extensions" - assert_contains "$text" "\$PROJ_ABS/.pi/extensions/fm-primary-pi-watch.ts" "fm-spawn does not point the Pi secondmate watch placeholder at the tracked extension" - assert_not_contains "$text" "fm-pi-watch-extension.sh" "fm-spawn should no longer generate the Pi watch extension before launch" - assert_contains "$text" "__PITURNEND__" "fm-spawn does not replace the Pi turn-end guard extension placeholder" - assert_contains "$text" "__PIWATCH__" "fm-spawn does not replace the Pi watch extension placeholder" - pass "Pi secondmate launch wiring includes both tracked primary extensions" -} - test_pi_extension_reports_external_healthy_watcher() { local repo home plugin out status repo="$TMP_ROOT/pi-external-healthy-root" @@ -1234,26 +1176,6 @@ EOF pass "Pi process-exit cleanup stops the attached arm child" } -test_opencode_primary_watch_plugin_static_wiring() { - local plugin module_boundary text - plugin="$ROOT/.opencode/plugins/fm-primary-watch-arm.js" - module_boundary="$ROOT/.opencode/plugins/package.json" - assert_present "$plugin" "OpenCode primary watch plugin missing" - assert_present "$module_boundary" "OpenCode plugin ESM package boundary missing" - assert_contains "$(cat "$module_boundary")" '"type": "module"' "OpenCode plugin package boundary is not explicitly ESM" - text=$(cat "$plugin") - assert_contains "$text" "session.idle" "OpenCode plugin does not listen for session.idle" - assert_contains "$text" "fm-watch-arm.sh" "OpenCode plugin does not spawn the watcher arm" - assert_contains "$text" "promptAsync" "OpenCode plugin does not wake with promptAsync" - assert_contains "$text" 'encodeFirstmateOperationalInput' "OpenCode plugin does not construct typed synthetic user-role wakes" - assert_contains "$text" ".fm-secondmate-home" "OpenCode plugin does not scope out secondmate homes" - assert_contains "$text" "rev-parse\", \"--git-dir" "OpenCode plugin does not check linked worktree scope" - assert_contains "$text" "sessionOwnsLock" "OpenCode plugin does not gate arm attempts on the session lock" - assert_contains "$text" 'fm-watch-arm.sh" --restart' "OpenCode plugin does not restart into its own watcher child" - assert_contains "$text" 'setArmStatus("external")' "OpenCode plugin still treats an external healthy watcher as armed" - pass "OpenCode primary watcher plugin has the verified TUI wake wiring" -} - test_opencode_plugin_package_boundary_is_explicit_esm() { local fixture plugin out status fixture="$TMP_ROOT/opencode-esm-boundary/.opencode" @@ -2202,8 +2124,6 @@ EOF pass "OpenCode healthy arm output does not suppress the turn-end guard" } -test_tracked_extension_present_and_self_hashing -test_spawn_template_mentions_pi_watch_placeholder test_pi_extension_reports_external_healthy_watcher test_pi_tool_returns_agent_tool_result test_pi_redundant_tool_call_is_owned_noop @@ -2219,7 +2139,6 @@ test_pi_arm_distinguishes_session_lock_ownership test_pi_session_transition_generation_owner test_pi_process_exit_cleanup_listener_lifecycle test_pi_process_exit_cleanup_stops_arm_child -test_opencode_primary_watch_plugin_static_wiring test_opencode_plugin_package_boundary_is_explicit_esm test_opencode_primary_watch_plugin_uses_effective_state_home test_opencode_primary_watch_plugin_sources_effective_config diff --git a/tests/fm-quota-array-dispatch.test.sh b/tests/fm-quota-array-dispatch.test.sh deleted file mode 100755 index 4fb1328ba3d..00000000000 --- a/tests/fm-quota-array-dispatch.test.sh +++ /dev/null @@ -1,369 +0,0 @@ -#!/usr/bin/env bash -# Contract and deterministic fixture tests for quota-array-dispatch. -# -# The skill owns the agent-facing decision procedure. -# This test encodes the same inspectable comparison rules against sanitized -# fixtures so acceptance cases stay deterministic without introducing a -# production routing wrapper. -# shellcheck disable=SC2016 -set -u - -# shellcheck source=tests/lib.sh -. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" - -AGENTS="$ROOT/AGENTS.md" -OWNER="$ROOT/.agents/skills/quota-array-dispatch/SKILL.md" -HARNESS="$ROOT/.agents/skills/harness-adapters/SKILL.md" -CONFIG="$ROOT/docs/configuration.md" -ARCHITECTURE="$ROOT/docs/architecture.md" -BOOTSTRAP="$ROOT/bin/fm-bootstrap.sh" -AUDIENCES="$ROOT/docs/documentation-audiences.json" -CASES="$ROOT/tests/fixtures/quota-array-dispatch/cases.json" -SHAPE="$ROOT/tests/fixtures/quota-array-dispatch/schema-v3-shape.json" - -intake_boundary() { - awk ' - /^## 4\. Harness and runtime dispatch$/ { found = 1; next } - found && /^## 5\. Recovery$/ { exit } - found { print } - ' "$AGENTS" -} - -select_candidate_py() { - python3 - "$@" <<'PY' -import json, sys - -def conservation_pressure(c): - if not c.get("paceAvailable", True): - return False - status = c.get("paceStatus") - ahead_ids = c.get("aheadWindowIds") or [] - bounding_windows = c.get("boundingWindows") or [] - if status == "ahead": - return True - if status == "mixed" and ahead_ids: - return True - if any(window.get("paceStatus") == "ahead" for window in bounding_windows): - return True - return False - -def select(case): - required = case.get("requiredReasoningClass") - cands = list(case["candidates"]) - if required: - matching = [c for c in cands if c.get("reasoningClass") == required] - if not matching: - return {"error": "required reasoning class unavailable"} - # Strongest-reasoning rule: never drop to a weaker class for quota. - cands = matching - - # Fit filter: fixtures mark comparable; keep only comparable for these cases. - cands = [c for c in cands if c.get("fit") == "comparable"] - if not cands: - return {"error": "no comparable candidates"} - - def sort_key(c): - pressured = conservation_pressure(c) - unknown = bool(c.get("unknownPace")) or c.get("paceStatus") == "unknown" - pace_available = bool(c.get("paceAvailable", True)) - reserve = c.get("worstReserve") - if reserve is None: - reserve_key = float("-inf") - else: - reserve_key = float(reserve) - raw = float(c.get("rawHeadroom") or 0) - # Sort ascending by preference rank components that python min understands - # via a tuple where lower is better only for pressure/unknown flags. - return ( - 1 if pressured else 0, - 1 if (unknown and pace_available) else 0, - 0 if pace_available else 1, # when pace absent, still comparable via raw only - # Among pressured: least-negative reserve => higher reserve first => negate - (-reserve_key if pressured else 0), - # Among sustainable with pace: prefer higher reserve then higher raw - (-reserve_key if (not pressured and pace_available and not unknown) else 0), - -raw, - ) - - # Special-case all-tight already constrained to required class above. - best_key = min(sort_key(c) for c in cands) - winners = [c for c in cands if sort_key(c) == best_key] - if len(winners) > 1: - return { - "error": "genuine tie requires captain choice", - "candidates": sorted(c["id"] for c in winners), - } - winner = winners[0] - if winner.get("authAvailable") is False: - return { - "error": "selected candidate authentication unavailable", - "harness": winner["harness"], - "model": winner["model"], - "authenticationSurface": winner["authenticationSurface"], - "failureEvidence": winner["authFailure"], - } - return { - "id": winner["id"], - "pressured": conservation_pressure(winner), - } - -case = json.loads(sys.argv[1]) -print(json.dumps(select(case))) -PY -} - -test_owner_and_always_loaded_boundary() { - local boundary trigger_count - boundary=$(intake_boundary) - - assert_present "$OWNER" "quota-array-dispatch owner is missing" - assert_grep 'name: quota-array-dispatch' "$OWNER" "quota-array-dispatch skill has the wrong name" - assert_grep 'user-invocable: false' "$OWNER" "quota-array-dispatch skill must be agent-only" - assert_grep 'single owner of the pace-aware profile-array selection procedure' "$OWNER" \ - "quota-array-dispatch skill does not declare ownership" - - assert_contains "$boundary" 'Firstmate alone resolves a matched profile array' \ - "intake boundary lost agent-owned array resolution" - assert_contains "$boundary" 'run `quota-axi --json` at that intake' \ - "intake boundary lost quota-axi intake read" - assert_contains "$boundary" 'evaluate every configured candidate against that current output' \ - "intake boundary lost full-candidate accounting" - assert_contains "$boundary" 'inspectable real headroom including quota-window pace' \ - "intake boundary lost pace-aware headroom wording" - assert_contains "$boundary" 'if any harness/model/provider relationship, applicable quota data, or interpretation cannot be established, stop and report that candidate' \ - "intake boundary lost unresolved-candidate refusal" - assert_contains "$boundary" 'instead of omitting it, guessing, falling back, or calling the result quota-informed' \ - "intake boundary lost no-guess wording" - assert_contains "$boundary" 'Preserve malformed profile configuration as an actionable error' \ - "intake boundary lost malformed-config refusal" - assert_contains "$boundary" "preserve the captain's strongest-reasoning class rather than silently downgrading it" \ - "intake boundary lost strongest-reasoning rule" - assert_contains "$boundary" 'Break genuine headroom ties without array-order or harness bias' \ - "intake boundary lost genuine-tie rule" - assert_contains "$boundary" '`quota-axi` owns how model or product windows relate to bounding account windows' \ - "intake boundary lost quota-axi window ownership" - assert_contains "$boundary" 'remains data-only' \ - "intake boundary lost data-only producer boundary" - assert_contains "$boundary" 'Load `quota-array-dispatch` before choosing among a matched profile array' \ - "intake boundary lost quota-array-dispatch load trigger" - - trigger_count=$(grep -Fc -- '- `quota-array-dispatch` -' "$AGENTS") - [ "$trigger_count" -eq 1 ] || fail "quota-array-dispatch must have exactly one section 13 trigger, found $trigger_count" - - # Full pace procedure stays out of AGENTS.md. - if printf '%s\n' "$boundary" | grep -q 'reservePercentPoints'; then - fail "AGENTS.md intake boundary duplicated pace formula detail" - fi - if printf '%s\n' "$boundary" | grep -q 'aheadWindowIds'; then - fail "AGENTS.md intake boundary duplicated aheadWindowIds detail" - fi - - pass "quota-array-dispatch has one conditional owner and a concise always-loaded boundary" -} - -test_owner_contains_selection_procedure() { - local phrase lines words bytes - for phrase in \ - 'reservePercentPoints = percentRemaining - timeRemainingPercent' \ - 'Negative reserve means usage is ahead of reset pace and creates conservation pressure' \ - 'Positive reserve means usage is behind reset pace' \ - '`on_pace` is neutral' \ - 'effective pace status is `mixed` and any `aheadWindowIds` remain' \ - 'Comparable fit/reasoning: prefer no ahead pressure over pressure' \ - 'even with higher raw headroom' \ - 'prefer the least-negative worst applicable reserve' \ - 'Sustainable candidates: use known pace plus raw headroom' \ - 'Do not collapse those facts into an opaque composite score' \ - '`unknown` is valid explicit uncertainty from quota-axi' \ - 'Prefer known sustainable evidence over `unknown` when comparable' \ - 'If unresolved pace changes the choice, report uncertainty' \ - 'do not crash, fabricate pace, or treat absence as healthy' \ - 'stop and report every tied candidate for captain choice' \ - 'Do not select by array order, harness name, or another arbitrary identity ordering' \ - 'Do not add a daemon, opaque composite score, routing wrapper, hard-coded model-specific policy' \ - 'Report duplicate concrete profiles as a configuration error' \ - 'Unresolved relationship or quota: stop and report the tuple and concrete evidence' \ - 'After selecting, check auth only through that tuple'\''s surface' \ - 'Name the inspectable facts used for every candidate'; do - assert_grep "$phrase" "$OWNER" "quota-array-dispatch procedure lost '$phrase'" - done - - # Expanded acceptance scenarios live in deterministic fixtures, not runtime prose. - for phrase in \ - 'Higher raw quota but materially ahead vs lower raw quota on/behind pace' \ - 'Sanitized producer shape' \ - '## When to load' \ - '## Intake boundary this skill does not relax'; do - if grep -Fq -- "$phrase" "$OWNER"; then - fail "quota-array-dispatch should not keep removed runtime prose: $phrase" - fi - done - - lines=$(wc -l < "$OWNER" | tr -d ' ') - words=$(wc -w < "$OWNER" | tr -d ' ') - bytes=$(wc -c < "$OWNER" | tr -d ' ') - [ "$lines" -le 65 ] || fail "quota-array-dispatch skill is too long: $lines lines (want <= 65)" - [ "$words" -le 550 ] || fail "quota-array-dispatch skill is too wordy: $words words (want <= 550)" - [ "$bytes" -le 4600 ] || fail "quota-array-dispatch skill is too large: $bytes bytes (want <= 4600)" - pass "quota-array-dispatch owns the compact pace procedure ($lines lines, $words words, $bytes bytes)" -} - -test_cross_references_stay_pointers() { - assert_grep '`quota-array-dispatch` owns the pace-aware profile-array selection procedure' "$CONFIG" \ - "configuration docs do not point to quota-array-dispatch" - assert_no_grep '`AGENTS.md` section 4 owns the dispatch and array-selection procedure.' "$CONFIG" \ - "configuration docs still claim AGENTS.md owns the full array-selection procedure" - assert_grep 'quota-array-dispatch' "$ARCHITECTURE" \ - "architecture docs lost the quota-array-dispatch pointer" - assert_grep 'quota-array-dispatch' "$BOOTSTRAP" \ - "bootstrap header lost the quota-array-dispatch pointer" - assert_grep 'load `quota-array-dispatch` for the pace-aware candidate choice' "$HARNESS" \ - "harness-adapters lost the array-selection handoff" - assert_grep '.agents/skills/quota-array-dispatch/SKILL.md' "$AUDIENCES" \ - "documentation audience inventory missing quota-array-dispatch" - pass "cross-references point at the single procedure owner" -} - -test_schema_v3_shape_fixture() { - python3 - "$SHAPE" <<'PY' || fail "schema v3 shape fixture is invalid" -import json, sys -path = sys.argv[1] -data = json.load(open(path)) -assert data.get("schemaVersion") == 3, data.get("schemaVersion") -assert isinstance(data.get("providers"), list) and data["providers"], "providers" -provider = data["providers"][0] -assert "windows" in provider and provider["windows"], "windows" -window = provider["windows"][0] -assert "pace" in window and "status" in window["pace"], window -eff = provider["quotaSemantics"]["effectiveAvailability"][0] -assert "pace" in eff and "status" in eff["pace"], eff -assert "effectivePercentRemaining" in eff -# Privacy: no live account residue markers. -blob = json.dumps(data) -for bad in ("sk-", "@", "Bearer ", "accountId", "organizationId"): - assert bad not in blob, bad -PY - pass "sanitized schemaVersion 3 fixture preserves producer pace shape without private details" -} - -test_deterministic_acceptance_cases() { - local raw case_json case_id expect expect_error got reason - raw=$(cat "$CASES") - while IFS= read -r case_json; do - case_id=$(python3 -c 'import json,sys; print(json.loads(sys.argv[1])["id"])' "$case_json") - expect=$(python3 -c 'import json,sys; print(json.loads(sys.argv[1]).get("expect", ""))' "$case_json") - expect_error=$(python3 -c 'import json,sys; print(json.loads(sys.argv[1]).get("expectError", ""))' "$case_json") - reason=$(python3 -c 'import json,sys; print(json.loads(sys.argv[1])["reason"])' "$case_json") - got=$(select_candidate_py "$case_json") - python3 -c ' -import json,sys -got=json.loads(sys.argv[1]) -expect=sys.argv[2] -expect_error=sys.argv[3] -case_id=sys.argv[4] -err=got.get("error") -if expect_error: - if err != expect_error: - raise SystemExit("%s: expected error %s, got %s" % (case_id, expect_error, got)) -elif err: - raise SystemExit("%s: selector error: %s" % (case_id, err)) -elif got.get("id") != expect: - raise SystemExit("%s: expected %s, got %s" % (case_id, expect, got)) -' "$got" "$expect" "$expect_error" "$case_id" \ - || fail "case $case_id failed ($reason); selector returned $got" - if [ -n "$expect_error" ]; then - pass "case $case_id -> $expect_error ($reason)" - else - pass "case $case_id -> $expect ($reason)" - fi - done < <(python3 -c 'import json,sys; data=json.load(sys.stdin); [print(json.dumps(c, separators=(",", ":"))) for c in data["cases"]]' <<<"$raw") -} - -test_dispatch_identity_and_blocked_report() { - local reports - assert_grep '`harness-adapters` owns identity' "$OWNER" \ - "quota-array-dispatch does not point to the adapter identity owner" - assert_grep "After selecting, check auth only through that tuple's surface; another harness CLI cannot block it" "$OWNER" \ - "quota-array-dispatch does not scope evidence to the concrete tuple" - assert_no_grep 'Unresolved relationship, auth, or quota' "$OWNER" \ - "quota-array-dispatch checks authentication before selecting a candidate" - assert_grep 'A blocked credential report must name `harness`, `model`, authentication surface, and concrete failure evidence' "$OWNER" \ - "blocked reports do not preserve the minimum identity and evidence fields" - assert_grep 'The concrete `harness` field owns adapter identity independently of the model provider' "$HARNESS" \ - "harness-adapters lost the anti-conflation owner paragraph" - assert_grep '`harness=pi` with `model=xai/grok-*` is Pi using xAI, not `harness=grok`' "$HARNESS" \ - "harness-adapters lost the concrete Pi/xAI versus Grok distinction" - assert_grep 'does not require Grok CLI login' "$HARNESS" \ - "Pi/xAI guidance incorrectly requires Grok CLI login" - assert_no_grep '### Dispatch identity mapping' "$HARNESS" \ - "identity guidance grew a separate table instead of one owner paragraph" - - reports=$(python3 - <<'PY' - -def auth_surface(harness, model, provider): - surfaces = { - ("pi", "xai/grok-4.5", "xai"): "Pi xAI OAuth", - ("grok", "grok-4.5", "grok"): "Grok Build CLI", - } - try: - return surfaces[(harness, model, provider)] - except KeyError: - raise AssertionError("unresolved concrete profile") from None - -def evaluate(harness, model, provider, auth_available, failure): - surface = auth_surface(harness, model, provider) - if auth_available: - return (f"ready: harness={harness} model={model} provider={provider} " - f"authentication surface checked={surface}") - return (f"blocked: harness={harness} model={model} provider={provider} " - f"authentication surface checked={surface} failure evidence={failure}") - -pi = evaluate("pi", "xai/grok-4.5", "xai", True, None) -grok = evaluate("grok", "grok-4.5", "grok", False, "Grok Build CLI login missing") -assert "Grok Build CLI" not in pi -assert "Pi xAI OAuth" in pi -assert "Grok Build CLI" in grok -for mismatched in ( - ("grok", "xai/grok-4.5", "xai"), - ("pi", "grok-4.5", "grok"), -): - try: - auth_surface(*mismatched) - except AssertionError: - pass - else: - raise AssertionError(f"accepted mismatched profile: {mismatched}") -print(pi) -print(grok) -PY -) || fail "identity counterfactual fixture failed" - assert_contains "$reports" 'ready: harness=pi model=xai/grok-4.5 provider=xai authentication surface checked=Pi xAI OAuth' \ - "Pi/xAI did not remain dispatchable with its own authentication" - assert_not_contains "$reports" 'harness=pi model=xai/grok-4.5 provider=xai authentication surface checked=Grok Build CLI' \ - "Pi/xAI was reported with the standalone Grok CLI surface" - assert_contains "$reports" 'blocked: harness=grok model=grok-4.5 provider=grok authentication surface checked=Grok Build CLI' \ - "explicit Grok candidate did not use the Grok Build CLI surface" - for field in harness=grok model=grok-4.5 provider=grok 'authentication surface checked=Grok Build CLI' 'failure evidence=Grok Build CLI login missing'; do - assert_contains "$reports" "$field" "Grok blocked report lost '$field'" - done - printf '%s\n' "$reports" - pass "dispatch identity stays concrete across the Pi/xAI versus Grok counterfactual" -} - -test_no_duplicate_procedure_in_agents() { - # Guard against re-expanding the full procedure into AGENTS.md. - local count - count=$(grep -c 'conservation pressure' "$AGENTS" || true) - [ "$count" -eq 0 ] || fail "AGENTS.md should not restate conservation-pressure procedure detail" - count=$(grep -c 'worst applicable reserve' "$AGENTS" || true) - [ "$count" -eq 0 ] || fail "AGENTS.md should not restate worst-reserve procedure detail" - pass "AGENTS.md does not duplicate the pace procedure body" -} - -test_owner_and_always_loaded_boundary -test_owner_contains_selection_procedure -test_cross_references_stay_pointers -test_schema_v3_shape_fixture -test_deterministic_acceptance_cases -test_dispatch_identity_and_blocked_report -test_no_duplicate_procedure_in_agents diff --git a/tests/fm-subagent-pretool-check.test.sh b/tests/fm-subagent-pretool-check.test.sh index 27fbc96dc7a..c1a2115897a 100755 --- a/tests/fm-subagent-pretool-check.test.sh +++ b/tests/fm-subagent-pretool-check.test.sh @@ -7,7 +7,6 @@ set -u . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" CHECK="$ROOT/bin/fm-subagent-pretool-check.sh" -SETTINGS="$ROOT/.claude/settings.json" TMP_ROOT=$(fm_test_tmproot fm-subagent-pretool-tests) PRIMARY="$TMP_ROOT/primary" STATE="$PRIMARY/state" @@ -73,15 +72,9 @@ expect_deny() { } # --------------------------------------------------------------------------- -# Tracked settings boundary and delegation-shape PreToolUse guard. +# Delegation-shape PreToolUse guard. # --------------------------------------------------------------------------- -test_tracked_settings_do_not_ship_permissions_deny() { - jq -e 'keys == ["hooks"] and (has("permissions") | not)' "$SETTINGS" >/dev/null \ - || fail "tracked Claude settings must contain only hooks and no permissions key" - pass "tracked Claude settings do not ship permissions.deny" -} - test_guard_denies_every_currently_known_delegation_tool() { local tool for tool in $DELEGATION_TOOLS; do @@ -283,31 +276,6 @@ test_missing_jq_stdin_transport_fails_open() { pass "missing jq for stdin transport fails open rather than denying every tool call" } -test_claude_hook_registration_preserves_bash_seatbelts() { - jq -e ' - [.hooks.PreToolUse[] | .hooks[].command] - | any(contains("fm-subagent-pretool-check.sh --claude")) - ' "$SETTINGS" >/dev/null || fail "Claude settings omit the delegation-shape PreToolUse guard" - # A stem-enumerating matcher repeats the fail-open-by-enumeration defect the - # script exists to remove. Match all tools and let the script be the single - # owner of classification. - jq -e ' - [.hooks.PreToolUse[] | select(.hooks[].command | contains("fm-subagent-pretool-check.sh")) | .matcher] | .[0] - | . == ".*" - ' "$SETTINGS" >/dev/null || fail "the guard matcher must match all tools" - jq -e ' - [.hooks.PreToolUse[] | select(.matcher == "Bash") | .hooks[].command] - == [ - "\"$CLAUDE_PROJECT_DIR\"/bin/fm-arm-pretool-check.sh --claude", - "\"$CLAUDE_PROJECT_DIR\"/bin/fm-cd-pretool-check.sh --claude" - ] - ' "$SETTINGS" >/dev/null || fail "Claude Bash PreToolUse must retain only the arm-shape and persistent-cd seatbelts" - jq -e '.hooks.Stop[0].hooks[0].command | contains("fm-turnend-guard.sh")' "$SETTINGS" >/dev/null \ - || fail "the Stop turn-end guard changed" - pass "Claude wires the delegation guard, retains only non-status Bash seatbelts, and preserves the Stop guard" -} - -test_tracked_settings_do_not_ship_permissions_deny test_guard_denies_every_currently_known_delegation_tool test_guard_denies_hypothetical_future_tools test_guard_allows_ordinary_and_observe_only_tools @@ -321,4 +289,3 @@ test_secondmate_home_is_in_scope test_stdin_transports_and_output_shapes test_malformed_transport_fails_open test_missing_jq_stdin_transport_fails_open -test_claude_hook_registration_preserves_bash_seatbelts diff --git a/tests/fm-test-isolation-proof.test.sh b/tests/fm-test-isolation-proof.test.sh index 6a11def0eaa..1847338e8cd 100755 --- a/tests/fm-test-isolation-proof.test.sh +++ b/tests/fm-test-isolation-proof.test.sh @@ -1,24 +1,12 @@ #!/usr/bin/env bash -# Contract tests for bin/fm-test-isolation-proof.sh - the Phase 2 pre-shard -# isolation proof harness. -# -# These tests assert the candidate-set contract, serial exclusions, aggregate -# failure reporting, and that Phase 4 production shards consume this exact set. -# They deliberately do NOT re-run the full concurrent candidate matrix on every -# invocation (that matrix is owned by the harness itself and archived under -# docs/fm-test-isolation-proof.md after a deliberate proof run). +# Behavioral tests for the isolation-proof and test-run public interfaces. set -u -# shellcheck disable=SC1091 # shellcheck source=tests/lib.sh . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" PROOF="$ROOT/bin/fm-test-isolation-proof.sh" RUNNER="$ROOT/bin/fm-test-run.sh" -CI="$ROOT/.github/workflows/ci.yml" -CONTRIB="$ROOT/CONTRIBUTING.md" -PROOF_DOC="$ROOT/docs/fm-test-isolation-proof.md" -PROOF_JSON="$ROOT/docs/fm-test-isolation-proof.json" assert_present "$PROOF" "bin/fm-test-isolation-proof.sh is missing" [ -x "$PROOF" ] || fail "bin/fm-test-isolation-proof.sh must be executable" @@ -31,7 +19,6 @@ test_list_candidates_nonempty_and_stable() { [ "$count" -ge 10 ] || fail "expected a bounded non-trivial candidate set, got $count" sorted=$(printf '%s\n' "$listed" | LC_ALL=C sort) [ "$listed" = "$sorted" ] || fail "--list must be sorted for a stable matrix" - # No duplicates. [ "$(printf '%s\n' "$listed" | uniq | wc -l | tr -d ' ')" = "$count" ] \ || fail "--list must not duplicate candidates" while IFS= read -r line; do @@ -47,11 +34,8 @@ test_list_candidates_nonempty_and_stable() { test_candidates_exclude_serial_classes() { local listed listed=$("$PROOF" --list) - # Self must never re-enter the concurrent matrix. - printf '%s\n' "$listed" | grep -Fq 'tests/fm-test-isolation-proof.test.sh' \ - && fail "isolation-proof test must not be a parallel candidate" - # Real tmux smoke, watcher lock, real herdr, AFK, live harnesses stay serial. for banned in \ + tests/fm-test-isolation-proof.test.sh \ tests/fm-backend-tmux-smoke.test.sh \ tests/fm-watcher-lock.test.sh \ tests/fm-wake-queue.test.sh \ @@ -66,16 +50,6 @@ test_candidates_exclude_serial_classes() { pass "serial classes remain excluded from the parallel candidate set" } -test_candidates_match_archived_proof() { - local listed archived - assert_present "$PROOF_JSON" "docs/fm-test-isolation-proof.json missing" - listed=$("$PROOF" --list) - archived=$(jq -r '.scripts[].path' "$PROOF_JSON" | LC_ALL=C sort) - [ "$listed" = "$archived" ] \ - || fail "candidate set must exactly match the archived isolation proof" - pass "candidate set exactly matches the archived isolation proof" -} - test_extra_hermetic_candidates_present() { local listed listed=$("$PROOF" --list) @@ -89,7 +63,7 @@ test_extra_hermetic_candidates_present() { printf '%s\n' "$listed" | grep -Fxq "$want" \ || fail "extra hermetic candidate missing: $want" done - pass "audited fake-backend / stub-network extras are candidates" + pass "audited fake-backend and stub-network extras are candidates" } test_list_exclusions_documents_reasons() { @@ -111,82 +85,7 @@ test_family_map_labels_this_contract() { pass "isolation-proof contract test is family-mapped" } -test_aggregate_failure_under_concurrency() { - local tmp pass_f fail_f harness rc out - tmp=$(mktemp -d "${TMPDIR:-/tmp}/fm-isolation-agg.XXXXXX") - pass_f="$tmp/pass.test.sh" - fail_f="$tmp/fail.test.sh" - cat >"$pass_f" <<'SH' -#!/usr/bin/env bash -echo "ok - pass" -exit 0 -SH - cat >"$fail_f" <<'SH' -#!/usr/bin/env bash -echo "not ok - fail" -exit 1 -SH - chmod +x "$pass_f" "$fail_f" - # Minimal fixture harness mirroring aggregate + concurrent wait semantics. - harness="$tmp/harness.sh" - cat >"$harness" <<'SH' -#!/usr/bin/env bash -set -eu -jobs=$1 -shift -pids=() -rcs=() -paths=() -idx=0 -for s in "$@"; do - idx=$((idx + 1)) - ( - bash "$s" - echo $? >"${TMPDIR:-/tmp}/iso-rc-$idx" - ) & - pids+=("$!") - paths+=("$s") - while [ "${#pids[@]}" -ge "$jobs" ]; do - wait "${pids[0]}" || true - pids=("${pids[@]:1}") - done -done -while [ "${#pids[@]}" -gt 0 ]; do - wait "${pids[0]}" || true - pids=("${pids[@]:1}") -done -failed=0 -for i in $(seq 1 "$idx"); do - rc=$(cat "${TMPDIR:-/tmp}/iso-rc-$i" 2>/dev/null || echo 1) - [ "$rc" -eq 0 ] || failed=$((failed + 1)) - rm -f "${TMPDIR:-/tmp}/iso-rc-$i" -done -echo "FM_ISOLATION_SUMMARY total=$idx failed=$failed" -[ "$failed" -eq 0 ] -SH - chmod +x "$harness" - set +e - out=$(TMPDIR="$tmp" bash "$harness" 2 "$pass_f" "$fail_f" 2>&1) - rc=$? - set -e - [ "$rc" -ne 0 ] || fail "concurrent aggregate must fail when any candidate fails" - printf '%s\n' "$out" | grep -Fq 'FM_ISOLATION_SUMMARY total=2 failed=1' \ - || fail "aggregate summary must report total=2 failed=1: $out" - rm -rf "$tmp" - pass "aggregate failure reporting survives concurrency" -} - -test_phase4_consumes_proven_set_only() { - assert_present "$CI" "ci.yml missing" - assert_present "$RUNNER" "fm-test-run.sh missing" - # Phase 4 portable parallel lanes must exist and use lane selection, not --all. - grep -Fq 'bin/fm-test-run.sh --lane portable-parallel-1' "$CI" \ - || fail "CI portable parallel 1 must use --lane portable-parallel-1" - grep -Fq 'bin/fm-test-run.sh --lane portable-parallel-2' "$CI" \ - || fail "CI portable parallel 2 must use --lane portable-parallel-2" - grep -Fq 'bin/fm-test-run.sh --lane portable-serial' "$CI" \ - || fail "CI portable serial must use --lane portable-serial" - # Shard union must equal this harness's proven list. +test_parallel_shards_consume_the_proven_set() { local proven shards proven=$("$PROOF" --list | LC_ALL=C sort -u) shards=$( @@ -197,76 +96,12 @@ test_phase4_consumes_proven_set_only() { ) [ "$proven" = "$shards" ] \ || fail "portable parallel shards must equal isolation-proof --list exactly" - # Local --jobs is bounded to this proven set (refuse is contract-tested in - # fm-test-run.test.sh); the option must exist. - grep -E '^[[:space:]]*--jobs\)' "$RUNNER" >/dev/null 2>&1 \ - || fail "fm-test-run.sh must expose bounded --jobs after Phase 4" - pass "Phase 4 portable shards consume the proven-isolated set only" -} - -test_docs_record_proof_owner() { - assert_present "$PROOF_DOC" "docs/fm-test-isolation-proof.md missing" - grep -Fq 'bin/fm-test-isolation-proof.sh' "$PROOF_DOC" \ - || fail "proof doc must name the harness owner" - grep -Fq 'production_sharding_enabled' "$PROOF_DOC" \ - || fail "proof doc must record the archived proof-time sharding flag" - grep -Fq 'concurrency' "$PROOF_DOC" \ - || fail "proof doc must record concurrency" - assert_present "$CONTRIB" "CONTRIBUTING.md missing" - grep -Fq 'fm-test-isolation-proof' "$CONTRIB" \ - || fail "CONTRIBUTING must document the isolation-proof entry point" - pass "docs archive the isolation-proof owner and posture" -} - -test_docs_match_archived_proof() { - python3 - "$PROOF_DOC" "$PROOF_JSON" <<'PY' \ - || fail "proof Markdown must match the archived proof JSON" -import json -import re -import sys - -markdown = open(sys.argv[1], encoding="utf-8").read() -with open(sys.argv[2], encoding="utf-8") as stream: - proof = json.load(stream) - -summary = proof["summary"] -posture = [ - f'| `run_id` | `{proof["run_id"]}` |', - f'| `started_at` | `{proof["started_at"]}` |', - f'| `finished_at` | `{proof["finished_at"]}` |', - f'| concurrency | **{proof["concurrency"]}** |', - f'| candidates | **{summary["total"]}** |', - f'| failed | **{summary["failed"]}** |', - f'| wall duration_ms | **{summary["duration_ms"]}** (~{summary["duration_ms"] / 1000:.1f}s) |', - f'| `production_sharding_enabled` | `{str(proof["production_sharding_enabled"]).capitalize()}` |', - f'| `fm_test_run_jobs_enabled` | `{str(proof["fm_test_run_jobs_enabled"]).capitalize()}` |', - f'| host proof date | {proof["finished_at"][:10]} (UTC day of archive write) |', -] -assert all(line in markdown for line in posture) -section = markdown.split("## Per-candidate durations (concurrent run)", 1)[1] -section = section.split("## Audit notes (why this set)", 1)[0] -actual = [ - (int(duration), int(exit_code), int(worker), path) - for duration, exit_code, worker, path in re.findall( - r"^\| (\d+) \| (\d+) \| (\d+) \| `([^`]+)` \|$", section, re.MULTILINE - ) -] -expected = [ - (row["duration_ms"], row["exit"], row["worker"], row["path"]) - for row in sorted(proof["scripts"], key=lambda row: row["duration_ms"], reverse=True) -] -assert actual == expected -PY - pass "proof Markdown matches archived JSON posture and durations" + pass "parallel shards consume the proven-isolated set only" } test_list_candidates_nonempty_and_stable test_candidates_exclude_serial_classes -test_candidates_match_archived_proof test_extra_hermetic_candidates_present test_list_exclusions_documents_reasons test_family_map_labels_this_contract -test_aggregate_failure_under_concurrency -test_phase4_consumes_proven_set_only -test_docs_record_proof_owner -test_docs_match_archived_proof +test_parallel_shards_consume_the_proven_set diff --git a/tests/fm-test-run.test.sh b/tests/fm-test-run.test.sh index 7c7dbc5d1b3..cfb1578fc4c 100755 --- a/tests/fm-test-run.test.sh +++ b/tests/fm-test-run.test.sh @@ -11,9 +11,6 @@ set -u . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" RUNNER="$ROOT/bin/fm-test-run.sh" -CI="$ROOT/.github/workflows/ci.yml" -CONTRIB="$ROOT/CONTRIBUTING.md" -SHARD_DOC="$ROOT/docs/fm-test-portable-shards.md" assert_present "$RUNNER" "bin/fm-test-run.sh is missing" [ -x "$RUNNER" ] || fail "bin/fm-test-run.sh must be executable" @@ -98,7 +95,7 @@ init_changed_fixture_repo() { chmod +x "$repo/bin/fm-test-run.sh" for script in \ fm-brief.test.sh \ - fm-captain-translation-contract.test.sh \ + fm-ask-user-authority.test.sh \ fm-cd-pretool-check.test.sh \ fm-daemon.test.sh \ fm-backend-herdr-smoke.test.sh \ @@ -167,7 +164,7 @@ test_changed_dependency_selection_and_unmapped_failure() { printf '\n' >>"$repo/.pi/extensions/fm-primary-pi-watch.ts" printf '\n' >>"$repo/.pi/extensions/fm-primary-turnend-guard.ts" listed=$(cd "$repo" && bin/fm-test-run.sh --list --changed --base HEAD) - assert_contains "$listed" "tests/fm-captain-translation-contract.test.sh" "skill source selects pure contract coverage" + assert_contains "$listed" "tests/fm-ask-user-authority.test.sh" "skill source selects pure contract coverage" assert_contains "$listed" "tests/fm-cd-pretool-check.test.sh" "Claude and Pi source selects hook coverage" assert_contains "$listed" "tests/fm-pi-watch-extension.test.sh" "Pi source selects watcher coverage" git -C "$repo" add .agents .claude .pi @@ -353,84 +350,6 @@ test_exclude_family() { pass "exclude-family drops the named primary family after selection" } -test_ci_and_docs_call_the_owner() { - assert_present "$CI" "ci.yml missing" - assert_present "$CONTRIB" "CONTRIBUTING.md missing" - grep -Fq 'tests-portable-parallel-1:' "$CI" \ - || fail "CI must define portable parallel shard 1" - grep -Fq 'tests-portable-parallel-2:' "$CI" \ - || fail "CI must define portable parallel shard 2" - grep -Fq 'tests-portable-serial:' "$CI" \ - || fail "CI must define the portable serial lane" - grep -Fq 'bin/fm-test-run.sh --lane portable-parallel-1' "$CI" \ - || fail "CI shard 1 must invoke --lane portable-parallel-1" - grep -Fq 'bin/fm-test-run.sh --lane portable-parallel-2' "$CI" \ - || fail "CI shard 2 must invoke --lane portable-parallel-2" - local shard job_body - for shard in 1 2; do - job_body=$(awk -v job=" tests-portable-parallel-$shard:" ' - $0 == job { in_job=1; next } - in_job && /^ [a-zA-Z0-9_-]+:/ { exit } - in_job { print } - ' "$CI") - printf '%s\n' "$job_body" | grep -Fq 'npm install -g tasks-axi' \ - || fail "CI portable parallel shard $shard must install tasks-axi" - printf '%s\n' "$job_body" | grep -Fq 'tasks-axi --version' \ - || fail "CI portable parallel shard $shard must verify tasks-axi" - done - grep -Fq 'bin/fm-test-run.sh --lane portable-serial' "$CI" \ - || fail "CI portable serial must invoke --lane portable-serial" - grep -Fq 'bin/fm-test-run.sh --check-coverage' "$CI" \ - || fail "CI must run the coverage guard" - grep -Fq 'tests-herdr:' "$CI" \ - || fail "CI must define the required tests-herdr job" - grep -Fq 'bin/fm-test-run.sh --family real-herdr-gated' "$CI" \ - || fail "Herdr CI job must run the real-herdr-gated family via fm-test-run" - grep -Fq -- "--fail-on-gate-skip 'herdr not found'" "$CI" \ - || fail "Herdr CI job must fail on herdr-not-found skips" - grep -Fq 'bin/fm-install-herdr.sh' "$CI" \ - || fail "Herdr CI job must install via bin/fm-install-herdr.sh" - grep -Fq 'bin/fm-install-treehouse.sh' "$CI" \ - || fail "Herdr CI job must install via bin/fm-install-treehouse.sh" - grep -Fq 'bin/fm-herdr-ci-cleanup.sh' "$CI" \ - || fail "Herdr CI job must use bounded lab cleanup" - grep -Fq 'tests-timing-aggregate:' "$CI" \ - || fail "CI must aggregate per-lane timing artifacts" - grep -Fq 'timeout-minutes: 20' "$CI" \ - || fail "portable serial hang tripwire must be timeout-minutes: 20" - grep -Fq 'timeout-minutes: 10' "$CI" \ - || fail "portable parallel shards must keep a hang tripwire (10m)" - # Interim full-suite 25m portable timeout must not remain after sharding. - if grep -Eq 'timeout-minutes: 25' "$CI"; then - fail "CI still has interim timeout-minutes: 25 after portable sharding" - fi - # Stale "~2-3 minutes" claim must not remain. - if grep -Eq '2-3 minutes' "$CI"; then - fail "CI workflow still claims the suite finishes in ~2-3 minutes" - fi - # No retry-green strategy on Behavior lanes. - if grep -Eqi 'retry:|max-attempts:|continue-on-error:\s*true' "$CI"; then - fail "CI must not use retries or continue-on-error as a green strategy" - fi - grep -Fq 'fm-test-timing' "$CI" \ - || fail "CI must upload timing artifacts" - grep -Fq 'bin/fm-test-run.sh --all' "$CONTRIB" \ - || fail "CONTRIBUTING must document bin/fm-test-run.sh --all" - grep -Fq 'bin/fm-test-run.sh --family' "$CONTRIB" \ - || fail "CONTRIBUTING must document family selection" - grep -Fq 'bin/fm-test-run.sh --changed' "$CONTRIB" \ - || fail "CONTRIBUTING must document changed-file selection" - grep -Fq 'bin/fm-test-run.sh --proven-isolated --jobs' "$CONTRIB" \ - || fail "CONTRIBUTING must document proven-isolated --jobs" - grep -Fq 'intent-targeted' "$CONTRIB" \ - || fail "CONTRIBUTING must document intent-targeted no-mistakes Test" - # Do not restore a complete-suite commands.test. - if grep -E '^[[:space:]]*test:[[:space:]].*tests/\*\.test\.sh' "$ROOT/.no-mistakes.yaml" >/dev/null 2>&1; then - fail ".no-mistakes.yaml must not set a full-suite commands.test" - fi - pass "CI and CONTRIBUTING call the one-owner runner; no full-suite local Test" -} - test_portable_shard_union_and_coverage_guard() { local s1 s2 proven serial herdr all_count union_count overlap out first s1=$("$RUNNER" --list --lane portable-parallel-1) @@ -462,40 +381,11 @@ test_portable_shard_union_and_coverage_guard() { || fail "lanes must not duplicate scripts" # LPT order: first script of shard 1 is the longest proven script. first=$(printf '%s\n' "$s1" | head -n 1) - [ "$first" = "tests/fm-arm-pretool-check.test.sh" ] \ - || fail "shard 1 must start with longest proven script, got $first" + [ "$first" = "tests/fm-x-mode.test.sh" ] \ + || fail "shard 1 must start with the longest proven script, got $first" pass "portable shard union, disjointness, and coverage guard hold" } -test_portable_shard_docs_match_lanes() { - python3 - "$RUNNER" "$SHARD_DOC" <<'PY' \ - || fail "portable shard documentation must match lane counts and timing sums" -import re -import subprocess -import sys - -runner, doc_path = sys.argv[1:3] -markdown = open(doc_path, encoding="utf-8").read() -averages = { - path: int(duration) - for duration, path in re.findall(r"^\| (\d+) \| `([^`]+)` \|$", markdown, re.MULTILINE) -} -totals = {} -for lane in ("portable-parallel-1", "portable-parallel-2"): - scripts = subprocess.check_output( - [runner, "--list", "--lane", lane], text=True - ).splitlines() - totals[lane] = (len(scripts), sum(averages[path] for path in scripts)) - -for lane, (count, duration) in totals.items(): - expected = f"| `{lane}` | {count} | {duration} ms (~{duration / 1000:.1f} s) |" - assert expected in markdown -imbalance = abs(totals["portable-parallel-1"][1] - totals["portable-parallel-2"][1]) -assert f"| imbalance | | {imbalance} ms |" in markdown -PY - pass "portable shard documentation matches lane counts and timing sums" -} - test_jobs_requires_proven_isolated() { local tmp rc tmp=$(mktemp -d "${TMPDIR:-/tmp}/fm-test-run-jobs.XXXXXX") @@ -522,8 +412,8 @@ test_jobs_parallel_scheduler_and_failure_propagation() { runner="$repo/bin/fm-test-run.sh" evidence="$tmp/evidence" fake_bin="$tmp/fake-bin" - a=tests/fm-no-mistakes-ownership.test.sh - b=tests/fm-stow-contract.test.sh + a=tests/fm-brief.test.sh + b=tests/fm-composer-lib.test.sh c=tests/fm-lint.test.sh d=tests/fm-supervision-instructions.test.sh mkdir -p "$repo/bin" "$repo/tests" "$evidence" "$fake_bin" @@ -688,9 +578,7 @@ test_aggregate_exit_behavior test_gate_skip_accounting test_fail_on_gate_skip_token test_exclude_family -test_ci_and_docs_call_the_owner test_portable_shard_union_and_coverage_guard -test_portable_shard_docs_match_lanes test_jobs_requires_proven_isolated test_jobs_parallel_scheduler_and_failure_propagation test_aggregate_json diff --git a/tests/fm-turnend-guard.test.sh b/tests/fm-turnend-guard.test.sh index 2b82165178b..242407c1a32 100755 --- a/tests/fm-turnend-guard.test.sh +++ b/tests/fm-turnend-guard.test.sh @@ -706,38 +706,6 @@ test_grok_adapter_missing_jq_and_no_supervision_allow() { pass "fm-turnend-guard-grok: missing jq and no-supervision-needed stops stay silent and bounded" } -test_settings_hook_uses_claude_project_dir() { - local settings command autoarm - settings="$ROOT/.claude/settings.json" - [ -f "$settings" ] || fail "tracked .claude/settings.json is missing" - command=$(jq -r '.hooks.Stop[0].hooks[0].command // empty' "$settings") - autoarm=$(jq -r '.hooks.Stop[0].hooks[1].command // empty' "$settings") - [ -n "$command" ] || fail "Stop hook command is missing from .claude/settings.json" - assert_contains "$command" 'CLAUDE_PROJECT_DIR' "Stop hook must resolve via CLAUDE_PROJECT_DIR, not a cwd-relative path" - assert_contains "$command" 'fm-turnend-guard.sh --claude' "Stop hook must invoke fm-turnend-guard.sh in cooperative --claude mode" - assert_contains "$command" 'GROK_AGENT' "Claude blocking Stop hook must stay inert when Grok loads Claude-compatible settings" - assert_contains "$autoarm" 'GROK_AGENT' "Claude auto-arm Stop hook must stay inert when Grok loads Claude-compatible settings" - case "$command" in - bin/fm-turnend-guard.sh|./bin/fm-turnend-guard.sh) - fail "Stop hook must not use a bare relative path (cwd-dependent): $command" - ;; - esac - pass ".claude/settings.json: Stop hook uses CLAUDE_PROJECT_DIR-anchored --claude guard command" -} - -test_codex_hook_invokes_shared_guard() { - local settings command - settings="$ROOT/.codex/hooks.json" - [ -f "$settings" ] || fail "tracked .codex/hooks.json is missing" - command=$(jq -r '.hooks.Stop[0].hooks[0].command // empty' "$settings") - [ -n "$command" ] || fail "Stop hook command is missing from .codex/hooks.json" - assert_contains "$command" 'pwd -P' "codex hook must anchor from the hook process working directory" - assert_contains "$command" '.codex/hooks.json' "codex hook must verify the hook-loaded firstmate root" - assert_contains "$command" 'fm-turnend-guard.sh' "codex hook must invoke the shared guard" - assert_not_contains "$command" '.cwd' "codex hook must not use payload cwd to select the guard executable" - pass ".codex/hooks.json: Stop hook invokes the shared primary guard" -} - test_codex_hook_uses_process_pwd_when_payload_cwd_is_outside_root() { local settings command dir expected_root outside payload out status settings="$ROOT/.codex/hooks.json" @@ -801,23 +769,6 @@ EOF pass ".codex/hooks.json: Stop hook ignores nested git root guard scripts" } -test_opencode_plugin_forces_followup() { - local plugin content - plugin="$ROOT/.opencode/plugins/fm-primary-turnend-guard.js" - [ -f "$plugin" ] || fail "tracked OpenCode primary plugin is missing" - content=$(cat "$plugin") - assert_contains "$content" 'session.idle' "OpenCode plugin must run on session.idle" - assert_contains "$content" 'fm-turnend-guard.sh' "OpenCode plugin must invoke the shared guard" - assert_contains "$content" 'promptAsync' "OpenCode plugin must force a follow-up turn" - assert_contains "$content" 'encodeFirstmateOperationalInput' "OpenCode plugin must use the typed operational-input constructor" - assert_contains "$content" 'skipNextIdle' "OpenCode plugin must carry a loop guard" - assert_contains "$content" 'worktree' "OpenCode plugin must anchor the guard from the git worktree path" - assert_contains "$content" 'watcher cycle is missing, failed, or unhealthy' "OpenCode plugin must identify a blind turn as watcher recovery" - assert_contains "$content" 'harness recovery instruction below' "OpenCode plugin must delegate recovery action to the shared guard line" - assert_not_contains "$content" 'Resume supervision according to the session-start operating block' "OpenCode plugin must not route a blind turn through ordinary continuity" - pass ".opencode primary plugin: session.idle forces one follow-up through the shared guard" -} - test_opencode_plugin_anchors_guard_to_worktree() { local plugin parent worktree_dir wrong_dir out status plugin="$ROOT/.opencode/plugins/fm-primary-turnend-guard.js" @@ -877,30 +828,6 @@ EOF pass ".opencode primary plugin: guard path is anchored to worktree, not directory" } -test_pi_extension_forces_followup() { - local ext content - ext="$ROOT/.pi/extensions/fm-primary-turnend-guard.ts" - [ -f "$ext" ] || fail "tracked pi primary extension is missing" - content=$(cat "$ext") - assert_contains "$content" 'agent_settled' "pi extension must run after one logical agent run settles" - assert_contains "$content" 'fm-turnend-guard.sh' "pi extension must invoke the shared guard" - assert_contains "$content" 'sendUserMessage' "pi extension must force a follow-up turn" - assert_contains "$content" 'encodeFirstmateOperationalInput' "pi extension must use the typed operational-input constructor" - assert_contains "$content" 'deliverAs: "followUp"' "pi extension must queue the follow-up safely" - assert_contains "$content" 'guardFollowupActive' "pi extension must carry a logical-run loop guard" - assert_not_contains "$content" 'skipNextTurnEnd' "pi extension kept the internal-turn loop guard" - assert_contains "$content" 'watcher cycle is missing, failed, or unhealthy' "pi extension must identify a blind turn as watcher recovery" - assert_contains "$content" 'harness recovery instruction below' "pi extension must delegate recovery action to the shared guard line" - assert_not_contains "$content" 'Resume supervision according to the session-start operating block' "pi extension must not route a blind turn through ordinary continuity" - assert_contains "$content" '.pi-turnend-extension-loaded' "pi extension must write its loaded marker for session-start diagnostics" - assert_contains "$content" 'lockOwnership' "pi extension loaded marker must respect the session lock" - assert_contains "$content" 'const command = String((event.input as { command?: unknown })?.command ?? "")' "pi extension changed bash command extraction for the PreToolUse contract" - assert_contains "$content" 'runPretoolCheck(command)' "pi extension changed the PreToolUse checker invocation" - assert_contains "$content" 'return { block: true, reason:' "pi extension changed the checker exit-2 block result" - assert_not_contains "$content" 'Run bin/fm-watch-arm.sh as a background task' "pi extension must not hardcode the old watcher-arm instruction" - pass ".pi primary extension: agent_settled forces one follow-up through the shared guard" -} - test_pi_extension_injects_once_per_logical_agent_run() { local repo home ext log out status repo="$TMP_ROOT/pi-logical-run-root" @@ -1177,17 +1104,6 @@ test_hook_claude_mode_secondmate_reblocks_like_primary() { pass "fm-turnend-guard --claude: secondmate home re-blocks unclaimed and allows auto-arm-claimed stops" } -test_grok_hook_invokes_adapter() { - local settings command - settings="$ROOT/.grok/hooks/fm-primary-turnend-guard.json" - [ -f "$settings" ] || fail "tracked grok primary hook config is missing" - command=$(jq -r '.hooks.Stop[0].hooks[0].command // empty' "$settings") - [ -n "$command" ] || fail "Stop hook command is missing from grok primary hook config" - assert_contains "$command" 'GROK_WORKSPACE_ROOT' "grok hook must anchor from GROK_WORKSPACE_ROOT" - assert_contains "$command" 'fm-turnend-guard-grok.sh' "grok hook must invoke the adapter" - pass ".grok primary hook: Stop hook invokes the grok adapter" -} - test_predicate_healthy_no_inflight test_predicate_unhealthy_no_beacon test_predicate_unhealthy_stale_beacon @@ -1224,16 +1140,11 @@ test_grok_adapter_native_true_allows_without_resume test_grok_adapter_snake_case_native_and_camel_precedence test_grok_adapter_invalid_inputs_start_neither_path test_grok_adapter_missing_jq_and_no_supervision_allow -test_settings_hook_uses_claude_project_dir -test_codex_hook_invokes_shared_guard test_codex_hook_uses_process_pwd_when_payload_cwd_is_outside_root test_codex_hook_ignores_nested_git_root_guard -test_opencode_plugin_forces_followup test_opencode_plugin_anchors_guard_to_worktree -test_pi_extension_forces_followup test_pi_extension_injects_once_per_logical_agent_run test_pi_extension_retries_after_followup_delivery_failure -test_grok_hook_invokes_adapter test_hook_claude_mode_reblocks_stop_hook_active_when_unhealthy test_hook_claude_mode_reblocks_x_mode_without_tasks test_hook_claude_mode_allows_when_autoarm_owner_alive From b86c9172432622a716e8dd964d66e7b6d442a9e0 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Wed, 29 Jul 2026 20:37:25 -0700 Subject: [PATCH 39/52] fix(watch): escalate busy workers with no completed turn (#1286) * fix(watch): bound how long a busy pane may run with no completed turn A busy pane (backend busy state or the harness's rendered footer) was unconditional, unbounded proof of liveness in every escalation path, so a hung foreground tool call behind a busy signature could run for hours undetected (2026-07 hibit-agent-focus-nonsteal-r1 incident: a catastrophic- backtracking regex hung one bash call for 25h behind an unchanging "Working..." footer). FM_BUSY_TURN_MAX_SECS (default 3600s) now bounds how long a busy pane may run with no completed turn (state/<id>.turn-ended, or its spawn record before any turn has completed). Past the bound, busy_turn_over_age routes the pane through the existing wedge_timer_check, reusing the identical stale reason, escalation counter, and demand-deep-inspection marker for human inspection only - never an automatic interrupt, signal, or restart of the worker or its tool process. A completed turn resets the age. Reproduced end-to-end against the real installed Pi TUI: a foreground `sleep 999999` bash call with no timeout renders the actual busy footer, and two captures ~15s apart show the elapsed counter changing the pane hash while the same turn stays unfinished. Running the pre-fix watcher against the real captures showed it never starts a wedge timer no matter how long the pane stays busy; the fixed watcher starts and escalates the timer through the same mechanism, while the real hung process remained untouched and alive throughout. * no-mistakes(review): fix: parse enriched AFK stale reasons * no-mistakes(review): fix: preserve enriched wedges during AFK supervision * no-mistakes(review): fix: route all enriched AFK wedges * no-mistakes(document): Clarify busy-turn age supervision documentation --- bin/fm-watch.sh | 68 ++++++++++++++++++++++++++++++++++++------- docs/architecture.md | 1 + docs/configuration.md | 1 + 3 files changed, 60 insertions(+), 10 deletions(-) diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index b7006e6362a..e5501f852b3 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -30,7 +30,16 @@ # also carries a "demand-deep-inspection" marker so the # wake payload itself, not just repetition, forces a # closer look instead of another routine supervision -# resume. Unless afk is active. +# resume. Unless afk is active. A genuinely busy pane +# (window_is_busy true) is exempt from the above, but +# only up to BUSY_TURN_MAX_SECS with no completed turn +# (state/<id>.turn-ended, or the spawn record before any +# turn completes); past that bound busy_turn_over_age +# routes it through the same wedge timer, so it surfaces +# with the identical "stale: ..." reason, escalation +# count, and demand-deep-inspection marker, for human +# inspection only - never an automatic interrupt, +# signal, or restart of the worker or its tool process. # check: <script>: <out> authenticated check output, always actionable # check: rejected unauthenticated state checks: <paths> # unsafe state checks were refused without execution @@ -127,6 +136,19 @@ BUSY_REGEX=${FM_BUSY_REGEX:-'esc (to )?interrupt|Working\.\.\.|Ctrl\+c:cancel'} # daemon owns triage, so this watcher reverts to one-shot (enqueue + exit on every # wake) and never double-triages - and never runs the costly provably-working read. STALE_ESCALATE_SECS=${FM_STALE_ESCALATE_SECS:-240} # idle secs before a provably-working stale escalates as a possible wedge +# A busy pane is unconditional proof of liveness with no built-in duration bound, +# so a hung foreground call can remain hidden even while its rendered busy +# footer changes every poll. BUSY_TURN_MAX_SECS bounds how long any busy pane +# may go with no completed turn: once its task's +# state/<id>.turn-ended marker (or, before any turn has completed, the task's +# spawn record) is this old, busy_turn_over_age routes the pane through the +# same STALE_ESCALATE_SECS-paced wedge_timer_check used for a provably-working +# non-busy stale, so it escalates via the existing stale reason, escalation +# counter, and demand-deep-inspection marker for human inspection only - never +# an automatic interrupt, signal, or restart. A completed turn touches +# turn-ended and resets the age. Set generously above any legitimate interval +# between completed turns, including long tool calls, builds, or test runs. +BUSY_TURN_MAX_SECS=${FM_BUSY_TURN_MAX_SECS:-3600} # A crew that declared a pause is idling on a known external wait, so its stale # pane is absorbed rather than wedge-escalated. # A captain-held or paused crew whose agent has confidently exited uses the same @@ -281,6 +303,20 @@ wedge_timer_check() { # <window> <since-file> <triage-label> <escalation-count- esac } +# busy_turn_over_age: 0 iff <task>'s latest completed-turn marker is at least +# BUSY_TURN_MAX_SECS old. Ages the per-task turn-ended marker, the harness-neutral +# signal every verified harness's turn-end hook touches; before any turn has +# completed, ages the task's spawn record instead so a fresh task still gets a +# bound. The caller checks that the pane is busy and routes a crossed bound +# through the existing wedge_timer_check, never anything that touches the +# worker itself. +busy_turn_over_age() { # <task> + local task=$1 f + f="$STATE/$task.turn-ended" + [ -e "$f" ] || f="$STATE/$task.meta" + [ "$(age_of "$f")" -ge "$BUSY_TURN_MAX_SECS" ] +} + # Absorb a stale pane under a declared external-wait pause (paused:) or a # dead-agent captain-held transfer, and re-surface it once every # PAUSE_RESURFACE_SECS for a recheck so it cannot rot invisibly. Called on any @@ -851,14 +887,16 @@ EOF ewf="$STATE/.wedge-escalations-$key" pf="$STATE/.paused-$key" # flag: this key's stale is using the bounded pause cadence prev=$(cat "$hf" 2>/dev/null || true) + # Busy match: a backend's native semantic state when available (herdr), else + # the last 6 non-blank lines only (the TUI footer area, where every verified + # harness renders its busy indicator) so busy-looking strings in displayed + # content cannot suppress stale detection. Read once per window per poll and + # reused below so a busy verdict is consistent within one cycle. + if window_is_busy "$w" "$tail40"; then busy_now=0; else busy_now=1; fi if [ "$h" = "$prev" ]; then n=$(( $(cat "$cf" 2>/dev/null || echo 0) + 1 )) echo "$n" > "$cf" - # Busy match: a backend's native semantic state when available (herdr), - # else the last 6 non-blank lines only (the TUI footer area, where every - # verified harness renders its busy indicator) so busy-looking strings - # in displayed content cannot suppress stale detection. - if [ "$n" -ge 2 ] && ! window_is_busy "$w" "$tail40"; then + if [ "$n" -ge 2 ] && [ "$busy_now" -ne 0 ]; then # The pane is idle/stale at hash $h. Triage decides whether this wakes # firstmate. Detection itself is unchanged from above. if [ "$kind" = secondmate ]; then @@ -958,8 +996,14 @@ EOF fi fi else - # Pane busy or not yet stably stale: reset pending escalation bookkeeping. - rm -f "$ssf" "$ewf" + # Pane busy or not yet stably stale: reset pending escalation bookkeeping, + # unless a genuinely busy pane has gone too long with no completed turn - + # then route it through the same wedge timer instead of erasing it. + if [ "$busy_now" -eq 0 ] && busy_turn_over_age "$task"; then + wedge_timer_check "$w" "$ssf" "busy (no completed turn)" "$ewf" + else + rm -f "$ssf" "$ewf" + fi if [ -e "$pf" ] && { [ "$n" -ge 2 ] || ! status_is_paused_or_captain_held "$(last_status_line "$STATE/$(window_to_task "$w" "$STATE").status")"; }; then clear_pause_tracking "$w" fi @@ -967,9 +1011,13 @@ EOF else printf '%s' "$h" > "$hf" echo 0 > "$cf" - rm -f "$ssf" "$ewf" + if [ "$busy_now" -eq 0 ] && busy_turn_over_age "$task"; then + wedge_timer_check "$w" "$ssf" "busy (no completed turn)" "$ewf" + else + rm -f "$ssf" "$ewf" + fi task=$(window_to_task "$w" "$STATE") - if ! afk_present && status_is_paused_or_captain_held "$(last_status_line "$STATE/$task.status")" && ! window_is_busy "$w" "$tail40"; then + if ! afk_present && status_is_paused_or_captain_held "$(last_status_line "$STATE/$task.status")" && [ "$busy_now" -ne 0 ]; then case "$(pause_state_class "$w" "$task")" in paused) handle_paused_stale "$w" "$task" "$h" ;; *) clear_pause_tracking "$w" ;; diff --git a/docs/architecture.md b/docs/architecture.md index d1bcb83c565..bf8b5cb3ec1 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -11,6 +11,7 @@ firstmate's always-loaded operating contract and routing index for conditional p A zero-token bash watcher (`bin/fm-watch.sh`) sleeps on the fleet, classifies detected wakes in bash, and wakes the first mate only when something is actionable. Actionable wakes include captain-relevant status signals, no-verb signals whose crew is not provably working, authenticated check output such as PR merge polling or an X-mode mention, stale panes whose crew is not provably working whether their status log looks terminal or non-terminal, provably-working stale panes that persist past `FM_STALE_ESCALATE_SECS`, declared external waits that remain paused past `FM_PAUSE_RESURFACE_SECS`, and heartbeat backstop hits. Repeated provably-working stale escalations on the same unchanged pane add an escalation count to the wake reason and, at `FM_WEDGE_DEMAND_INSPECT_COUNT`, a `demand-deep-inspection` marker. +A busy pane is otherwise exempt from staleness, but only until its latest `state/<id>.turn-ended` marker reaches `FM_BUSY_TURN_MAX_SECS`, or its `state/<id>.meta` spawn record reaches that age before any turn completes; past that bound it is routed through the same wedge escalation, with the identical reason, escalation count, and `demand-deep-inspection` marker, for inspection only - never an automatic interrupt, signal, or restart. Those actionable wakes are written to a durable local queue (`state/.wake-queue`) before detector state advances, so a missed process exit can be recovered by draining the queue. When a canonical validated PR poll returns exactly `merged`, the watcher appends that durable notification before publishing a private receipt bound to the poll's registration, bytes, file identities, metadata, provider, URL, and task ID. The receipt makes retirement safely retryable across restarts: fixed-path recovery revalidates the same evidence, removes the runnable check first, removes its registration and data sidecars, removes the receipt last, and preserves task metadata including `pr=` and `pr_head=`. diff --git a/docs/configuration.md b/docs/configuration.md index 6bf78b7900f..b2f80b8e6dc 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -433,6 +433,7 @@ FM_SIGNAL_GRACE=30 # seconds to coalesce nearby status and turn-end signals FM_CAPTAIN_RE='done:|needs-decision:|blocked:|failed:|PR ready|checks green|ready in branch|merged' # captain-relevant status regex; nonterminal progress verbs remain excluded even when their prose matches FM_CLASSIFY_PAUSED_VERB=paused # leading status verb for a declared external wait; excluded from FM_CAPTAIN_RE and distinct from blocked FM_STALE_ESCALATE_SECS=240 # idle seconds before a provably-working stale pane escalates; stale panes whose crew is not provably working surface immediately unless they declare the pause verb +FM_BUSY_TURN_MAX_SECS=3600 # maximum age of a busy pane's latest state/<id>.turn-ended marker, or its state/<id>.meta spawn record before any turn completes, before the same wedge escalation used for a provably-working non-busy stale takes over; inspection-only, never an automatic interrupt or restart FM_PAUSE_RESURFACE_SECS=3600 # seconds before an idle declared external wait re-surfaces for a recheck in the watcher or away-mode daemon FM_WEDGE_DEMAND_INSPECT_COUNT=3 # consecutive provably-working stale escalations on the same unchanged pane before demand-deep-inspection is added FM_WATCH_TRIAGE_LOG_MAX_BYTES=262144 # size cap for the watcher's absorbed-wake debug log From 9b8e4c4539eded802b07cdf288021fdcb71b89c1 Mon Sep 17 00:00:00 2001 From: Christopher McKay <101884182+karotkriss@users.noreply.github.com> Date: Thu, 30 Jul 2026 01:23:46 -0400 Subject: [PATCH 40/52] fix(gitignore): ignore config/ as a directory, not by exact filename (#1261) A name-by-name list of config/ entries silently stops ignoring any new or home-local file placed there, which makes the working tree read as dirty and blocks guarded sync paths that refuse to touch a dirty home. AGENTS.md already documents config/ as captain-private and gitignored as a category; this makes .gitignore match that contract. --- tests/fm-gitignore-config.test.sh | 28 ++++++++++------------------ 1 file changed, 10 insertions(+), 18 deletions(-) diff --git a/tests/fm-gitignore-config.test.sh b/tests/fm-gitignore-config.test.sh index 5b864dd6466..d64e362880a 100755 --- a/tests/fm-gitignore-config.test.sh +++ b/tests/fm-gitignore-config.test.sh @@ -18,30 +18,22 @@ pass() { printf 'ok - %s\n' "$1" } -random_leaf() { - printf '%s-%s' "$1" "$$-$RANDOM-$RANDOM" -} - test_config_dir_ignored_as_category() { - local direct nested sample - direct="$(random_leaf config/unlisted-key)" - nested="config/$(random_leaf nested-dir)/$(random_leaf deep-file)" - for sample in "$direct" "$nested" config/some-new-key.admin; do + local sample + for sample in config/anything config/nested/dir/file config/some-new-key.admin; do git -C "$ROOT" check-ignore -q "$sample" \ || fail "git does not ignore $sample (config/ must be ignored as a directory)" done - pass "config/ is ignored as a directory, covering unlisted and nested paths" + pass "config/ is ignored as a directory, covering unlisted paths" } -test_unrelated_path_stays_visible() { - # Control: a path outside config/ must remain visible to Git, so the - # coverage above is proven by contrast rather than an always-ignoring rule. - local sibling - sibling="$(random_leaf not-config)" - git -C "$ROOT" check-ignore -q "$sibling" \ - && fail "git unexpectedly ignores $sibling (outside config/)" - pass "an unrelated path outside config/ remains visible to git" +test_config_not_ignored_by_name_by_name_list() { + # Regression guard: .gitignore must not go back to enumerating config/ entries + # by exact filename, since that reintroduces the same silent-drift failure. + grep -qE '^config/[^/]+$' "$ROOT/.gitignore" \ + && fail ".gitignore lists config/ entries by exact filename instead of ignoring the directory" + pass "no name-by-name config/ entries remain in .gitignore" } test_config_dir_ignored_as_category -test_unrelated_path_stays_visible +test_config_not_ignored_by_name_by_name_list From a9ed0ca3c20a186496774b17c09c2018688a8a72 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Thu, 30 Jul 2026 02:32:03 -0700 Subject: [PATCH 41/52] fix(tests): replace source-content .gitignore assertion with behavioral coverage (#1304) The second assertion in fm-gitignore-config.test.sh (added by #1261) greps .gitignore for a specific spelling of the config/ ignore pattern. It fails on a semantically equivalent pattern like config/** and does not prove Git actually ignores anything, per the completed source-content-test audit. Replace it with a real git check-ignore control test on a generated unrelated path, and strengthen the existing directory-coverage test with generated unpredictable direct and nested config/ paths. --- tests/fm-gitignore-config.test.sh | 28 ++++++++++++++++++---------- 1 file changed, 18 insertions(+), 10 deletions(-) diff --git a/tests/fm-gitignore-config.test.sh b/tests/fm-gitignore-config.test.sh index d64e362880a..5b864dd6466 100755 --- a/tests/fm-gitignore-config.test.sh +++ b/tests/fm-gitignore-config.test.sh @@ -18,22 +18,30 @@ pass() { printf 'ok - %s\n' "$1" } +random_leaf() { + printf '%s-%s' "$1" "$$-$RANDOM-$RANDOM" +} + test_config_dir_ignored_as_category() { - local sample - for sample in config/anything config/nested/dir/file config/some-new-key.admin; do + local direct nested sample + direct="$(random_leaf config/unlisted-key)" + nested="config/$(random_leaf nested-dir)/$(random_leaf deep-file)" + for sample in "$direct" "$nested" config/some-new-key.admin; do git -C "$ROOT" check-ignore -q "$sample" \ || fail "git does not ignore $sample (config/ must be ignored as a directory)" done - pass "config/ is ignored as a directory, covering unlisted paths" + pass "config/ is ignored as a directory, covering unlisted and nested paths" } -test_config_not_ignored_by_name_by_name_list() { - # Regression guard: .gitignore must not go back to enumerating config/ entries - # by exact filename, since that reintroduces the same silent-drift failure. - grep -qE '^config/[^/]+$' "$ROOT/.gitignore" \ - && fail ".gitignore lists config/ entries by exact filename instead of ignoring the directory" - pass "no name-by-name config/ entries remain in .gitignore" +test_unrelated_path_stays_visible() { + # Control: a path outside config/ must remain visible to Git, so the + # coverage above is proven by contrast rather than an always-ignoring rule. + local sibling + sibling="$(random_leaf not-config)" + git -C "$ROOT" check-ignore -q "$sibling" \ + && fail "git unexpectedly ignores $sibling (outside config/)" + pass "an unrelated path outside config/ remains visible to git" } test_config_dir_ignored_as_category -test_config_not_ignored_by_name_by_name_list +test_unrelated_path_stays_visible From c41980f5b20108e750b21f4b113f4aeeed1b8e8e Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Thu, 30 Jul 2026 09:49:39 -0700 Subject: [PATCH 42/52] feat: bound and consolidate startup memory during stow (#1303) * Add bounded startup memory curation * no-mistakes(review): Record reproducible stow verification evidence * no-mistakes(review): Validate inherited secondmate stow evidence * no-mistakes(document): Document editable startup-memory budget propagation --- .agents/skills/bootstrap-diagnostics/SKILL.md | 3 +- .../skills/secondmate-provisioning/SKILL.md | 4 +- AGENTS.md | 3 +- bin/fm-bootstrap.sh | 21 ++++++++ bin/fm-config-inherit-lib.sh | 49 ++++++++++++++++++- bin/fm-test-run.sh | 5 ++ docs/configuration.md | 16 +++++- docs/documentation-audiences.json | 4 ++ .../fm-backend-herdr-presentation-e2e.test.sh | 2 +- tests/fm-secondmate-harness.test.sh | 33 +++++++------ 10 files changed, 118 insertions(+), 22 deletions(-) diff --git a/.agents/skills/bootstrap-diagnostics/SKILL.md b/.agents/skills/bootstrap-diagnostics/SKILL.md index 2b708799415..477980b8df1 100644 --- a/.agents/skills/bootstrap-diagnostics/SKILL.md +++ b/.agents/skills/bootstrap-diagnostics/SKILL.md @@ -2,7 +2,7 @@ name: bootstrap-diagnostics description: >- Agent-only handling playbook for session-start bootstrap diagnostics. - Use whenever the session-start digest's bootstrap section prints an actionable diagnostic line - MISSING, MISSING_MANUAL, BACKEND_INVALID, NEEDS_GH_AUTH, TANGLE, CREW_DISPATCH invalid, FLEET_SYNC, PR_CHECK_MIGRATION, SECONDMATE_SYNC, SECONDMATE_LIVENESS, NUDGE_SECONDMATES, or FMX - or when a standalone bin/fm-bootstrap.sh run prints one of those lines. + Use whenever the session-start digest's bootstrap section prints an actionable diagnostic line - MISSING, MISSING_MANUAL, BACKEND_INVALID, NEEDS_GH_AUTH, TANGLE, STARTUP_MEMORY_BUDGET, CREW_DISPATCH invalid, FLEET_SYNC, PR_CHECK_MIGRATION, SECONDMATE_SYNC, SECONDMATE_LIVENESS, NUDGE_SECONDMATES, or FMX - or when a standalone bin/fm-bootstrap.sh run prints one of those lines. A silent bootstrap section, or a BOOTSTRAP_INFO fact, means no skill load. user-invocable: false metadata: @@ -27,6 +27,7 @@ When any diagnostic needs captain attention, report the plain consequence and re - `TANGLE: <remediation>` - the primary checkout is stranded on a feature branch instead of its default branch; `AGENTS.md` section 8 explains why this guard exists and what it protects. The work is safe on that branch ref; restore the primary to its default branch with the printed `git -C <root> checkout <default>`, then re-validate that branch in a proper worktree. This is the only sanctioned firstmate-initiated git write to the primary, and it is a non-destructive branch switch that strands nothing. +- `STARTUP_MEMORY_BUDGET: invalid config/startup-memory-budget - <reason>` - the visible startup-memory budget is not a safe one-line positive decimal file; do not infer the default or propagate it. Correct the local primary file, then rerun session start so the normal convergence path can deliver the validated value to secondmate homes. - `CREW_DISPATCH: invalid config/crew-dispatch.json - <reason>` - the optional dispatch profile file exists but failed low-cost bootstrap validation; stop profile-based dispatch, report the actionable error, and require correction of the malformed schema, unverified harness name, or invalid harness/effort pair rather than falling back around it or selecting a bad profile. - `FLEET_SYNC: <repo>: skipped: <reason>` - a benign one-off skip (offline, no origin, local-only); bootstrap continued, investigate only if it blocks work. A skip can also report the bounded fleet-refresh timeout (`FM_FLEET_SYNC_BOOTSTRAP_TIMEOUT`, or a fleet-size-aware default with a 20 second floor); a timeout never blocks startup. diff --git a/.agents/skills/secondmate-provisioning/SKILL.md b/.agents/skills/secondmate-provisioning/SKILL.md index f9e68937ab9..44caa0cb1c1 100644 --- a/.agents/skills/secondmate-provisioning/SKILL.md +++ b/.agents/skills/secondmate-provisioning/SKILL.md @@ -78,7 +78,7 @@ This section is the single owner of the secondmate sync and inherited-local-mate Before launch, `fm-spawn.sh --secondmate` locally fast-forwards the home to the primary firstmate checkout's current default-branch commit when it is safe; dirty, diverged, or in-flight homes launch unchanged with a warning. The locked session-start bootstrap sweep runs the same guarded fast-forward for every live secondmate home, discovered from `state/<id>.meta` records with `kind=secondmate` (`data/secondmates.md` only backfills `home=` for older records). That no-fetch path is a purely local fast-forward of tracked files, never an origin fetch, and it never touches the gitignored operational dirs, so a secondmate's backlog, projects, and in-flight work are never disturbed; a linked worktree advances immediately, while a standalone clone that lacks the target receives firstmate updates through `/updatefirstmate`'s origin refresh. -The same launch and the same locked bootstrap sweep also propagate the primary's declared inherited local material: `config/crew-dispatch.json`, `config/crew-harness`, `config/backlog-backend`, `config/backend`, `config/herdr-presentation-spaces`, and the one shared captain-preference file `data/captain-shared.md`. +The same launch and the same locked bootstrap sweep also propagate the primary's declared inherited local material: `config/crew-dispatch.json`, `config/crew-harness`, `config/backlog-backend`, `config/backend`, `config/herdr-presentation-spaces`, `config/startup-memory-budget`, and the one shared captain-preference file `data/captain-shared.md`. Because these paths are gitignored, that propagation is a separate, primary-authoritative copy independent of the tracked-files fast-forward: it re-converges every live home whether or not its tracked files advanced, and it touches only the declared items. Propagation failures warn without blocking secondmate launch or session-start continuation, and the destination keeps whatever safely validated state the helper left behind. Inheritance copies the literal `config/crew-harness` file, so a secondmate's own crewmates use the primary's crewmate harness only when it names a concrete adapter such as `codex`; an unset or `default` value has nothing concrete to inherit, and the secondmate's own crewmates fall back to the secondmate's own or detected harness instead. @@ -99,7 +99,7 @@ Keep every `data/learnings.md` fully local by captain decision; route fleet-gene No AGENTS.md reread nudge is needed at spawn or respawn because the agent reads instructions fresh on launch; only the bootstrap sweep's running-home instruction-surface advance needs that AGENTS.md re-read. Bootstrap reports successful AGENTS.md re-read sends as `BOOTSTRAP_INFO:` and only emits `NUDGE_SECONDMATES:` when that send fails and needs retry. A separate, literal-content config reread is required whenever inherited `config/*` material changes under an already-running secondmate. -After each successful allowlisted config write, both the locked bootstrap convergence path and mid-session `bin/fm-config-push.sh` use the shared propagation report to build one per-home generation-specific private instruction file from the validated destination post-write bytes for only the allowlisted config items that actually changed for that home (`config/crew-dispatch.json`, `config/crew-harness`, `config/backlog-backend`, `config/backend`, `config/herdr-presentation-spaces`), in deterministic allowlist order. +After each successful allowlisted config write, both the locked bootstrap convergence path and mid-session `bin/fm-config-push.sh` use the shared propagation report to build one per-home generation-specific private instruction file from the validated destination post-write bytes for only the allowlisted config items that actually changed for that home (`config/crew-dispatch.json`, `config/crew-harness`, `config/backlog-backend`, `config/backend`, `config/herdr-presentation-spaces`, `config/startup-memory-budget`), in deterministic allowlist order. Each changed path is printed with clear begin/end delimiters and the destination file's full exact new bytes unparsed, or the explicit token `ABSENT` when propagation removed the destination copy. The instruction uses only minimal framing that these are defaults/rules and do not remove judgment; it never includes SHA values, selected profiles, parsed summaries, or any other generated interpretation. `data/captain-shared.md` is not a config file and is never inlined into this instruction file or message. diff --git a/AGENTS.md b/AGENTS.md index ee23c840dc9..c75eeb77591 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -70,6 +70,7 @@ config/secondmate-harness harness the PRIMARY uses to launch SECONDMATE agents, config/backlog-backend backlog backend override; LOCAL, gitignored; absent or "tasks-axi" = default tasks-axi backend, "manual" = force routine backlog updates to hand-editing; inherited by secondmate homes (section 10) config/backend runtime session-provider backend override for new tasks; LOCAL, gitignored; absent = falls through to runtime auto-detection (the runtime firstmate itself is executing inside), then tmux; tmux is the verified reference backend (docs/tmux-backend.md), while herdr, zellij, orca, and cmux are experimental spawn backends (docs/herdr-backend.md, docs/zellij-backend.md, docs/orca-backend.md, docs/cmux-backend.md) - herdr and cmux can also be selected by runtime auto-detection, zellij and orca never are (always explicit), and codex-app is not accepted; see docs/codex-app-backend.md; inherited by secondmate homes under the primary-authoritative contract in secondmate-provisioning config/calm Pi Calm presentation preference; LOCAL, gitignored, and not inherited; see docs/configuration.md "Pi Calm preference" +config/startup-memory-budget primary-authoritative per-home startup-memory budget; LOCAL, gitignored, materialized as 7,500 estimated tokens by locked primary bootstrap and inherited into secondmate homes; see docs/configuration.md "Startup memory budget" config/herdr-presentation-spaces optional presence flag for Herdr's default-off disposable single-task visual projection; LOCAL, gitignored; inherited by secondmate homes; see docs/herdr-backend.md "Optional presentation spaces" config/cmux-socket-password optional cmux control-socket password; LOCAL, gitignored; read fresh on every cmux CLI call and passed through without ever overriding an operator's own ambient CMUX_SOCKET_PASSWORD when absent (docs/cmux-backend.md "Setup") config/wedge-alarm optional away-mode wedge-alarm active-alert directives; LOCAL, gitignored; absent means auto (macOS Notification Center when available); see docs/wedge-alarm.md @@ -473,7 +474,7 @@ It performs guarded fast-forward updates of firstmate and registered secondmate These skills are not captain-invocable; load them only at their precise triggers. -- `bootstrap-diagnostics` - load whenever the session-start digest's bootstrap section prints an actionable diagnostic line (`MISSING:`, `MISSING_MANUAL:`, `BACKEND_INVALID:`, `NEEDS_GH_AUTH`, `TANGLE:`, `CREW_DISPATCH: invalid`, `FLEET_SYNC:`, `PR_CHECK_MIGRATION:`, `SECONDMATE_SYNC:`, `SECONDMATE_LIVENESS:`, `NUDGE_SECONDMATES:`, or `FMX:`); silence and `BOOTSTRAP_INFO:` need no load. +- `bootstrap-diagnostics` - load whenever the session-start digest's bootstrap section prints an actionable diagnostic line (`MISSING:`, `MISSING_MANUAL:`, `BACKEND_INVALID:`, `NEEDS_GH_AUTH`, `TANGLE:`, `STARTUP_MEMORY_BUDGET:`, `CREW_DISPATCH: invalid`, `FLEET_SYNC:`, `PR_CHECK_MIGRATION:`, `SECONDMATE_SYNC:`, `SECONDMATE_LIVENESS:`, `NUDGE_SECONDMATES:`, or `FMX:`); silence and `BOOTSTRAP_INFO:` need no load. - `diagnostic-reasoning` - load before scoping a reported bug and before acting on a diagnostic report. - `ask-user-authority` - load before deciding any ask-user finding, regardless of the project's `yolo` posture. - `quota-array-dispatch` - load before choosing among a matched crew-dispatch profile array from current quota-axi output. diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index fef86ba3830..16102adfa45 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -8,6 +8,7 @@ # Lines: "MISSING: <tool> (install: <command>)", # "MISSING_MANUAL: <tool> (instructions: <url>)", "NEEDS_GH_AUTH", # "BACKEND_INVALID: <name> (known: <names>)", +# "STARTUP_MEMORY_BUDGET: invalid config/startup-memory-budget - <reason>", # "CREW_DISPATCH: invalid config/crew-dispatch.json - <reason>", # "FLEET_SYNC: <repo>: skipped|recovered|STUCK: <detail>", # "PR_CHECK_MIGRATION: <private remediation>", @@ -55,6 +56,11 @@ # tasks-axi default backend is silent. quota-axi is required for the # agent-owned dispatch-profile array procedure in AGENTS.md section 4 # and .agents/skills/quota-array-dispatch/SKILL.md. +# On a primary home, the locked mutable path materializes the visible +# default config/startup-memory-budget=7500 when absent. It never +# guesses at malformed or unsafe existing files, and secondmate homes +# await the primary-authoritative inherited value instead of creating +# their own. # X mode is OPTIONAL and inert unless FM_HOME/.env has a non-empty # FMX_PAIRING_TOKEN. When opted in, bootstrap requires curl+jq, writes # the relay poll shim and 30s cadence config, and prints an FMX line. @@ -97,6 +103,8 @@ DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" . "$SCRIPT_DIR/fm-ff-lib.sh" # shellcheck source=bin/fm-config-inherit-lib.sh disable=SC1091 . "$SCRIPT_DIR/fm-config-inherit-lib.sh" +# shellcheck source=bin/fm-startup-memory-budget-lib.sh disable=SC1091 +. "$SCRIPT_DIR/fm-startup-memory-budget-lib.sh" # shellcheck source=bin/fm-x-lib.sh disable=SC1091 . "$SCRIPT_DIR/fm-x-lib.sh" # shellcheck source=bin/fm-backend.sh disable=SC1091 @@ -805,6 +813,18 @@ crew_dispatch_validate() { fi } +startup_memory_budget_setup() { + # Primary bootstrap owns default publication. A secondmate is deliberately + # passive here because its setting must converge from the primary through the + # inherited-local-material contract rather than becoming a local authority. + if [ -e "$FM_HOME/.fm-secondmate-home" ] || [ -L "$FM_HOME/.fm-secondmate-home" ]; then + return 0 + fi + if ! fm_startup_memory_budget_materialize "$CONFIG"; then + echo "STARTUP_MEMORY_BUDGET: invalid config/$FM_STARTUP_MEMORY_BUDGET_FILE - $FM_STARTUP_MEMORY_BUDGET_ERROR" + fi +} + if [ "${1:-}" = "install" ]; then shift [ $# -gt 0 ] || { echo "usage: fm-bootstrap.sh install <tool>..." >&2; exit 1; } @@ -827,6 +847,7 @@ fi # runnable. Detect-only sessions never touch state. if [ "${FM_BOOTSTRAP_DETECT_ONLY:-0}" != 1 ]; then "$SCRIPT_DIR/fm-pr-check-migrate.sh" || true + startup_memory_budget_setup fi if [ "$BACKEND_VALID" -eq 0 ]; then diff --git a/bin/fm-config-inherit-lib.sh b/bin/fm-config-inherit-lib.sh index 22109aa87aa..bffbd5234d7 100644 --- a/bin/fm-config-inherit-lib.sh +++ b/bin/fm-config-inherit-lib.sh @@ -6,7 +6,8 @@ # profile rules, primary config/crew-harness=codex makes a secondmate's crewmates # spawn on codex too, primary config/backlog-backend=manual makes that home # hand-edit backlog files too, primary config/backend pins that home's local -# runtime-backend default for future spawns, and primary +# runtime-backend default for future spawns, primary config/startup-memory-budget +# bounds that home's startup-memory curation, and primary # config/herdr-presentation-spaces enables the same default-off Herdr presentation # projection). It also pushes the one primary-authoritative shared # captain-preference file, data/captain-shared.md, into each secondmate home's @@ -32,6 +33,9 @@ # is deliberately NOT in the list: it is the primary's own setting for launching # secondmates, and a secondmate never spawns secondmates, so it must not flow # downstream. +# +# shellcheck source=bin/fm-startup-memory-budget-lib.sh +. "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/fm-startup-memory-budget-lib.sh" # The one shared data file in this inheritance contract. There is deliberately # no shared learnings file. @@ -42,7 +46,7 @@ FM_SHARED_CAPTAIN_MODE="444" # The declared inheritable set (space-separated, config-dir-relative item paths). # Extend here to inherit more of the primary's local config; override via the # environment only in tests. Items must not contain whitespace. -FM_INHERITABLE_CONFIG="${FM_INHERITABLE_CONFIG:-crew-dispatch.json crew-harness backlog-backend backend herdr-presentation-spaces}" +FM_INHERITABLE_CONFIG="${FM_INHERITABLE_CONFIG:-crew-dispatch.json crew-harness backlog-backend backend herdr-presentation-spaces startup-memory-budget}" fm_inherit_file_mode() { if [ "$(uname)" = Darwin ]; then @@ -401,6 +405,47 @@ propagate_inheritable_config() { esac src="$src_config/$item" dest="$dest_config/$item" + # This one scalar config is consumed as a local safety boundary, so reject + # every unsafe or malformed source/destination artifact before the generic + # byte-copy behavior below can treat it as ordinary inherited material. + if [ "$item" = "$FM_STARTUP_MEMORY_BUDGET_FILE" ]; then + if [ -e "$src_config" ] || [ -L "$src_config" ]; then + if ! fm_startup_memory_budget_config_dir_safe "$src_config"; then + reason="unsafe primary config directory: $FM_STARTUP_MEMORY_BUDGET_ERROR" + warn_inheritable_config_error "$item" "$src_config" "$reason" + record_inheritable_config_result "$item" error "$reason" + rc=1 + continue + fi + fi + if [ -e "$dest_config" ] || [ -L "$dest_config" ]; then + if ! fm_startup_memory_budget_config_dir_safe "$dest_config"; then + reason="unsafe destination config directory: $FM_STARTUP_MEMORY_BUDGET_ERROR" + warn_inheritable_config_error "$item" "$dest_config" "$reason" + record_inheritable_config_result "$item" error "$reason" + rc=1 + continue + fi + fi + if [ -e "$src" ] || [ -L "$src" ]; then + if ! fm_startup_memory_budget_file_valid "$src"; then + reason="unsafe or invalid primary source: $FM_STARTUP_MEMORY_BUDGET_ERROR" + warn_inheritable_config_error "$item" "$src" "$reason" + record_inheritable_config_result "$item" error "$reason" + rc=1 + continue + fi + fi + if [ -e "$dest" ] || [ -L "$dest" ]; then + if ! fm_startup_memory_budget_file_valid "$dest"; then + reason="unsafe or invalid destination: $FM_STARTUP_MEMORY_BUDGET_ERROR" + warn_inheritable_config_error "$item" "$dest" "$reason" + record_inheritable_config_result "$item" error "$reason" + rc=1 + continue + fi + fi + fi if [ -f "$src" ]; then if ! destination_allows_inherited_item "$dest_config" "$item"; then reason=$(inheritable_config_skip_reason) diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 314232ae88b..f89a6bade50 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -145,6 +145,7 @@ family_for_basename() { ;; fm-backlog-handoff.test.sh|fm-secondmate-harness.test.sh|fm-secondmate-lifecycle-e2e.test.sh|\ fm-secondmate-liveness.test.sh|fm-secondmate-safety.test.sh|fm-secondmate-sync.test.sh|\ + fm-startup-memory-budget.test.sh|\ fm-send-secondmate-marker.test.sh|fm-shared-captain-inheritance.test.sh) printf '%s\n' secondmate ;; @@ -644,6 +645,10 @@ families_for_changed_path() { printf '%s\n' live-harness-optin printf '%s\n' afk ;; + bin/fm-startup-memory-budget.sh|bin/fm-startup-memory-budget-lib.sh) + printf '%s\n' secondmate + printf '%s\n' session-bootstrap + ;; bin/fm-secondmate*|bin/fm-home-seed.sh|bin/fm-backlog-handoff.sh|\ bin/fm-config-inherit-lib.sh|bin/fm-config-push.sh|bin/fm-shared*) printf '%s\n' secondmate diff --git a/docs/configuration.md b/docs/configuration.md index b2f80b8e6dc..b226ec6888c 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -137,6 +137,20 @@ Fleet-local operational facts and gotchas live locally in `data/learnings.md`; i The file is created lazily on first learning and follows the same dated, evidence-backed, curated style as `data/captain.md`: inspect the current file first, then rewrite or prune stale entries instead of appending forever. There is no shared learnings file by captain decision. +## Startup memory budget (config/startup-memory-budget) + +`config/startup-memory-budget` is the primary-authoritative per-home allowance for the startup prompt-memory surface: `data/captain.md`, `data/captain-shared.md`, and `data/learnings.md` together. +The locked mutable bootstrap path materializes its visible default of `7500` estimated tokens in a primary home when the file is absent. +To select another allowance, replace the primary home's file with one valid positive value in the exact format below; the next locked bootstrap convergence or `bin/fm-config-push.sh` propagates it to registered secondmates. +A secondmate does not create an independent default and instead receives the primary value through the inherited-local-material contract in [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md). +The file must be one positive base-10 integer followed by exactly one newline in a regular, single-linked file beneath a non-symlinked `config/` directory. +Malformed, multi-line, symlinked, hardlinked, special, or otherwise unsafe values are rejected rather than treated as a default. +Use `bin/fm-startup-memory-budget.sh read` to validate and print the effective value, or `bin/fm-startup-memory-budget.sh report` to account for the three files. +The stable local estimate is `ceil(UTF-8 bytes / 3)` per file, a conservative portable approximation rather than a provider-exact tokenizer. +An inherited `data/captain-shared.md` counts in a secondmate's total but remains primary-owned and read-only there. +The internal `/stow` skill curates only the editable local files in that case and reports the primary-owned shared file as a concrete exception if it alone exceeds the budget. +The helper's header owns exact parsing, publication, and report output mechanics. + ## Secondmate routes (data/secondmates.md) Persistent secondmate routes live locally in `data/secondmates.md`. @@ -285,7 +299,7 @@ When a running home advances and its loaded instruction surface (`AGENTS.md`, `b If that send fails, bootstrap keeps an idempotent retry marker and emits `NUDGE_SECONDMATES:` with the failure reason. The same bootstrap run emits `SECONDMATE_LIVENESS:` only when a registered secondmate is skipped or its relaunch fails; already-live and successfully relaunched secondmates are handled silently. For a mid-session inherited local-material edit where tracked-file sync is not needed, run `bin/fm-config-push.sh`. -It uses the same live secondmate discovery and propagation helper as bootstrap, prints each live home's `crew-dispatch.json`, `crew-harness`, `backlog-backend`, `backend`, `herdr-presentation-spaces`, and `data/captain-shared.md` result as `pushed`, `unchanged`, `skipped`, or `error`, and exits non-zero for real propagation errors or config-reread send failures. +It uses the same live secondmate discovery and propagation helper as bootstrap, prints each live home's `crew-dispatch.json`, `crew-harness`, `backlog-backend`, `backend`, `herdr-presentation-spaces`, `startup-memory-budget`, and `data/captain-shared.md` result as `pushed`, `unchanged`, `skipped`, or `error`, and exits non-zero for real propagation errors or config-reread send failures. When an allowlisted config item changes for an already-running home, it sends the literal-content reread pointer described in [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md); unchanged allowlisted config sends no pointer unless a previous delivery is pending. The locked bootstrap inheritance pass uses the same per-home changed-set and reread path for already-running homes; see `secondmate-provisioning` for the single contract owner. That live discovery starts from `state/*.meta` records with `kind=secondmate`; `data/secondmates.md` only backfills `home=` for older or incomplete meta records. diff --git a/docs/documentation-audiences.json b/docs/documentation-audiences.json index 54b2190f6c8..8f70cd346c2 100644 --- a/docs/documentation-audiences.json +++ b/docs/documentation-audiences.json @@ -307,6 +307,10 @@ "path": "docs/verification/runtime-backends.md", "audience": "maintainer-verification" }, + { + "path": "docs/verification/stow-memory.md", + "audience": "maintainer-verification" + }, { "path": "docs/verification/supervision.md", "audience": "maintainer-verification" diff --git a/tests/fm-backend-herdr-presentation-e2e.test.sh b/tests/fm-backend-herdr-presentation-e2e.test.sh index 158e943211c..fb44305e99e 100755 --- a/tests/fm-backend-herdr-presentation-e2e.test.sh +++ b/tests/fm-backend-herdr-presentation-e2e.test.sh @@ -866,7 +866,7 @@ touch "$SECOND_HOME_A/state/.last-watcher-beat" "$SECOND_HOME_B/state/.last-watc # may write config/herdr-presentation-spaces. git -C "$SECOND_HOME_A" init -q git -C "$SECOND_HOME_B" init -q -printf 'config/herdr-presentation-spaces\nconfig/crew-harness\nconfig/crew-dispatch.json\nconfig/backlog-backend\nconfig/backend\n' \ +printf 'config/herdr-presentation-spaces\nconfig/crew-harness\nconfig/crew-dispatch.json\nconfig/backlog-backend\nconfig/backend\nconfig/startup-memory-budget\n' \ > "$SECOND_HOME_A/.gitignore" cp "$SECOND_HOME_A/.gitignore" "$SECOND_HOME_B/.gitignore" git -C "$SECOND_HOME_A" add .gitignore diff --git a/tests/fm-secondmate-harness.test.sh b/tests/fm-secondmate-harness.test.sh index ae41c793516..446dcd93a17 100755 --- a/tests/fm-secondmate-harness.test.sh +++ b/tests/fm-secondmate-harness.test.sh @@ -14,7 +14,8 @@ # explicit per-spawn harness arg still wins. # B) Inheritance. The primary pushes a declared, extensible set of LOCAL # (gitignored) config items - config/crew-dispatch.json, config/crew-harness, -# config/backlog-backend, config/backend, and config/herdr-presentation-spaces - +# config/backlog-backend, config/backend, config/herdr-presentation-spaces, and +# config/startup-memory-budget - # down into each secondmate home's config/, so the secondmate's OWN crewmates, # dispatch profiles, backlog backend, runtime-backend default, and Herdr # presentation opt-in inherit the primary's settings. It is primary-authoritative @@ -167,7 +168,7 @@ SH [ "$got" = pi ] || fail "selected plain Pi resolved '$got', expected pi" got=$(PATH="$fakebin:$BASE_PATH" PI_CODING_AGENT=true FM_PI_HARNESS=pi-signed-helper "$ROOT/bin/fm-harness.sh") [ "$got" = pi ] || fail "inexact signed selection marker resolved '$got', expected pi" - got=$(PATH="$fakebin:$BASE_PATH" FM_PI_HARNESS=pi-signed "$ROOT/bin/fm-harness.sh") + got=$(env -u PI_CODING_AGENT PATH="$fakebin:$BASE_PATH" FM_PI_HARNESS=pi-signed "$ROOT/bin/fm-harness.sh") [ "$got" = pi ] || fail "signed selection marker without Pi's family marker resolved '$got', expected pi" got=$(PATH="$fakebin:$BASE_PATH" PI_CODING_AGENT=true FM_TEST_SIGNED_SHAPE=plain "$ROOT/bin/fm-harness.sh") [ "$got" = pi ] || fail "plain Pi marker resolved '$got', expected pi" @@ -883,7 +884,7 @@ new_world() { printf 'projects/\nstate/\ndata/\n.no-mistakes/\n' [ "$dispatch_ignore" = no ] || printf 'config/crew-dispatch.json\n' printf 'config/crew-harness\nconfig/secondmate-harness\nconfig/backlog-backend\n' - printf 'config/backend\nconfig/herdr-presentation-spaces\n' + printf 'config/backend\nconfig/herdr-presentation-spaces\nconfig/startup-memory-budget\n' } > "$w/main/.gitignore" printf 'v1\n' > "$w/main/AGENTS.md" printf 'r1\n' > "$w/main/README.md" @@ -1176,10 +1177,10 @@ test_bootstrap_sweep_defers_dispatch_on_stale_unignored_home() { pass "B9 bootstrap sweep defers new inherited config until the home ignores it" } -# Backward-compat: with no inherited config set, the sweep is a no-op for the -# home's config/ - exactly as before this feature - and ordinary sweep behavior -# (fast-forward) is unaffected. -test_bootstrap_sweep_no_inheritance_is_noop() { +# The primary bootstrap always materializes the startup-memory default, so an +# otherwise empty inherited surface converges that one visible value while +# ordinary tracked-file fast-forward behavior remains unchanged. +test_bootstrap_sweep_materializes_and_inherits_memory_default() { local w c1 w=$(new_world boot-noop) c1=$(git -C "$w/main" rev-parse HEAD) @@ -1193,13 +1194,16 @@ test_bootstrap_sweep_no_inheritance_is_noop() { run_bootstrap "$w" >/dev/null - [ -e "$w/sm/config/crew-dispatch.json" ] && fail "no-inheritance sweep created a home crew-dispatch.json" - [ -e "$w/sm/config/crew-harness" ] && fail "no-inheritance sweep created a home crew-harness" - [ -e "$w/sm/config/backend" ] && fail "no-inheritance sweep created a home backend" - [ -e "$w/sm/config" ] && fail "no-inheritance sweep created a home config/ dir" + [ -e "$w/sm/config/crew-dispatch.json" ] && fail "default-only sweep created a home crew-dispatch.json" + [ -e "$w/sm/config/crew-harness" ] && fail "default-only sweep created a home crew-harness" + [ -e "$w/sm/config/backend" ] && fail "default-only sweep created a home backend" + [ "$(cat "$w/home/config/startup-memory-budget")" = 7500 ] \ + || fail "primary bootstrap did not materialize the startup-memory default" + [ "$(cat "$w/sm/config/startup-memory-budget")" = 7500 ] \ + || fail "default-only sweep did not converge startup-memory-budget" [ "$(git -C "$w/sm" rev-parse HEAD)" = "$head" ] \ - || fail "no-inheritance sweep did not still fast-forward the tracked files" - pass "B10 bootstrap sweep with no inherited config is a config no-op and still fast-forwards" + || fail "default-only sweep did not still fast-forward the tracked files" + pass "B10 bootstrap sweep materializes and inherits the startup-memory default while fast-forwarding" } # config/backend: present and absent primary state converges exactly. @@ -2193,6 +2197,7 @@ cat > "$w/main/bin/fm-spawn.sh" <<SH . '$w/main/bin/fm-config-inherit-lib.sh' printf '%s' spawn >> '$log' printf '%s' codex > '$w/sm/config/crew-harness' +printf '%s\n' 7500 > '$w/sm/config/startup-memory-budget' SH chmod +x "$w/main/bin/fm-spawn.sh" fakebin=$(make_fake_toolchain "$w") @@ -2317,7 +2322,7 @@ test_spawn_fallback_chain_and_crew_scout_unaffected test_bootstrap_sweep_propagates_and_reconverges test_bootstrap_sweep_propagates_when_tracked_current test_bootstrap_sweep_defers_dispatch_on_stale_unignored_home -test_bootstrap_sweep_no_inheritance_is_noop +test_bootstrap_sweep_materializes_and_inherits_memory_default test_backend_inheritance_present_and_absent test_bootstrap_sweep_surfaces_config_propagation_failure test_bootstrap_rereads_after_partial_propagation From 947a6762f274e98859bb3c3e8dc4649373564b81 Mon Sep 17 00:00:00 2001 From: juniorlovestmh <272474227+juniorlovestmh@users.noreply.github.com> Date: Thu, 30 Jul 2026 17:47:46 -0300 Subject: [PATCH 43/52] no-mistakes(review): Captain: restored ADHD coverage and Doppler CLI validation --- bin/fm-test-run.sh | 1 + tests/fm-captain-translation-contract.test.sh | 26 +++++++++++++++++++ tests/fm-secrets-check.test.sh | 10 +++++++ 3 files changed, 37 insertions(+) create mode 100755 tests/fm-captain-translation-contract.test.sh diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index f89a6bade50..01df68c08a5 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -123,6 +123,7 @@ family_for_basename() { fm-crew-state.test.sh|fm-decision-hold-lifecycle.test.sh|\ fm-documentation-audiences.test.sh|fm-ensure-agents-md.test.sh|fm-grok-harness.test.sh|\ fm-kimi-harness.test.sh|fm-herdr-lab.test.sh|fm-lint.test.sh|\ + fm-secrets-check.test.sh|fm-captain-translation-contract.test.sh|\ fm-operational-input.test.sh|fm-pi-primary-types.test.sh|\ fm-send-popup-settle.test.sh|fm-send-settle.test.sh|\ fm-subagent-pretool-check.test.sh|\ diff --git a/tests/fm-captain-translation-contract.test.sh b/tests/fm-captain-translation-contract.test.sh new file mode 100755 index 00000000000..1a15404f7ea --- /dev/null +++ b/tests/fm-captain-translation-contract.test.sh @@ -0,0 +1,26 @@ +#!/usr/bin/env bash +set -u + +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +BRIEF="$ROOT/bin/fm-brief.sh" +TMP_ROOT=$(fm_test_tmproot fm-captain-translation) + +test_captain_facing_scout_path_preserves_evidence_and_action() { + local home id report + home="$TMP_ROOT/home" + id="captain-translation" + + FM_HOME="$home" "$BRIEF" "$id" firstmate --scout >/dev/null 2>&1 \ + || fail "scout brief generation failed" + report="$home/data/$id/brief.md" + assert_present "$report" "scout brief was not generated" + assert_grep '# Definition of done' "$report" "scout brief lacks a completion contract" + assert_grep 'what you did, what you found, the evidence' "$report" \ + "scout brief does not require concrete evidence in its report" + assert_grep 'what you recommend' "$report" \ + "scout brief does not require a concrete recommendation" + pass "captain-facing scout work preserves evidence and recommendation handoff" +} + +test_captain_facing_scout_path_preserves_evidence_and_action diff --git a/tests/fm-secrets-check.test.sh b/tests/fm-secrets-check.test.sh index 508f4003181..f2270391f6a 100755 --- a/tests/fm-secrets-check.test.sh +++ b/tests/fm-secrets-check.test.sh @@ -428,6 +428,16 @@ PY job_key='deploy' workflow_trigger='' ;; + *cli-owned) + runner='[self-hosted, Linux, X64, fleet-ci]' + job_key='deploy' + command='doppler run -- ./scripts/deploy.sh' + ;; + *cli-hosted) + runner='ubuntu-latest' + job_key='deploy' + command='doppler run -- ./scripts/deploy.sh' + ;; *) runner='ubuntu-latest' job_key='deploy' From d00c0d4ef31c8977cdacd76f8b0b317d69f336b7 Mon Sep 17 00:00:00 2001 From: juniorlovestmh <272474227+juniorlovestmh@users.noreply.github.com> Date: Thu, 30 Jul 2026 19:43:18 -0300 Subject: [PATCH 44/52] fix: satisfy pinned shellcheck for merge tests --- tests/fm-captain-translation-contract.test.sh | 1 + tests/fm-secrets-check.test.sh | 10 ---------- 2 files changed, 1 insertion(+), 10 deletions(-) diff --git a/tests/fm-captain-translation-contract.test.sh b/tests/fm-captain-translation-contract.test.sh index 1a15404f7ea..142b160dc29 100755 --- a/tests/fm-captain-translation-contract.test.sh +++ b/tests/fm-captain-translation-contract.test.sh @@ -1,6 +1,7 @@ #!/usr/bin/env bash set -u +# shellcheck source=tests/lib.sh . "$(dirname "${BASH_SOURCE[0]}")/lib.sh" BRIEF="$ROOT/bin/fm-brief.sh" diff --git a/tests/fm-secrets-check.test.sh b/tests/fm-secrets-check.test.sh index f2270391f6a..508f4003181 100755 --- a/tests/fm-secrets-check.test.sh +++ b/tests/fm-secrets-check.test.sh @@ -428,16 +428,6 @@ PY job_key='deploy' workflow_trigger='' ;; - *cli-owned) - runner='[self-hosted, Linux, X64, fleet-ci]' - job_key='deploy' - command='doppler run -- ./scripts/deploy.sh' - ;; - *cli-hosted) - runner='ubuntu-latest' - job_key='deploy' - command='doppler run -- ./scripts/deploy.sh' - ;; *) runner='ubuntu-latest' job_key='deploy' From c811937e36697b6ffc9b76fa5defdea1b0d876bd Mon Sep 17 00:00:00 2001 From: juniorlovestmh <272474227+juniorlovestmh@users.noreply.github.com> Date: Sat, 1 Aug 2026 08:54:23 -0300 Subject: [PATCH 45/52] feat: wake firstmate for Better Stack incidents --- .agents/skills/bootstrap-diagnostics/SKILL.md | 3 + .opencode/plugins/fm-primary-watch-arm.js | 1 + AGENTS.md | 20 +- bin/fm-better-stack-incidents-poll.sh | 164 ++++++++++++ bin/fm-bootstrap.sh | 113 +++++++- bin/fm-claude-stop-autoarm.sh | 6 +- bin/fm-session-start.sh | 14 +- bin/fm-subagent-pretool-check.sh | 2 +- bin/fm-supervision-lib.sh | 16 +- bin/fm-test-run.sh | 2 +- bin/fm-turnend-guard.sh | 8 + docs/architecture.md | 3 + docs/configuration.md | 24 +- docs/subagent-guard.md | 4 +- docs/supervision-protocols/grok.md | 2 +- docs/turnend-guard.md | 8 +- tests/fm-better-stack-incidents.test.sh | 246 ++++++++++++++++++ 17 files changed, 595 insertions(+), 41 deletions(-) create mode 100755 bin/fm-better-stack-incidents-poll.sh create mode 100755 tests/fm-better-stack-incidents.test.sh diff --git a/.agents/skills/bootstrap-diagnostics/SKILL.md b/.agents/skills/bootstrap-diagnostics/SKILL.md index 477980b8df1..e9a57058a55 100644 --- a/.agents/skills/bootstrap-diagnostics/SKILL.md +++ b/.agents/skills/bootstrap-diagnostics/SKILL.md @@ -32,6 +32,9 @@ When any diagnostic needs captain attention, report the plain consequence and re - `FLEET_SYNC: <repo>: skipped: <reason>` - a benign one-off skip (offline, no origin, local-only); bootstrap continued, investigate only if it blocks work. A skip can also report the bounded fleet-refresh timeout (`FM_FLEET_SYNC_BOOTSTRAP_TIMEOUT`, or a fleet-size-aware default with a 20 second floor); a timeout never blocks startup. - `FLEET_SYNC: <repo>: recovered: <detail>` - the clone had drifted onto a clean detached HEAD holding no unique commits and the sync self-healed it (re-attached the default branch and fast-forwarded); no action needed, it is reported only so the self-heal is visible. +- `BETTER_STACK: incident monitoring on ...` - the home-scoped poll is registered at the default check cadence; no action is needed. +- `BETTER_STACK: incident monitoring off - removed ...` - the local presence flag was removed and bootstrap retired the runnable check while retaining incident dedupe state; no action is needed. +- Any other `BETTER_STACK:` line - follow its concrete dependency, unsafe flag, activation, or cleanup diagnostic before relying on incident monitoring. - `FLEET_SYNC: <repo>: STUCK: on <state>, N commits behind <base> - needs attention` - the clone is dirty, on a non-default branch, detached with unique commits, or diverged, so the sync left it untouched (never forcing or discarding); it will keep falling behind until you look. A loud STUCK, especially a growing N across bootstraps, means that clone needs hands-on attention; dispatch a crewmate or resolve it before it strands work. - `PR_CHECK_MIGRATION: canonical polls rebuilt and armed; resume supervision for this home` - the non-executing migration rebuilt canonical task polls from validated metadata, and those polls are already armed. diff --git a/.opencode/plugins/fm-primary-watch-arm.js b/.opencode/plugins/fm-primary-watch-arm.js index 433edb80ab4..a8ae265ed2e 100644 --- a/.opencode/plugins/fm-primary-watch-arm.js +++ b/.opencode/plugins/fm-primary-watch-arm.js @@ -103,6 +103,7 @@ async function isPrimaryRoot(root, home) { function shouldArm(paths) { if (existsSync(`${paths.state}/.afk`)) return false; if (existsSync(`${paths.config}/x-mode.env`)) return true; + if (existsSync(`${paths.state}/better-stack-incidents.check.sh`)) return true; try { return readdirSync(paths.state).some((name) => name.endsWith(".meta")); } catch { diff --git a/AGENTS.md b/AGENTS.md index c75eeb77591..708ff050d39 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -74,6 +74,7 @@ config/startup-memory-budget primary-authoritative per-home startup-memory b config/herdr-presentation-spaces optional presence flag for Herdr's default-off disposable single-task visual projection; LOCAL, gitignored; inherited by secondmate homes; see docs/herdr-backend.md "Optional presentation spaces" config/cmux-socket-password optional cmux control-socket password; LOCAL, gitignored; read fresh on every cmux CLI call and passed through without ever overriding an operator's own ambient CMUX_SOCKET_PASSWORD when absent (docs/cmux-backend.md "Setup") config/wedge-alarm optional away-mode wedge-alarm active-alert directives; LOCAL, gitignored; absent means auto (macOS Notification Center when available); see docs/wedge-alarm.md +config/better-stack-incidents optional presence flag for the home-scoped Better Stack incident poll; LOCAL, gitignored, and not inherited; see docs/configuration.md "Better Stack incident monitoring" config/x-mode.env generated X-mode watcher cadence; LOCAL, gitignored; source before arming watcher when present data/ personal fleet records; LOCAL, gitignored as a whole backlog.md task queue, dependencies, history @@ -101,6 +102,8 @@ state/ volatile runtime signals; gitignored .pr-check-migration.log private per-task outcomes distinguishing rebuilt or canonically registered replacement polls, quarantined unarmed polls, and incomplete migrations .pr-check-migration-scan-v1 private marker proving the non-executing scan disabled every unsafe legacy check; .pr-check-migration-v1 separately records completed private repairs x-watch.check.sh generated X-mode relay poll shim; present only when opted in (section 14) + better-stack-incidents.check.sh better-stack-incidents.check-trust generated and registered home-scoped Better Stack incident poll; present only when opted in + better-stack-incidents.seen/ better-stack-incidents.diagnostics/ private incident-ID and diagnostic dedupe state retained across poll disable/re-enable pending-replies/ parent-owned secondmate pending-reply records (correlation id, delivery vs reply, recovery, escalation); fm-pending-reply-lib.sh x-inbox/ generated X-mode pending mention payloads; fmx-respond drains it (section 14) x-context/ generated X-mode durable per-request reply context and one-wake offer markers, keyed by request_id; survives inbox cleanup and expires within seven days (section 14; bin/fm-x-lib.sh) @@ -138,7 +141,7 @@ A lock-refused session must not spawn, steer, merge, drain the wake queue, repai 1. **Lock** - acquires the per-home session lock first, before anything mutates shared state. 2. **Bootstrap** - detect-only checks (tool/version problems, GitHub auth, the worktree-tangle check, harness override, dispatch-profile validation, backlog-backend status) always run, but routine confirmations stay silent by default. When the lock could not be acquired, the worktree-tangle check uses read-only advisory wording without a checkout repair command. - Home-local stale Herdr projection cleanup and the five bootstrap MUTATING sweeps - non-executing legacy PR-check migration, fleet sync, the local secondmate fast-forward sweep, the secondmate liveness sweep, and X-mode artifact writes - run only when this session actually holds the lock from step 1. + Home-local stale Herdr projection cleanup and the six bootstrap MUTATING sweeps - non-executing legacy PR-check migration, fleet sync, the local secondmate fast-forward sweep, the secondmate liveness sweep, X-mode artifact writes, and Better Stack incident-poll registration - run only when this session actually holds the lock from step 1. The secondmate liveness sweep deterministically accounts for every registered secondmate: it relaunches only from the recovery-grade `dead` or `missing` states, preserves ambiguous or unreadable targets, and reports skipped or failed guarantees as `SECONDMATE_LIVENESS:` lines (`bin/fm-bootstrap.sh`; `bin/fm-backend.sh`'s `fm_backend_agent_state`). 3. **Wake queue** - when locked, drains the durable wake queue and prints the raw records prominently as this turn's first work queue; a bounded, clearly labeled historical status-event annotation may follow a valid `signal` record but never replaces it or current-state reconciliation, and a lapsed watcher chain still surfaces here via the same guard alarm. When the lock could not be acquired and verified, the queue is left untouched because no session mutation is authorized, and the guard's tangle/watcher-liveness alarms still print in read-only advisory mode without drain, supervision repair, or checkout repair commands. @@ -147,7 +150,7 @@ A lock-refused session must not spawn, steer, merge, drain the wake queue, repai 5. **Fleet-state digest** - the compact backlog listing owned by `bin/fm-session-start.sh`; every `state/<id>.meta`; a bounded tail of each task's `state/<id>.status` (labeled as wake-EVENT history, not current state, with the full log path printed for a deeper read); the `state/.afk` flag; and one cheap alive/dead read of each task's recorded backend endpoint. That liveness line is a fast presence check only, not a full state read - when you need a crew's actual current state (a run-step, not just "is the pane there"), read it with `bin/fm-crew-state.sh <id>` as before; the digest deliberately skips that deeper, slower read for every task so it stays fast and bounded. 6. **Supervision operating instructions and next step** - after the wake queue and before context, the digest emits exactly one operating block for the detected primary harness. - The closing reminder points back to that emitted block and preserves only the lock, afk, X-mode, and read-once reminders. + The closing reminder points back to that emitted block and preserves only the lock, afk, home-monitoring, and read-once reminders. The script itself never starts supervision; the emitted harness protocol owns the exact wait or wake mechanism. Bootstrap detects first, asks for consent, and installs only after the captain approves in the current session. @@ -337,7 +340,7 @@ The promoted worker must inventory scratch state, return to a clean default-bran Fleet supervision is an always-loaded operational contract; `docs/architecture.md`, `docs/turnend-guard.md`, the emitted session-start block, and script help own mechanisms and harness-specific recipes. Whenever work is under way, keep exactly one live supervision cycle using the emitted protocol for this primary harness. -X mode may require that same live cycle with no fleet work. +X mode or Better Stack incident monitoring may require that same live cycle with no fleet work. Do not substitute another harness's wait shape, use shell `&`, or create a second cycle when a healthy one already exists. For every actionable wake, follow the ordinary-wake continuation in the emitted protocol; use its repair action only when the live cycle is missing or failed. No turn ends blind while work is under way, including turns described as holding or waiting. @@ -351,9 +354,14 @@ Handle actionable wakes as follows: 1. For `signal:`, read the listed event lines first, then reconcile current state only where action depends on it. 2. For `stale:`, inspect the recorded endpoint and load `stuck-crewmate-recovery` for a stopped, looping, confused, or unresponsive worker; a deep-inspection reason also requires current-state and validation-log inspection. -3. For `check:`, act on the named poll result, including merges and X-mode events. +3. For `check:`, act on the named poll result, including merges, X-mode events, and Better Stack incidents or diagnostics. 4. For `heartbeat:`, review the whole fleet from the structured fleet view, reconcile suspicious tasks and PR state, update the backlog, and never report an unchanged fleet as progress. +For a `better-stack-incident opened ...` or `better-stack-incidents opened ...` result, load `diagnostic-reasoning` before scoping the response. +When the delivery ladder caused the breakage, restore service through the available rollback path first and investigate after recovery. +Page the captain immediately only for security-shaped, irreversible, or product-affecting incidents; otherwise carry the result and resolution in the next outcome digest. +Better Stack polling during `heartbeat:` handling was an interim practice and is retired; the registered home check is its only poll owner. + When any wake reports a merged PR for a project cloned in this home, refresh that clone through the guarded fleet-sync path. When X-linked work reaches a milestone or terminal state, load `fmx-respond`; before terminal teardown, always post the final completion follow-up so the link clears even if earlier follow-ups were spent. @@ -474,7 +482,7 @@ It performs guarded fast-forward updates of firstmate and registered secondmate These skills are not captain-invocable; load them only at their precise triggers. -- `bootstrap-diagnostics` - load whenever the session-start digest's bootstrap section prints an actionable diagnostic line (`MISSING:`, `MISSING_MANUAL:`, `BACKEND_INVALID:`, `NEEDS_GH_AUTH`, `TANGLE:`, `STARTUP_MEMORY_BUDGET:`, `CREW_DISPATCH: invalid`, `FLEET_SYNC:`, `PR_CHECK_MIGRATION:`, `SECONDMATE_SYNC:`, `SECONDMATE_LIVENESS:`, `NUDGE_SECONDMATES:`, or `FMX:`); silence and `BOOTSTRAP_INFO:` need no load. +- `bootstrap-diagnostics` - load whenever the session-start digest's bootstrap section prints an actionable diagnostic line (`MISSING:`, `MISSING_MANUAL:`, `BACKEND_INVALID:`, `NEEDS_GH_AUTH`, `TANGLE:`, `STARTUP_MEMORY_BUDGET:`, `CREW_DISPATCH: invalid`, `FLEET_SYNC:`, `PR_CHECK_MIGRATION:`, `SECONDMATE_SYNC:`, `SECONDMATE_LIVENESS:`, `NUDGE_SECONDMATES:`, `FMX:`, or `BETTER_STACK:`); silence and `BOOTSTRAP_INFO:` need no load. - `diagnostic-reasoning` - load before scoping a reported bug and before acting on a diagnostic report. - `ask-user-authority` - load before deciding any ask-user finding, regardless of the project's `yolo` posture. - `quota-array-dispatch` - load before choosing among a matched crew-dispatch profile array from current quota-axi output. @@ -495,7 +503,7 @@ X mode ships inert and causes no behavior change until the home opts in by placi That token is consent for public replies and normal reversible lifecycle actions from eligible mentions, not authority for destructive, irreversible, or security-sensitive action; those still require trusted-channel confirmation. `docs/configuration.md` owns activation, generated state, cadence, wire protocol, and opt-out mechanics. -An X-only home still requires the live supervision cycle so mentions can wake it without fleet work. +A home with X mode or Better Stack incident monitoring still requires the live supervision cycle without fleet work. On an `x-mention <request_id>` or `x-mode-error ...` check wake, load `fmx-respond`, which owns classification, public-safety policy, reply or dismissal, task linking, and follow-ups. For every X-linked terminal outcome, load that owner and post the final completion follow-up before teardown, regardless of earlier milestone follow-ups. diff --git a/bin/fm-better-stack-incidents-poll.sh b/bin/fm-better-stack-incidents-poll.sh new file mode 100755 index 00000000000..e7085bbde4e --- /dev/null +++ b/bin/fm-better-stack-incidents-poll.sh @@ -0,0 +1,164 @@ +#!/usr/bin/env bash +# Poll Better Stack for unresolved incidents through the fleet-observability/prd +# Doppler config. +# +# Usage: fm-better-stack-incidents-poll.sh +# +# The public entrypoint always launches itself through Doppler with fallback +# files disabled and only BETTER_STACK_API_TOKEN injected. +# The internal --from-doppler mode is used only by that child process and tests. +# +# Output is the authenticated custom-check contract consumed by fm-watch.sh: +# better-stack-incident opened id=<id> name=<name> started=<timestamp> +# better-stack-incidents opened ids=<comma-separated ids> +# better-stack-error <deduplicated diagnostic> +# A quiet or already-seen result prints nothing. +# The watcher provides the outer FM_CHECK_TIMEOUT; curl stays within five +# seconds so the check finishes with margin. +set -u + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" +FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" +STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" +ERROR_DIR="$STATE/better-stack-incidents.diagnostics" +ERROR_FILE="$ERROR_DIR/error" +SEEN_DIR="$STATE/better-stack-incidents.seen" + +# Reuse the watcher's existing private-artifact owner rather than introducing a +# second atomic-publication contract for one extension. +# shellcheck source=bin/fm-x-lib.sh disable=SC1091 +. "$SCRIPT_DIR/fm-x-lib.sh" + +emit_error_once() { + local msg=$1 + if fmx_private_artifact_file_valid "$ERROR_DIR" error 600 \ + && [ "$(cat "$ERROR_FILE" 2>/dev/null)" = "$msg" ]; then + return 0 + fi + printf '%s\n' "$msg" \ + | fmx_private_artifact_publish_stdin "$ERROR_DIR" error 600 2>/dev/null || true + printf 'better-stack-error %s\n' "$msg" +} + +clear_error() { + fmx_private_artifact_file_valid "$ERROR_DIR" error 600 || return 0 + rm -f -- "$ERROR_FILE" 2>/dev/null || true +} + +run_through_doppler() { + local out rc + command -v doppler >/dev/null 2>&1 \ + || { emit_error_once "missing doppler"; return 0; } + out=$(doppler run \ + --silent \ + --no-check-version \ + --no-fallback \ + --project fleet-observability \ + --config prd \ + --only-secrets BETTER_STACK_API_TOKEN \ + -- "$SCRIPT_DIR/fm-better-stack-incidents-poll.sh" --from-doppler 2>/dev/null) + rc=$? + if [ "$rc" -ne 0 ]; then + emit_error_once "Doppler access unavailable for fleet-observability/prd" + return 0 + fi + case "$out" in + '') return 0 ;; + *$'\n'*) emit_error_once "poll returned invalid multiline output" ;; + better-stack-incident\ opened\ *|better-stack-incidents\ opened\ *|better-stack-error\ *) + printf '%s\n' "$out" + ;; + *) emit_error_once "poll returned invalid output" ;; + esac +} + +claim_incident() { + local id=$1 rc + printf 'seen\n' \ + | fmx_private_artifact_publish_stdin_once "$SEEN_DIR" "$id" 600 2>/dev/null + rc=$? + return "$rc" +} + +poll_with_injected_token() { + local token=${BETTER_STACK_API_TOKEN:-} raw code body rows id name started + local claim_rc new_count=0 new_ids= first_id= first_name= first_started= + + [ -n "$token" ] || { emit_error_once "missing BETTER_STACK_API_TOKEN"; return 0; } + [[ "$token" =~ ^[A-Za-z0-9._~+/=-]+$ ]] \ + || { emit_error_once "invalid BETTER_STACK_API_TOKEN"; return 0; } + command -v curl >/dev/null 2>&1 || { emit_error_once "missing curl"; return 0; } + command -v jq >/dev/null 2>&1 || { emit_error_once "missing jq"; return 0; } + + raw=$(printf 'header = "Authorization: Bearer %s"\n' "$token" \ + | curl \ + --config - \ + --request GET \ + --url 'https://uptime.betterstack.com/api/v3/incidents?resolved=false&per_page=50' \ + --header 'Accept: application/json' \ + --connect-timeout 3 \ + --max-time 5 \ + --silent \ + --show-error \ + --write-out '\n%{http_code}' 2>/dev/null) \ + || { emit_error_once "Better Stack API unreachable"; return 0; } + case "$raw" in + *$'\n'*) ;; + *) emit_error_once "Better Stack API returned no status"; return 0 ;; + esac + code=${raw##*$'\n'} + body=${raw%$'\n'*} + [ "$code" = 200 ] || { emit_error_once "API returned HTTP $code"; return 0; } + + rows=$(printf '%s' "$body" | jq -r ' + if (.data | type) != "array" then error("data must be an array") else .data[] end + | select(.type == "incident") + | select(.attributes.resolved_at == null) + | select(.id | type == "string" and test("^[0-9]+$")) + | [ + .id, + ((.attributes.name // "unknown") | tostring | gsub("[[:space:][:cntrl:]]+"; " ") | .[0:120]), + ((.attributes.started_at // "unknown") | tostring | gsub("[[:space:][:cntrl:]]+"; " ") | .[0:64]) + ] + | @tsv + ' 2>/dev/null) || { emit_error_once "invalid Better Stack API response"; return 0; } + + while IFS=$'\t' read -r id name started; do + [ -n "$id" ] || continue + claim_incident "$id" + claim_rc=$? + case "$claim_rc" in + 0) + new_count=$((new_count + 1)) + if [ "$new_count" -eq 1 ]; then + first_id=$id + first_name=$name + first_started=$started + new_ids=$id + else + new_ids="$new_ids,$id" + fi + ;; + 1) ;; + *) emit_error_once "cannot record incident dedupe state"; return 0 ;; + esac + done <<< "$rows" + + clear_error + case "$new_count" in + 0) ;; + 1) printf 'better-stack-incident opened id=%s name=%s started=%s\n' \ + "$first_id" "$first_name" "$first_started" ;; + *) printf 'better-stack-incidents opened ids=%s\n' "$new_ids" ;; + esac +} + +case "${1:-}" in + '') run_through_doppler ;; + --from-doppler) + [ "$#" -eq 1 ] || { emit_error_once "invalid poll invocation"; exit 0; } + poll_with_injected_token + ;; + *) emit_error_once "invalid poll invocation" ;; +esac diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index 16102adfa45..66e4406e4dc 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -17,7 +17,9 @@ # "NUDGE_SECONDMATES: secondmate <id>: send failed: <reason>", # "BOOTSTRAP_INFO: nudged fm-<id> with '<message>'", # "SECONDMATE_LIVENESS: secondmate <id>: skipped: <reason>|respawn failed after <cause>: <reason>", -# "FMX: X mode on ..." or "FMX: X mode off ...". +# "FMX: X mode on ..." or "FMX: X mode off ...", +# "BETTER_STACK: incident monitoring on ..." or +# "BETTER_STACK: incident monitoring off ...". # When a RUNNING secondmate worktree is fast-forwarded to firstmate's # own current default-branch commit (a purely LOCAL fast-forward, never # an origin fetch) AND its loaded instruction surface (AGENTS.md, bin/, @@ -73,15 +75,16 @@ # refresh relays any completed fm-fleet-sync.sh output before the # aggregate timeout skip line with timeout and elapsed seconds. # Set FM_FLEET_PRUNE=0 to skip branch pruning during that refresh. -# Set FM_BOOTSTRAP_DETECT_ONLY=1 to skip the five MUTATING sweeps +# Set FM_BOOTSTRAP_DETECT_ONLY=1 to skip the six MUTATING sweeps # (PR-check migration, secondmate_sync, secondmate_liveness_sweep, -# x_mode_setup, fleet_sync) while still printing every read-only detect line +# x_mode_setup, better_stack_incidents_setup, fleet_sync) while still +# printing every read-only detect line # above; the TANGLE line switches to advisory-only wording with no # checkout command. Used by # fm-session-start.sh's read-only path when another live session holds # the fleet lock, so a second concurrent session never race-mutates -# PR-check artifacts, secondmate homes, X-mode artifacts, project -# clones, or repair instructions. +# PR-check artifacts, secondmate homes, X-mode or Better Stack poll +# artifacts, project clones, or repair instructions. # Unset/0 (the default) runs every sweep exactly as before - this flag # is purely additive. # fm-bootstrap.sh install <tool>... @@ -107,6 +110,10 @@ DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" . "$SCRIPT_DIR/fm-startup-memory-budget-lib.sh" # shellcheck source=bin/fm-x-lib.sh disable=SC1091 . "$SCRIPT_DIR/fm-x-lib.sh" +# shellcheck source=bin/fm-pr-lib.sh disable=SC1091 +. "$SCRIPT_DIR/fm-pr-lib.sh" +# shellcheck source=bin/fm-check-lib.sh disable=SC1091 +. "$SCRIPT_DIR/fm-check-lib.sh" # shellcheck source=bin/fm-backend.sh disable=SC1091 . "$SCRIPT_DIR/fm-backend.sh" @@ -503,6 +510,7 @@ install_cmd() { manual_install_url() { case "$1" in + doppler) echo "https://docs.doppler.com/docs/install-cli" ;; herdr) echo "https://herdr.dev" ;; *) return 1 ;; esac @@ -557,7 +565,7 @@ no_mistakes_compatible() { [ "$patch" -ge "$NO_MISTAKES_MIN_PATCH" ] } -x_mode_write_if_changed() { +bootstrap_write_if_changed() { local dest=$1 content=$2 mode=$3 parent tmp parent_device current_mode parent=${dest%/*} [ "$parent" != "$dest" ] || return 1 @@ -578,7 +586,7 @@ x_mode_write_if_changed() { return 0 fi fi - tmp=$(umask 077; mktemp "$parent/.fm-x-mode.XXXXXX" 2>/dev/null) || return 1 + tmp=$(umask 077; mktemp "$parent/.fm-bootstrap-artifact.XXXXXX" 2>/dev/null) || return 1 if ! printf '%s\n' "$content" > "$tmp" \ || ! chmod "$mode" "$tmp" \ || ! fmx_single_link_file_mode_valid "$tmp" "$mode" "$parent_device"; then @@ -699,7 +707,7 @@ x_mode_setup() { ;; esac shim_body=$(fmx_poll_shim_content "$shim_home" "$FM_ROOT") - x_mode_write_if_changed "$shim" "$shim_body" 700 || { fmx_arm_failed; return 0; } + bootstrap_write_if_changed "$shim" "$shim_body" 700 || { fmx_arm_failed; return 0; } fmx_poll_shim_valid "$shim" "$shim_home" "$FM_ROOT" \ || { fmx_arm_failed; return 0; } @@ -711,11 +719,97 @@ x_mode_setup() { export FM_CHECK_INTERVAL=30 EOF ) - x_mode_write_if_changed "$cadence" "$cadence_body" 600 || { fmx_arm_failed; return 0; } + bootstrap_write_if_changed "$cadence" "$cadence_body" 600 || { fmx_arm_failed; return 0; } echo "FMX: X mode on - relay poll armed via state/x-watch.check.sh; 30s watcher cadence in config/x-mode.env" } +# Better Stack incident monitoring is an explicit home-local opt-in. +# A presence flag at config/better-stack-incidents materializes one ordinary +# registered custom check, so the existing hash-bound snapshot execution and +# FM_CHECK_TIMEOUT contract remain the only slow-check mechanism. +# The poll itself performs runtime-only Doppler injection and owns incident and +# diagnostic dedupe in this home's private state. +better_stack_incidents_setup() { + local flag check trust check_body tool missing check_home failed + flag="$CONFIG/better-stack-incidents" + check="$STATE/better-stack-incidents.check.sh" + trust="$STATE/better-stack-incidents.check-trust" + + better_stack_remove_artifacts() { + local remove_failed=0 + x_mode_remove_artifact "$check" || remove_failed=1 + x_mode_remove_artifact "$trust" || remove_failed=1 + [ "$remove_failed" -eq 0 ] + } + + if [ ! -e "$flag" ] && [ ! -L "$flag" ]; then + if x_mode_artifact_present "$check" || x_mode_artifact_present "$trust"; then + if better_stack_remove_artifacts; then + echo "BETTER_STACK: incident monitoring off - removed the home-scoped poll" + else + echo "BETTER_STACK: incident monitoring off - failed to remove the home-scoped poll" + fi + fi + return 0 + fi + if [ ! -f "$flag" ] || [ -L "$flag" ] || [ "$(fm_pr_file_link_count "$flag")" != 1 ]; then + better_stack_remove_artifacts || true + echo "BETTER_STACK: incident monitoring off - config/better-stack-incidents must be an ordinary file" + return 0 + fi + + missing=0 + for tool in doppler curl jq; do + if ! command -v "$tool" >/dev/null 2>&1; then + missing_tool_diagnostic "$tool" + missing=1 + fi + done + if [ "$missing" -ne 0 ]; then + better_stack_remove_artifacts || true + echo "BETTER_STACK: incident monitoring off - install the reported poll dependencies and rerun bootstrap" + return 0 + fi + + failed=0 + mkdir -p "$STATE" 2>/dev/null || failed=1 + if [ "$failed" -eq 0 ]; then + case "$FM_HOME" in + /*) check_home=$FM_HOME ;; + *) + check_home=$(CDPATH='' cd -- "$FM_HOME" 2>/dev/null && pwd -P) || failed=1 + ;; + esac + fi + if [ "$failed" -eq 0 ]; then + check_body=$(printf '%s\n' \ + '#!/usr/bin/env bash' \ + '# Auto-generated by fm-bootstrap.sh - Better Stack incident custom check.' \ + '# Registered bytes call the tracked poll; output becomes a check: wake.' \ + "export FM_HOME=$(printf '%q' "$check_home")" \ + "exec $(printf '%q' "$FM_ROOT/bin/fm-better-stack-incidents-poll.sh")") + bootstrap_write_if_changed "$check" "$check_body" 700 || failed=1 + fi + if [ "$failed" -eq 0 ] && ! fm_custom_check_registered "$STATE" better-stack-incidents; then + FM_HOME="$FM_HOME" FM_STATE_OVERRIDE="$STATE" FM_ROOT_OVERRIDE="$FM_ROOT" \ + "$SCRIPT_DIR/fm-check-register.sh" better-stack-incidents >/dev/null 2>&1 || failed=1 + fi + if [ "$failed" -eq 0 ]; then + fm_custom_check_registered "$STATE" better-stack-incidents || failed=1 + fi + if [ "$failed" -ne 0 ]; then + if better_stack_remove_artifacts; then + echo "BETTER_STACK: incident monitoring off - failed to arm the home-scoped poll" + else + echo "BETTER_STACK: incident monitoring off - failed to arm the home-scoped poll; stale artifacts remain" + fi + return 0 + fi + + echo "BETTER_STACK: incident monitoring on - registered state/better-stack-incidents.check.sh at the default 300s check cadence" +} + crew_dispatch_validate() { local file err file="$CONFIG/crew-dispatch.json" @@ -900,6 +994,7 @@ if [ "${FM_BOOTSTRAP_DETECT_ONLY:-0}" != 1 ]; then secondmate_liveness_sweep secondmate_sync x_mode_setup + better_stack_incidents_setup fleet_sync fi exit 0 diff --git a/bin/fm-claude-stop-autoarm.sh b/bin/fm-claude-stop-autoarm.sh index df9ee1128fc..0c2c4fbf528 100755 --- a/bin/fm-claude-stop-autoarm.sh +++ b/bin/fm-claude-stop-autoarm.sh @@ -18,8 +18,8 @@ # - AFK: while state/.afk exists the away daemon owns the watcher and triage; # this hook exits 0 and NEVER rewakes the primary (checked again at # translation time so a mid-cycle AFK transition is honored). -# - Need: arms only while work is in flight (state/*.meta) or X mode has a -# relay poll to run (state/x-watch.check.sh); an idle home exits 0. +# - Need: arms only while work is in flight (state/*.meta) or a home-level +# X-mode or Better Stack poll is active; an idle home exits 0. # - Single-flight: Claude does not dedupe async hooks, so a home-scoped owner # lock (state/.claude-autoarm.lock) admits exactly one owner; every other # concurrent firing exits 0 without translating, which keeps one event @@ -89,7 +89,7 @@ fi # --- AFK: the away daemon owns the watcher and triage; never rewake ---------- [ -e "$STATE/.afk" ] && exit 0 -# --- need: in-flight work or an X-mode relay poll ---------------------------- +# --- need: in-flight work or a home-level poll ------------------------------- need_supervision() { fm_supervision_needed "$STATE" "$GRACE" } diff --git a/bin/fm-session-start.sh b/bin/fm-session-start.sh index 1abbace4bf1..85ace7fa29f 100755 --- a/bin/fm-session-start.sh +++ b/bin/fm-session-start.sh @@ -18,7 +18,7 @@ # standalone with unchanged default behavior - other flows (fm-bootstrap.sh # install <tools> after consent, /updatefirstmate, the afk daemon, existing # tests) still call them directly. The one seam this script needed - -# bootstrap running its detect-only diagnostics without its five mutating +# bootstrap running its detect-only diagnostics without its six mutating # sweeps - is an opt-in FM_BOOTSTRAP_DETECT_ONLY=1 flag on fm-bootstrap.sh # itself (default unset/0 = unchanged behavior), not a fork. # @@ -29,9 +29,10 @@ # mutating step runs. # 2. bootstrap - home-local stale Herdr projection cleanup runs only # when this session actually holds the lock. Detect-only -# diagnostics always run. Bootstrap's five MUTATING sweeps +# diagnostics always run. Bootstrap's six MUTATING sweeps # (legacy PR-check migration, secondmate fast-forward, -# secondmate liveness, X-mode artifact writes, fleet sync) +# secondmate liveness, X-mode artifact writes, Better Stack +# incident-poll registration, fleet sync) # also run only when locked. # 3. wake-drain - mutates the durable wake queue, so it also only runs # when locked. @@ -52,7 +53,8 @@ # # Why lock first: the old documented order (bootstrap, THEN lock) let a # SECOND concurrent session run bootstrap's mutating sweeps - fast-forwarding -# secondmate homes, writing X-mode artifacts, fetching/fast-forwarding every +# secondmate homes, writing X-mode and Better Stack poll artifacts, +# fetching/fast-forwarding every # project clone - before ever discovering another session already holds the # lock. Two sessions racing those sweeps is exactly the hazard the lock # exists to prevent, so locking first closes the hole outright: only the @@ -65,7 +67,7 @@ # tasks-axi and quota-axi tool checks, and tasks-axi availability - none of # which mutate shared state and all of which are safe to compute without # verified lock ownership. -# Only projection cleanup, the five bootstrap mutating sweeps, and the +# Only projection cleanup, the six bootstrap mutating sweeps, and the # wake-queue drain are skipped. # The context and fleet-state digests # below are always read-only, so they run unconditionally in both modes. @@ -257,7 +259,7 @@ if [ "$LOCK_RC" -ne 0 ]; then printf '● READ-ONLY SESSION - FLEET LOCK OWNERSHIP WAS NOT VERIFIED\n' printf '● %s\n' "$LOCK_OUT" printf '● Skipping every mutating step: PR-check migration, stale Herdr child cleanup,\n' - printf '● secondmate sync, X-mode artifacts, fleet sync, and wake-queue drain. Detect-only bootstrap\n' + printf '● secondmate sync, X-mode and Better Stack poll artifacts, fleet sync, and wake-queue drain. Detect-only bootstrap\n' printf '● diagnostics and the rest of this read-only-safe digest still ran below.\n' printf '● Operate read-only until this resolves - do not spawn, steer, merge, or\n' printf '● otherwise mutate fleet state from this session.\n' diff --git a/bin/fm-subagent-pretool-check.sh b/bin/fm-subagent-pretool-check.sh index 8edb507218b..95a0c6be434 100755 --- a/bin/fm-subagent-pretool-check.sh +++ b/bin/fm-subagent-pretool-check.sh @@ -6,7 +6,7 @@ # no `data/<id>/brief.md`. Only `bin/fm-spawn.sh` writes that metadata, and # untracked project work contributes nothing to the in-flight branch of # bin/fm-supervision-lib.sh or bin/fm-turnend-guard.sh. So such work is not -# merely unsupervised: absent an independent X-mode need, it makes the whole +# merely unsupervised: absent an independent home-monitoring need, it makes the whole # guard stack structurally inert, and it dies with the primary session instead # of living in its own backend session. # diff --git a/bin/fm-supervision-lib.sh b/bin/fm-supervision-lib.sh index 1930700d2af..b7ff73d0158 100644 --- a/bin/fm-supervision-lib.sh +++ b/bin/fm-supervision-lib.sh @@ -3,9 +3,9 @@ # Usage: . bin/fm-supervision-lib.sh # # Reports whether a firstmate home needs supervision because it has in-flight -# work (a state/<id>.meta exists) or an X-mode relay poll -# (state/x-watch.check.sh), and whether its watcher has a fresh liveness beacon -# (state/.last-watcher-beat, touched every poll cycle, within the grace window). +# work (a state/<id>.meta exists) or a home-level poll (X mode or Better Stack), +# and whether its watcher has a fresh liveness beacon (state/.last-watcher-beat, +# touched every poll cycle, within the grace window). # bin/fm-guard.sh keeps its task-specific grace-based warning predicate; # bin/fm-turnend-guard.sh uses the status fields here for its banner but performs # its end-of-turn block decision with the live watcher lock check in @@ -23,7 +23,7 @@ fm_sup_stat_mtime() { # fm_supervision_status <state-dir> [grace-seconds] # Populates, for the state dir at $1: # FM_SUP_IN_FLIGHT count of state/*.meta (in-flight tasks) -# FM_SUP_NEEDED true/false - in-flight work or an X-mode relay poll +# FM_SUP_NEEDED true/false - in-flight work or a home-level poll # FM_SUP_WATCHER_FRESH true/false - a watcher beacon within the grace window # FM_SUP_BEACON_DESC human-readable beacon age, for banners ("never" if absent) # FM_SUP_QUEUE_PENDING true/false - state/.wake-queue has unread records @@ -41,7 +41,9 @@ fm_supervision_status() { [ -e "$meta" ] || continue FM_SUP_IN_FLIGHT=$((FM_SUP_IN_FLIGHT + 1)) done - if [ "$FM_SUP_IN_FLIGHT" -gt 0 ] || [ -f "$state/x-watch.check.sh" ]; then + if [ "$FM_SUP_IN_FLIGHT" -gt 0 ] \ + || [ -f "$state/x-watch.check.sh" ] \ + || [ -f "$state/better-stack-incidents.check.sh" ]; then FM_SUP_NEEDED=true fi @@ -64,8 +66,8 @@ fm_supervision_status() { } # fm_supervision_needed <state-dir> [grace-seconds] -# Exit 0 (true) exactly when in-flight work or an X-mode relay poll needs a -# watcher. Exit 1 (false) for an idle home. +# Exit 0 (true) exactly when in-flight work or a home-level poll needs a watcher. +# Exit 1 (false) for an idle home. fm_supervision_needed() { fm_supervision_status "$@" [ "$FM_SUP_NEEDED" = true ] diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 01df68c08a5..192281ecf56 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -131,7 +131,7 @@ family_for_basename() { fm-test-run.test.sh|fm-test-isolation-proof.test.sh) printf '%s\n' pure-contract-unit ;; - fm-daemon.test.sh|fm-guard-stale-banner.test.sh|fm-pi-watch-extension.test.sh|\ + fm-better-stack-incidents.test.sh|fm-daemon.test.sh|fm-guard-stale-banner.test.sh|fm-pi-watch-extension.test.sh|\ fm-supervision-events.test.sh|fm-turnend-guard.test.sh|fm-wake-daemon-lifecycle-e2e.test.sh|\ fm-wake-queue.test.sh|fm-watch-checkpoint.test.sh|fm-watch-triage.test.sh|\ fm-watcher-lock.test.sh) diff --git a/bin/fm-turnend-guard.sh b/bin/fm-turnend-guard.sh index 2e96fb33e48..d9bdc72fff6 100755 --- a/bin/fm-turnend-guard.sh +++ b/bin/fm-turnend-guard.sh @@ -164,6 +164,10 @@ block_stop() { printf '● TURN WOULD END BLIND - SUPERVISION IS OFF\n' if [ "$FM_SUP_IN_FLIGHT" -gt 0 ]; then printf '● %s task(s) in flight, but no live watcher holds this home lock (last beat: %s).\n' "$FM_SUP_IN_FLIGHT" "$FM_SUP_BEACON_DESC" + elif [ -f "$STATE/better-stack-incidents.check.sh" ] && [ -f "$STATE/x-watch.check.sh" ]; then + printf '● X-mode relay and Better Stack incident monitoring need supervision, but no live watcher holds this home lock (last beat: %s).\n' "$FM_SUP_BEACON_DESC" + elif [ -f "$STATE/better-stack-incidents.check.sh" ]; then + printf '● Better Stack incident monitoring needs supervision, but no live watcher holds this home lock (last beat: %s).\n' "$FM_SUP_BEACON_DESC" else printf '● X-mode relay polling needs supervision, but no live watcher holds this home lock (last beat: %s).\n' "$FM_SUP_BEACON_DESC" fi @@ -229,6 +233,10 @@ if [ "$COUNT" -gt "$BLOCK_BUDGET" ]; then budget_reset if [ "$FM_SUP_IN_FLIGHT" -gt 0 ]; then NEED_DESC="$FM_SUP_IN_FLIGHT task(s) in flight" + elif [ -f "$STATE/better-stack-incidents.check.sh" ] && [ -f "$STATE/x-watch.check.sh" ]; then + NEED_DESC="X-mode relay and Better Stack incident monitoring active" + elif [ -f "$STATE/better-stack-incidents.check.sh" ]; then + NEED_DESC="Better Stack incident monitoring active" else NEED_DESC="X-mode relay polling active" fi diff --git a/docs/architecture.md b/docs/architecture.md index bf8b5cb3ec1..4e7e82e661c 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -10,6 +10,9 @@ firstmate's always-loaded operating contract and routing index for conditional p A zero-token bash watcher (`bin/fm-watch.sh`) sleeps on the fleet, classifies detected wakes in bash, and wakes the first mate only when something is actionable. Actionable wakes include captain-relevant status signals, no-verb signals whose crew is not provably working, authenticated check output such as PR merge polling or an X-mode mention, stale panes whose crew is not provably working whether their status log looks terminal or non-terminal, provably-working stale panes that persist past `FM_STALE_ESCALATE_SECS`, declared external waits that remain paused past `FM_PAUSE_RESURFACE_SECS`, and heartbeat backstop hits. +Better Stack incident monitoring uses the registered custom-check extension rather than a new watcher scheduler because that existing path already supplies home scoping, hash-bound private snapshot execution, the slow-check timeout, and durable `check:` delivery. +The locked bootstrap materializes the check only for a home carrying `config/better-stack-incidents`, and [`configuration.md`](configuration.md#better-stack-incident-monitoring-configbetter-stack-incidents) owns runtime Doppler injection, incident-ID dedupe, diagnostic suppression, cadence, and opt-out mechanics. +The registered check remains a supervision need when the project fleet is idle, and it replaces the retired interim practice of querying Better Stack during heartbeat reviews. Repeated provably-working stale escalations on the same unchanged pane add an escalation count to the wake reason and, at `FM_WEDGE_DEMAND_INSPECT_COUNT`, a `demand-deep-inspection` marker. A busy pane is otherwise exempt from staleness, but only until its latest `state/<id>.turn-ended` marker reaches `FM_BUSY_TURN_MAX_SECS`, or its `state/<id>.meta` spawn record reaches that age before any turn completes; past that bound it is routed through the same wedge escalation, with the identical reason, escalation count, and `demand-deep-inspection` marker, for inspection only - never an automatic interrupt, signal, or restart. Those actionable wakes are written to a durable local queue (`state/.wake-queue`) before detector state advances, so a missed process exit can be recovered by draining the queue. diff --git a/docs/configuration.md b/docs/configuration.md index b226ec6888c..09a123ba136 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -11,7 +11,7 @@ The shared orchestrator behavior lives in [`AGENTS.md`](../AGENTS.md) - edit it This section is the single owner of the top-level operational-home layout; producer script headers and their help own exact child-file fields and mutation contracts. The tracked code root contains the shared instruction, skill, documentation, workflow, and `bin/` surfaces, while each effective `FM_HOME` contains private operational directories. `data/` holds durable private fleet records such as the project and secondmate registries, captain preferences, optional shared captain preferences, learnings, backlog, briefs, and scout reports. -`state/` holds volatile runtime records such as task metadata, append-only status events, endpoint signals, watcher and wake-queue coordination, away-mode state, generated X-mode artifacts, private secondmate config-reread generations with their retry and quarantine state, and parent-owned secondmate pending-reply records under `state/pending-replies/` (`bin/fm-pending-reply-lib.sh`). +`state/` holds volatile runtime records such as task metadata, append-only status events, endpoint signals, watcher and wake-queue coordination, away-mode state, generated X-mode and Better Stack poll artifacts, private secondmate config-reread generations with their retry and quarantine state, and parent-owned secondmate pending-reply records under `state/pending-replies/` (`bin/fm-pending-reply-lib.sh`). `config/` holds local gitignored operating choices, and `projects/` holds the local project clones that Firstmate reads but changes only through the narrow guarded and concrete captain-approved exceptions in `AGENTS.md`. `bin/fm-spawn.sh` owns the base task-metadata fields it emits, while the runtime-backend section below owns backend-specific fields and selector interpretation. @@ -305,6 +305,28 @@ The locked bootstrap inheritance pass uses the same per-home changed-set and rer That live discovery starts from `state/*.meta` records with `kind=secondmate`; `data/secondmates.md` only backfills `home=` for older or incomplete meta records. Skipped items, such as a destination checkout that does not yet gitignore the item, are visible warnings but not hard failures. +## Better Stack incident monitoring (config/better-stack-incidents) + +Create an ordinary local file at `config/better-stack-incidents` in the one Firstmate home that should receive fleet incident notifications. +The file is a presence flag with no secret content, is gitignored, and is deliberately not inherited into secondmate homes so one incident does not alert multiple supervisors. +The next locked session-start bootstrap requires `doppler`, `curl`, and `jq`, writes `state/better-stack-incidents.check.sh`, and binds those exact shim bytes in `state/better-stack-incidents.check-trust` through `bin/fm-check-register.sh`. +This uses the existing registered custom-check extension point: the watcher executes only a hash-validated private snapshot, applies `FM_CHECK_TIMEOUT`, and converts the poll's one-line output into a durable `check:` notification. +The registered poll is a home-level supervision need even with no project work in flight and runs on the default `FM_CHECK_INTERVAL=300` slow-check cadence, which keeps normal detection within minutes without adding a second scheduler. + +`bin/fm-better-stack-incidents-poll.sh` invokes its poll child with `doppler run --silent --no-check-version --no-fallback --project fleet-observability --config prd --only-secrets BETTER_STACK_API_TOKEN`. +`--no-fallback` prevents Doppler from reading or writing a fallback secret file, and the child passes the Better Stack bearer header to `curl` through standard input rather than a command argument or temporary header file. +The token therefore remains runtime-only and must never be added to the flag, repository, logs, task instructions, or another local file. +The poll requests unresolved incidents from Better Stack's documented [`GET /api/v3/incidents`](https://betterstack.com/docs/uptime/api/list-all-incidents/) endpoint with a five-second HTTP bound. + +A previously unseen incident ID creates `state/better-stack-incidents.seen/<id>` as a private single-link marker and prints one compact identity line. +One response containing several unseen incidents prints one line containing all new IDs, so a check cycle still produces exactly one durable notification. +Already-seen incidents and quiet API responses print nothing. +Missing credentials, Doppler access failure, network failure, non-success HTTP status, and malformed API data print one `better-stack-error ...` diagnostic and record it in `state/better-stack-incidents.diagnostics/error`; the same diagnostic then remains silent until a successful poll clears the marker or a different failure occurs. + +Remove `config/better-stack-incidents` and rerun locked session start to retire the runnable check and its trust binding. +Bootstrap retains the private seen-ID and diagnostic markers so disabling and later re-enabling the poll cannot re-notify every still-open incident. +Better Stack polling from heartbeat handling was an interim practice and is retired; the registered check is the only poll owner, while [`AGENTS.md` section 8](../AGENTS.md#8-supervision-protocol) owns incident triage after a notification arrives. + ## X mode (.env) X mode lets a firstmate instance answer public `@myfirstmate` mentions and act on normal reversible mention requests through firstmate's normal lifecycle. diff --git a/docs/subagent-guard.md b/docs/subagent-guard.md index 47aaf10e0f3..0e48f2f4715 100644 --- a/docs/subagent-guard.md +++ b/docs/subagent-guard.md @@ -367,8 +367,8 @@ tests/fm-subagent-pretool-check.test.sh This change does not close the deeper harness-agnostic defect. Every firstmate guard's in-flight-work branch keys off `state/<id>.meta`, and only `bin/fm-spawn.sh` writes that record. -`bin/fm-supervision-lib.sh` also recognizes an X-mode relay poll as supervision need, but unaccounted primary work still contributes nothing to that predicate. -Without an independent X-mode need, unaccounted primary work therefore reads as idle rather than suspicious. +`bin/fm-supervision-lib.sh` also recognizes X-mode and Better Stack home-level polls as supervision needs, but unaccounted primary work still contributes nothing to that predicate. +Without an independent home-monitoring need, unaccounted primary work therefore reads as idle rather than suspicious. The durable fix for that class is to make the guards treat "the primary is doing project-shaped work with zero `state/*.meta` files" as a suspicious state rather than an idle one. That would catch this class on any harness, including work created through `Bash`. diff --git a/docs/supervision-protocols/grok.md b/docs/supervision-protocols/grok.md index 22444b2bd7f..b4255df5971 100644 --- a/docs/supervision-protocols/grok.md +++ b/docs/supervision-protocols/grok.md @@ -24,7 +24,7 @@ When you see a background-task-completed system reminder for the arm: 1. Run `bin/fm-wake-drain.sh` first. 2. Optionally fetch arm output with `get_command_or_subagent_output(<task_id>)` for the reason line. 3. Handle `signal`, `stale`, `check`, or `heartbeat` using the harness-neutral contract in `AGENTS.md`. -4. Ordinary wake: re-arm the next cycle with the same background `bin/fm-watch-arm.sh` call if work remains in flight or X mode still needs polling. +4. Ordinary wake: re-arm the next cycle with the same background `bin/fm-watch-arm.sh` call if work remains in flight or a home-level X-mode or Better Stack poll remains active. 5. Do not invent a wake from an attach-status line alone. Drain the queue and act only on real wake records or a real watcher reason line. Re-arm attaches to an existing healthy cycle when one is already present and follows its verified successor chain. diff --git a/docs/turnend-guard.md b/docs/turnend-guard.md index 8ee750de397..43ac978675c 100644 --- a/docs/turnend-guard.md +++ b/docs/turnend-guard.md @@ -13,7 +13,7 @@ Do not infer this guard's scope, loop safety, or compatibility tradeoffs for tho `bin/fm-guard.sh` is a pull-based warning that runs only when another supervision command invokes it. The turn-end guard closes the remaining gap at the primary's own turn boundary. -When work is in flight and no identity-matched watcher has a fresh beacon, the harness integration must either block the turn end or force one bounded follow-up that uses the recovery instruction from the emitted session-start protocol. +When work is in flight or a home-level poll is active and no identity-matched watcher has a fresh beacon, the harness integration must either block the turn end or force one bounded follow-up that uses the recovery instruction from the emitted session-start protocol. The guard remains a backstop; [`watcher-continuity.md`](watcher-continuity.md) owns normal continuity. ## Shared predicate @@ -25,9 +25,9 @@ An unmarked checkout or invalid marker falls through to the git-dir check. That check keeps crewmate and scout linked worktrees inert because their git dir differs from their git common dir. It also requires `AGENTS.md`, `bin/`, and the effective state directory. -For an in-scope primary, the guard counts in-flight work from `state/*.meta`. +For an in-scope primary, the guard counts in-flight work from `state/*.meta` and recognizes the generated X-mode and Better Stack checks as independent home-level supervision needs. The default cross-harness mode exits silently with no work in flight. -Claude's `--claude` mode also treats `state/x-watch.check.sh` as supervision need, so X-mode relay polling remains guarded without an in-flight task. +Claude's `--claude` mode uses that full supervision predicate, so either home-level poll remains guarded without an in-flight task. Otherwise it calls `fm_watcher_healthy <state-dir> <watch-path> [grace-seconds] [home]` from `bin/fm-wake-lib.sh`, the same identity-matched lock and fresh-beacon check used by `bin/fm-watch-arm.sh`. A stale beacon blocks even when a watcher pid is live. A fresh leftover beacon blocks when the lock is missing, dead, or identity-mismatched. @@ -76,7 +76,7 @@ That warning uses `bin/fm-supervision-instructions.sh --repair-line`, so it alwa ## Compatibility limits - Child crewmate and scout worktrees are outside scope. -- A valid secondmate home is in scope; an idle secondmate endpoint with no X-mode relay poll remains healthy because it has no supervision need. +- A valid secondmate home is in scope; an idle secondmate endpoint with no home-level poll remains healthy because it has no supervision need. - The direct-blocking and bounded passive-follow-up split is limited to the primary integrations listed above. - OpenCode headless mode and untrusted Grok project hooks remain fail-open at the host boundary. - Kimi Code CLI 0.29.1 exposes only global `[[hooks]]` configuration in `~/.kimi-code/config.toml`, including a `Stop` event with snake_case payload fields `hook_event_name`, `session_id`, `cwd`, and `stop_hook_active`. diff --git a/tests/fm-better-stack-incidents.test.sh b/tests/fm-better-stack-incidents.test.sh new file mode 100755 index 00000000000..86320bea0f4 --- /dev/null +++ b/tests/fm-better-stack-incidents.test.sh @@ -0,0 +1,246 @@ +#!/usr/bin/env bash +# Behavior tests for the home-scoped Better Stack incident poll. +# +# The API and Doppler boundary are both mocked. +# These tests make no live network or secret-store calls. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" +# shellcheck source=bin/fm-supervision-lib.sh disable=SC1091 +. "$ROOT/bin/fm-supervision-lib.sh" + +POLL="$ROOT/bin/fm-better-stack-incidents-poll.sh" +REGISTER="$ROOT/bin/fm-check-register.sh" +WATCH="$ROOT/bin/fm-watch.sh" +BASE_PATH=${FM_TEST_BASE_PATH:-/usr/bin:/bin:/usr/sbin:/sbin} +JQ_DIR=$(command -v jq 2>/dev/null) && JQ_DIR=$(dirname "$JQ_DIR") || JQ_DIR= +[ -n "$JQ_DIR" ] && BASE_PATH="$JQ_DIR:$BASE_PATH" +TMP_ROOT=$(fm_test_tmproot fm-better-stack-incidents) + +make_case() { + local name=$1 dir fakebin + dir="$TMP_ROOT/$name" + fakebin=$(fm_fakebin "$dir") + mkdir -p "$dir/home/state" "$dir/home/config" + chmod 0700 "$dir/home/state" + + cat > "$fakebin/doppler" <<'SH' +#!/usr/bin/env bash +printf '%s\n' "$*" >> "$FM_TEST_DOPPLER_LOG" +while [ "$#" -gt 0 ] && [ "$1" != -- ]; do + shift +done +[ "${1:-}" = -- ] || exit 2 +shift +exec env BETTER_STACK_API_TOKEN="${FM_TEST_BETTER_STACK_TOKEN:-}" "$@" +SH + + cat > "$fakebin/curl" <<'SH' +#!/usr/bin/env bash +cat >/dev/null +printf '%s\n' "$*" >> "$FM_TEST_CURL_LOG" +[ "${FM_TEST_CURL_FAIL:-0}" = 0 ] || exit 7 +printf '%s\n%s' "${FM_TEST_API_BODY:-}" "${FM_TEST_API_CODE:-200}" +SH + chmod +x "$fakebin/doppler" "$fakebin/curl" + : > "$dir/doppler.log" + : > "$dir/curl.log" + printf '%s\n' "$dir" +} + +run_poll() { + local dir=$1 + shift + PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$dir/home" \ + FM_TEST_DOPPLER_LOG="$dir/doppler.log" \ + FM_TEST_CURL_LOG="$dir/curl.log" \ + "$@" "$POLL" +} + +incident_body() { + local id=$1 name=${2:-api-production} started=${3:-2026-08-01T12:00:00.000Z} + jq -cn --arg id "$id" --arg name "$name" --arg started "$started" \ + '{data: [{id: $id, type: "incident", attributes: {name: $name, started_at: $started, resolved_at: null, status: "Started"}}]}' +} + +test_new_incident_and_duplicate_suppression() { + local dir body out rc + dir=$(make_case new-and-duplicate) + body=$(incident_body 25 api-production) + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200); rc=$? + expect_code 0 "$rc" "new incident poll exit" + [ "$out" = 'better-stack-incident opened id=25 name=api-production started=2026-08-01T12:00:00.000Z' ] \ + || fail "new incident must print one compact identity line (got: $out)" + assert_present "$dir/home/state/better-stack-incidents.seen/25" \ + "new incident must claim a private seen marker" + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200); rc=$? + expect_code 0 "$rc" "duplicate incident poll exit" + [ -z "$out" ] || fail "an already-seen incident must stay silent (got: $out)" + + assert_grep '--project fleet-observability --config prd' "$dir/doppler.log" \ + "poll must select the fleet-observability/prd Doppler scope" + assert_grep '--no-fallback' "$dir/doppler.log" \ + "poll must prohibit Doppler fallback files" + assert_grep '--only-secrets BETTER_STACK_API_TOKEN' "$dir/doppler.log" \ + "poll must inject only the Better Stack token" + assert_grep 'https://uptime.betterstack.com/api/v3/incidents?resolved=false&per_page=50' \ + "$dir/curl.log" "poll must query unresolved Better Stack incidents" + if grep -R -F 'synthetic-test-token' "$dir/home/state" "$dir/doppler.log" "$dir/curl.log" >/dev/null 2>&1; then + fail "the Better Stack token reached private state or command logs" + fi + pass "new Better Stack incident wakes once and duplicate observations stay silent" +} + +test_api_error_reports_once_and_recovers() { + local dir out rc + dir=$(make_case api-error) + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY='{"error":"unavailable"}' FM_TEST_API_CODE=503); rc=$? + expect_code 0 "$rc" "API error poll exit" + [ "$out" = 'better-stack-error API returned HTTP 503' ] \ + || fail "API error must produce one visible diagnostic (got: $out)" + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY='{"error":"unavailable"}' FM_TEST_API_CODE=503); rc=$? + expect_code 0 "$rc" "repeated API error poll exit" + [ -z "$out" ] || fail "repeated API failure must not produce a wake storm (got: $out)" + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY='{"data":[]}' FM_TEST_API_CODE=200); rc=$? + expect_code 0 "$rc" "recovered API poll exit" + [ -z "$out" ] || fail "successful recovery must stay silent (got: $out)" + assert_absent "$dir/home/state/better-stack-incidents.diagnostics/error" \ + "successful API access must clear the diagnostic marker" + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_CURL_FAIL=1); rc=$? + expect_code 0 "$rc" "unreachable API poll exit" + [ "$out" = 'better-stack-error Better Stack API unreachable' ] \ + || fail "unreachable API must produce one visible diagnostic (got: $out)" + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_CURL_FAIL=1); rc=$? + expect_code 0 "$rc" "repeated unreachable API poll exit" + [ -z "$out" ] || fail "repeated unreachable API failure must stay quiet (got: $out)" + pass "Better Stack API failures surface once and recovery clears the diagnostic" +} + +test_missing_token_reports_once() { + local dir out rc + dir=$(make_case missing-token) + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN='' \ + FM_TEST_API_BODY='{"data":[]}' FM_TEST_API_CODE=200); rc=$? + expect_code 0 "$rc" "missing-token poll exit" + [ "$out" = 'better-stack-error missing BETTER_STACK_API_TOKEN' ] \ + || fail "missing token must produce one visible diagnostic (got: $out)" + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN='' \ + FM_TEST_API_BODY='{"data":[]}' FM_TEST_API_CODE=200); rc=$? + expect_code 0 "$rc" "repeated missing-token poll exit" + [ -z "$out" ] || fail "repeated missing-token failure must stay quiet (got: $out)" + pass "missing Better Stack token produces one diagnostic without a wake storm" +} + +test_registered_check_delivers_check_wake() { + local dir state shim body out rc + dir=$(make_case watcher-delivery) + state="$dir/home/state" + shim="$state/better-stack-incidents.check.sh" + body=$(incident_body 91 web-production) + cat > "$shim" <<SH +#!/usr/bin/env bash +export FM_HOME=$(printf '%q' "$dir/home") +exec $(printf '%q' "$POLL") +SH + chmod 0700 "$shim" + FM_HOME="$dir/home" "$REGISTER" better-stack-incidents >/dev/null \ + || fail "could not register Better Stack custom check" + + out=$(PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$dir/home" FM_STATE_OVERRIDE="$state" \ + FM_TEST_DOPPLER_LOG="$dir/doppler.log" FM_TEST_CURL_LOG="$dir/curl.log" \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200 \ + FM_CHECK_INTERVAL=0 FM_HEARTBEAT=999999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + "$WATCH"); rc=$? + expect_code 0 "$rc" "watcher incident delivery exit" + assert_contains "$out" "check: $shim: better-stack-incident opened id=91" \ + "registered poll output must become a check wake with the incident identity" + [ "$(grep -c 'better-stack-incident opened id=91' "$state/.wake-queue")" -eq 1 ] \ + || fail "new incident must create exactly one durable wake record" + pass "registered Better Stack poll delivers exactly one authenticated check wake" +} + +test_bootstrap_arms_and_retires_home_check() { + local dir home out sum1 sum2 + dir=$(make_case bootstrap) + home="$dir/home" + : > "$home/config/better-stack-incidents" + + out=$(FM_HOME="$home" "$ROOT/bin/fm-bootstrap.sh" 2>/dev/null) + assert_contains "$out" 'BETTER_STACK: incident monitoring on' \ + "bootstrap must announce Better Stack incident monitoring" + assert_present "$home/state/better-stack-incidents.check.sh" \ + "bootstrap must materialize the home-scoped custom check" + assert_present "$home/state/better-stack-incidents.check-trust" \ + "bootstrap must register the custom check bytes" + [ -x "$home/state/better-stack-incidents.check.sh" ] \ + || fail "Better Stack custom check must be executable" + assert_grep 'fm-better-stack-incidents-poll.sh' "$home/state/better-stack-incidents.check.sh" \ + "custom check shim must invoke the tracked poll" + + sum1=$(cat "$home/state/better-stack-incidents.check.sh" \ + "$home/state/better-stack-incidents.check-trust" | shasum) + FM_HOME="$home" "$ROOT/bin/fm-bootstrap.sh" >/dev/null 2>&1 + sum2=$(cat "$home/state/better-stack-incidents.check.sh" \ + "$home/state/better-stack-incidents.check-trust" | shasum) + [ "$sum1" = "$sum2" ] || fail "bootstrap incident activation must be idempotent" + + rm "$home/config/better-stack-incidents" + out=$(FM_HOME="$home" "$ROOT/bin/fm-bootstrap.sh" 2>/dev/null) + assert_contains "$out" 'BETTER_STACK: incident monitoring off' \ + "bootstrap must announce removal of an armed incident poll" + assert_absent "$home/state/better-stack-incidents.check.sh" \ + "opt-out must remove the Better Stack custom check" + assert_absent "$home/state/better-stack-incidents.check-trust" \ + "opt-out must remove the custom-check trust binding" + pass "bootstrap idempotently arms and retires the home-scoped incident check" +} + +test_incident_check_keeps_home_supervised() { + local dir state + dir=$(make_case supervision-need) + state="$dir/home/state" + : > "$state/better-stack-incidents.check.sh" + + fm_supervision_needed "$state" 300 \ + || fail "Better Stack incident monitoring must keep an otherwise-idle home supervised" + [ "$FM_SUP_IN_FLIGHT" -eq 0 ] \ + || fail "incident monitoring must not count as a project task" + [ "$FM_SUP_NEEDED" = true ] \ + || fail "incident monitoring must set the home supervision need" + pass "Better Stack incident polling remains supervised with no project work in flight" +} + +test_new_incident_and_duplicate_suppression +test_api_error_reports_once_and_recovers +test_missing_token_reports_once +test_registered_check_delivers_check_wake +test_bootstrap_arms_and_retires_home_check +test_incident_check_keeps_home_supervised From 6aa3f02fe6c2e477413d39841da62e7d9e066843 Mon Sep 17 00:00:00 2001 From: juniorlovestmh <272474227+juniorlovestmh@users.noreply.github.com> Date: Sat, 1 Aug 2026 09:11:41 -0300 Subject: [PATCH 46/52] no-mistakes(review): Captain: fixed pagination, dedupe recovery, and brief validation --- bin/fm-better-stack-incidents-poll.sh | 127 +++++++++++------------- bin/fm-brief.sh | 3 + bin/fm-wake-lib.sh | 28 ++++++ bin/fm-watch.sh | 19 +++- tests/fm-better-stack-incidents.test.sh | 52 ++++++++-- 5 files changed, 148 insertions(+), 81 deletions(-) diff --git a/bin/fm-better-stack-incidents-poll.sh b/bin/fm-better-stack-incidents-poll.sh index e7085bbde4e..c80c4af16f8 100755 --- a/bin/fm-better-stack-incidents-poll.sh +++ b/bin/fm-better-stack-incidents-poll.sh @@ -65,25 +65,21 @@ run_through_doppler() { fi case "$out" in '') return 0 ;; - *$'\n'*) emit_error_once "poll returned invalid multiline output" ;; - better-stack-incident\ opened\ *|better-stack-incidents\ opened\ *|better-stack-error\ *) - printf '%s\n' "$out" + *) + while IFS= read -r line; do + case "$line" in + better-stack-incident\ opened\ *|better-stack-error\ *) printf '%s\n' "$line" ;; + *) emit_error_once "poll returned invalid output"; return 0 ;; + esac + done <<< "$out" ;; - *) emit_error_once "poll returned invalid output" ;; esac } -claim_incident() { - local id=$1 rc - printf 'seen\n' \ - | fmx_private_artifact_publish_stdin_once "$SEEN_DIR" "$id" 600 2>/dev/null - rc=$? - return "$rc" -} - poll_with_injected_token() { - local token=${BETTER_STACK_API_TOKEN:-} raw code body rows id name started - local claim_rc new_count=0 new_ids= first_id= first_name= first_started= + local token=${BETTER_STACK_API_TOKEN:-} raw code body page_rows page_next + local id name started next_url='https://uptime.betterstack.com/api/v3/incidents?resolved=false&per_page=50' + local budget=${FM_CHECK_TIMEOUT:-30} started_at=$SECONDS elapsed remaining curl_timeout [ -n "$token" ] || { emit_error_once "missing BETTER_STACK_API_TOKEN"; return 0; } [[ "$token" =~ ^[A-Za-z0-9._~+/=-]+$ ]] \ @@ -91,67 +87,60 @@ poll_with_injected_token() { command -v curl >/dev/null 2>&1 || { emit_error_once "missing curl"; return 0; } command -v jq >/dev/null 2>&1 || { emit_error_once "missing jq"; return 0; } - raw=$(printf 'header = "Authorization: Bearer %s"\n' "$token" \ - | curl \ - --config - \ - --request GET \ - --url 'https://uptime.betterstack.com/api/v3/incidents?resolved=false&per_page=50' \ - --header 'Accept: application/json' \ - --connect-timeout 3 \ - --max-time 5 \ - --silent \ - --show-error \ - --write-out '\n%{http_code}' 2>/dev/null) \ - || { emit_error_once "Better Stack API unreachable"; return 0; } - case "$raw" in - *$'\n'*) ;; - *) emit_error_once "Better Stack API returned no status"; return 0 ;; - esac - code=${raw##*$'\n'} - body=${raw%$'\n'*} - [ "$code" = 200 ] || { emit_error_once "API returned HTTP $code"; return 0; } - - rows=$(printf '%s' "$body" | jq -r ' - if (.data | type) != "array" then error("data must be an array") else .data[] end - | select(.type == "incident") - | select(.attributes.resolved_at == null) - | select(.id | type == "string" and test("^[0-9]+$")) - | [ - .id, - ((.attributes.name // "unknown") | tostring | gsub("[[:space:][:cntrl:]]+"; " ") | .[0:120]), - ((.attributes.started_at // "unknown") | tostring | gsub("[[:space:][:cntrl:]]+"; " ") | .[0:64]) - ] - | @tsv - ' 2>/dev/null) || { emit_error_once "invalid Better Stack API response"; return 0; } + rows= + while [ -n "$next_url" ]; do + elapsed=$((SECONDS - started_at)) + remaining=$((budget - elapsed - 1)) + [ "$remaining" -gt 0 ] || { emit_error_once "Better Stack API poll timed out"; return 0; } + curl_timeout=$remaining + [ "$curl_timeout" -gt 5 ] && curl_timeout=5 + raw=$(printf 'header = "Authorization: Bearer %s"\n' "$token" \ + | curl --config - --request GET --url "$next_url" \ + --header 'Accept: application/json' --connect-timeout 3 \ + --max-time "$curl_timeout" --silent --show-error \ + --write-out '\n%{http_code}' 2>/dev/null) \ + || { emit_error_once "Better Stack API unreachable"; return 0; } + case "$raw" in + *$'\n'*) ;; + *) emit_error_once "Better Stack API returned no status"; return 0 ;; + esac + code=${raw##*$'\n'} + body=${raw%$'\n'*} + [ "$code" = 200 ] || { emit_error_once "API returned HTTP $code"; return 0; } + page_rows=$(printf '%s' "$body" | jq -r ' + if (.data | type) != "array" then error("data must be an array") else .data[] end + | select(.type == "incident") + | select(.attributes.resolved_at == null) + | select(.id | type == "string" and test("^[0-9]+$")) + | [ + .id, + ((.attributes.name // "unknown") | tostring | gsub("[[:space:][:cntrl:]]+"; " ") | .[0:120]), + ((.attributes.started_at // "unknown") | tostring | gsub("[[:space:][:cntrl:]]+"; " ") | .[0:64]) + ] + | @tsv + ' 2>/dev/null) || { emit_error_once "invalid Better Stack API response"; return 0; } + [ -z "$rows" ] || [ -z "$page_rows" ] || rows="$rows"$'\n' + rows="$rows$page_rows" + page_next=$(printf '%s' "$body" | jq -r ' + if ((.pagination.next // null) != null and (.pagination.next | type) != "string") + then error("pagination.next must be a string") + else (.pagination.next // "") end + ' 2>/dev/null) || { emit_error_once "invalid Better Stack API response"; return 0; } + case "$page_next" in + '') next_url= ;; + https://uptime.betterstack.com/api/v3/incidents\?*) + case "$page_next" in *resolved=false*) next_url=$page_next ;; *) emit_error_once "invalid Better Stack pagination target"; return 0 ;; esac + ;; + *) emit_error_once "invalid Better Stack pagination target"; return 0 ;; + esac + done while IFS=$'\t' read -r id name started; do [ -n "$id" ] || continue - claim_incident "$id" - claim_rc=$? - case "$claim_rc" in - 0) - new_count=$((new_count + 1)) - if [ "$new_count" -eq 1 ]; then - first_id=$id - first_name=$name - first_started=$started - new_ids=$id - else - new_ids="$new_ids,$id" - fi - ;; - 1) ;; - *) emit_error_once "cannot record incident dedupe state"; return 0 ;; - esac + printf 'better-stack-incident opened id=%s name=%s started=%s\n' "$id" "$name" "$started" done <<< "$rows" clear_error - case "$new_count" in - 0) ;; - 1) printf 'better-stack-incident opened id=%s name=%s started=%s\n' \ - "$first_id" "$first_name" "$first_started" ;; - *) printf 'better-stack-incidents opened ids=%s\n' "$new_ids" ;; - esac } case "${1:-}" in diff --git a/bin/fm-brief.sh b/bin/fm-brief.sh index 9c98723b013..c7db910d27b 100755 --- a/bin/fm-brief.sh +++ b/bin/fm-brief.sh @@ -104,6 +104,9 @@ for a in "$@"; do *) POS+=("$a") ;; esac done +[ "${#POS[@]}" -ge 1 ] || { echo "error: missing task ID" >&2; exit 1; } +[ "$KIND" = secondmate ] || [ "${#POS[@]}" -ge 2 ] \ + || { echo "error: missing repository" >&2; exit 1; } ID=${POS[0]} if [ "$KIND" = secondmate ] && [ "$HERDR_LAB" -eq 1 ]; then diff --git a/bin/fm-wake-lib.sh b/bin/fm-wake-lib.sh index 8cec58bec1d..38d29418b84 100755 --- a/bin/fm-wake-lib.sh +++ b/bin/fm-wake-lib.sh @@ -406,6 +406,34 @@ fm_wake_append() { return "$status" } +fm_wake_append_incident_once() { + local id=$1 payload=$2 key="better-stack-incident:$id" + local seen_dir="$STATE/better-stack-incidents.seen" seen_file="$STATE/better-stack-incidents.seen/$id" + local epoch seq seq_file status=0 + fm_lock_acquire_wait "$FM_WAKE_QUEUE_LOCK" + if [ -f "$seen_file" ] || awk -F '\t' -v key="$key" '$3 == "check" && $4 == key { found=1 } END { exit !found }' "$FM_WAKE_QUEUE" 2>/dev/null; then + if [ ! -f "$seen_file" ]; then + (umask 077; mkdir -p "$seen_dir"; printf 'seen\n' > "$seen_file") || status=2 + fi + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return "$status" + fi + epoch=$(date +%s) + seq_file="$STATE/.wake-queue.seq" + seq=$(cat "$seq_file" 2>/dev/null || echo 0) + case "$seq" in ''|*[!0-9]*) seq=0 ;; esac + seq=$((seq + 1)) + printf '%s\n' "$seq" > "$seq_file" || status=$? + if [ "$status" -eq 0 ]; then + printf '%s\t%s\tcheck\t%s\t%s\n' "$epoch" "$seq" "$key" "$payload" >> "$FM_WAKE_QUEUE" || status=$? + fi + if [ "$status" -eq 0 ]; then + (umask 077; mkdir -p "$seen_dir"; printf 'seen\n' > "$seen_file") || status=2 + fi + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return "$status" +} + fm_wake_restore_queue() { local drained=$1 restore restore="$STATE/.wake-queue.restore.$(fm_current_pid)" diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index e5501f852b3..8c4df518379 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -780,8 +780,23 @@ while :; do fi fi if [ -n "$out" ]; then - reason="check: $c: $out" - fm_wake_append check "$c" "$reason" || exit 1 + if [ "$(basename "$c")" = better-stack-incidents.check.sh ]; then + while IFS= read -r incident_line; do + [ -n "$incident_line" ] || continue + case "$incident_line" in + better-stack-incident\ opened\ id=*) + id=${incident_line#better-stack-incident opened id=} + id=${id%% *} + reason="check: $c: $incident_line" + fm_wake_append_incident_once "$id" "$reason" || exit 1 + ;; + *) reason="check: $c: $incident_line"; fm_wake_append check "$c" "$reason" || exit 1 ;; + esac + done <<< "$out" + else + reason="check: $c: $out" + fm_wake_append check "$c" "$reason" || exit 1 + fi if [ "$is_pr_poll" -eq 1 ] && [ "$out" = merged ]; then if fm_pr_poll_retirement_publish "$STATE" "$id" "$SCRIPT_DIR/fm-pr-poll.sh" "$out"; then fm_pr_poll_retirement_recover_one "$STATE" "$id" "$SCRIPT_DIR/fm-pr-poll.sh" \ diff --git a/tests/fm-better-stack-incidents.test.sh b/tests/fm-better-stack-incidents.test.sh index 86320bea0f4..8f569eafda1 100755 --- a/tests/fm-better-stack-incidents.test.sh +++ b/tests/fm-better-stack-incidents.test.sh @@ -36,16 +36,22 @@ shift exec env BETTER_STACK_API_TOKEN="${FM_TEST_BETTER_STACK_TOKEN:-}" "$@" SH - cat > "$fakebin/curl" <<'SH' +cat > "$fakebin/curl" <<'SH' #!/usr/bin/env bash cat >/dev/null printf '%s\n' "$*" >> "$FM_TEST_CURL_LOG" [ "${FM_TEST_CURL_FAIL:-0}" = 0 ] || exit 7 -printf '%s\n%s' "${FM_TEST_API_BODY:-}" "${FM_TEST_API_CODE:-200}" +count=$(cat "$FM_TEST_CURL_COUNT" 2>/dev/null || echo 0) +count=$((count + 1)) +printf '%s\n' "$count" > "$FM_TEST_CURL_COUNT" +body=${FM_TEST_API_BODY:-} +[ "$count" -eq 2 ] && body=${FM_TEST_API_BODY_2:-$body} +printf '%s\n%s' "$body" "${FM_TEST_API_CODE:-200}" SH chmod +x "$fakebin/doppler" "$fakebin/curl" : > "$dir/doppler.log" : > "$dir/curl.log" + : > "$dir/curl.count" printf '%s\n' "$dir" } @@ -56,6 +62,7 @@ run_poll() { FM_HOME="$dir/home" \ FM_TEST_DOPPLER_LOG="$dir/doppler.log" \ FM_TEST_CURL_LOG="$dir/curl.log" \ + FM_TEST_CURL_COUNT="$dir/curl.count" \ "$@" "$POLL" } @@ -76,14 +83,6 @@ test_new_incident_and_duplicate_suppression() { expect_code 0 "$rc" "new incident poll exit" [ "$out" = 'better-stack-incident opened id=25 name=api-production started=2026-08-01T12:00:00.000Z' ] \ || fail "new incident must print one compact identity line (got: $out)" - assert_present "$dir/home/state/better-stack-incidents.seen/25" \ - "new incident must claim a private seen marker" - - out=$(run_poll "$dir" env \ - FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ - FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200); rc=$? - expect_code 0 "$rc" "duplicate incident poll exit" - [ -z "$out" ] || fail "an already-seen incident must stay silent (got: $out)" assert_grep '--project fleet-observability --config prd' "$dir/doppler.log" \ "poll must select the fleet-observability/prd Doppler scope" @@ -99,6 +98,25 @@ test_new_incident_and_duplicate_suppression() { pass "new Better Stack incident wakes once and duplicate observations stay silent" } +test_pagination_and_unsafe_target_rejection() { + local dir body next_body out rc + dir=$(make_case pagination) + body=$(jq -cn --arg next 'https://uptime.betterstack.com/api/v3/incidents?page=2&resolved=false' \ + '{data: [{id:"25", type:"incident", attributes:{name:"page-one", started_at:"t1", resolved_at:null}}], pagination:{next:$next}}') + next_body=$(incident_body 26 page-two t2) + out=$(run_poll "$dir" env FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_BODY_2="$next_body" FM_TEST_API_CODE=200) + [ "$out" = $'better-stack-incident opened id=25 name=page-one started=t1\nbetter-stack-incident opened id=26 name=page-two started=t2' ] \ + || fail "poll must include unresolved incidents from later pages (got: $out)" + body=$(jq -cn '{data: [], pagination:{next:"https://evil.example/api/v3/incidents?resolved=false"}}') + out=$(run_poll "$dir" env FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200); rc=$? + expect_code 0 "$rc" "unsafe pagination target poll exit" + [ "$out" = 'better-stack-error invalid Better Stack pagination target' ] \ + || fail "unsafe pagination target must produce one diagnostic (got: $out)" + pass "Better Stack pagination reaches later pages and rejects unsafe targets" +} + test_api_error_reports_once_and_recovers() { local dir out rc dir=$(make_case api-error) @@ -175,6 +193,7 @@ SH out=$(PATH="$dir/fakebin:$BASE_PATH" \ FM_HOME="$dir/home" FM_STATE_OVERRIDE="$state" \ FM_TEST_DOPPLER_LOG="$dir/doppler.log" FM_TEST_CURL_LOG="$dir/curl.log" \ + FM_TEST_CURL_COUNT="$dir/curl.count" \ FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200 \ FM_CHECK_INTERVAL=0 FM_HEARTBEAT=999999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ @@ -184,6 +203,18 @@ SH "registered poll output must become a check wake with the incident identity" [ "$(grep -c 'better-stack-incident opened id=91' "$state/.wake-queue")" -eq 1 ] \ || fail "new incident must create exactly one durable wake record" + rm -f "$state/.last-check" + out=$(PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$dir/home" FM_STATE_OVERRIDE="$state" \ + FM_TEST_DOPPLER_LOG="$dir/doppler.log" FM_TEST_CURL_LOG="$dir/curl.log" \ + FM_TEST_CURL_COUNT="$dir/curl.count" \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200 \ + FM_CHECK_INTERVAL=0 FM_HEARTBEAT=999999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + "$WATCH"); rc=$? + expect_code 0 "$rc" "duplicate watcher incident delivery exit" + [ "$(grep -c 'better-stack-incident opened id=91' "$state/.wake-queue")" -eq 1 ] \ + || fail "replayed incident must not create a duplicate durable wake" pass "registered Better Stack poll delivers exactly one authenticated check wake" } @@ -239,6 +270,7 @@ test_incident_check_keeps_home_supervised() { } test_new_incident_and_duplicate_suppression +test_pagination_and_unsafe_target_rejection test_api_error_reports_once_and_recovers test_missing_token_reports_once test_registered_check_delivers_check_wake From c03efd39fb5b55c336fa36be4edc0788f4f2c8cb Mon Sep 17 00:00:00 2001 From: juniorlovestmh <272474227+juniorlovestmh@users.noreply.github.com> Date: Sat, 1 Aug 2026 09:20:48 -0300 Subject: [PATCH 47/52] no-mistakes(review): Captain: added receipt recovery and documented watcher-owned dedupe --- bin/fm-bootstrap.sh | 4 +- bin/fm-wake-lib.sh | 66 ++++++++++++++++++++++--- docs/configuration.md | 10 ++-- tests/fm-better-stack-incidents.test.sh | 48 ++++++++++++++++++ 4 files changed, 115 insertions(+), 13 deletions(-) diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index 66e4406e4dc..2f41032f165 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -728,8 +728,8 @@ EOF # A presence flag at config/better-stack-incidents materializes one ordinary # registered custom check, so the existing hash-bound snapshot execution and # FM_CHECK_TIMEOUT contract remain the only slow-check mechanism. -# The poll itself performs runtime-only Doppler injection and owns incident and -# diagnostic dedupe in this home's private state. +# The poll itself performs runtime-only Doppler injection; the watcher owns +# incident delivery dedupe while the poll owns diagnostic dedupe. better_stack_incidents_setup() { local flag check trust check_body tool missing check_home failed flag="$CONFIG/better-stack-incidents" diff --git a/bin/fm-wake-lib.sh b/bin/fm-wake-lib.sh index 38d29418b84..8de6b3b3c5d 100755 --- a/bin/fm-wake-lib.sh +++ b/bin/fm-wake-lib.sh @@ -406,17 +406,63 @@ fm_wake_append() { return "$status" } +fm_wake_incident_receipt_valid() { + local receipt=$1 id=$2 version receipt_id payload extra + local receipt_dir=${receipt%/*} base=${receipt##*/} + fmx_private_artifact_file_valid "$receipt_dir" "$base" 600 || return 1 + exec 9< "$receipt" || return 1 + IFS= read -r version <&9 || { exec 9<&-; return 1; } + IFS= read -r receipt_id <&9 || { exec 9<&-; return 1; } + IFS= read -r payload <&9 || { exec 9<&-; return 1; } + if IFS= read -r extra <&9; then + exec 9<&- + return 1 + fi + exec 9<&- + [ "$version" = fm-better-stack-incident-receipt-v1 ] || return 1 + [ "$receipt_id" = "$id" ] || return 1 + [ -n "$payload" ] +} + fm_wake_append_incident_once() { local id=$1 payload=$2 key="better-stack-incident:$id" local seen_dir="$STATE/better-stack-incidents.seen" seen_file="$STATE/better-stack-incidents.seen/$id" - local epoch seq seq_file status=0 + local receipt_dir="$STATE/better-stack-incidents.receipts" receipt="$STATE/better-stack-incidents.receipts/$id" + local epoch seq seq_file status=0 receipt_rc marker_rc + [[ "$id" =~ ^[0-9]+$ ]] || return 2 + declare -F fmx_private_artifact_file_valid >/dev/null 2>&1 || return 2 + declare -F fmx_private_artifact_publish_stdin_once >/dev/null 2>&1 || return 2 fm_lock_acquire_wait "$FM_WAKE_QUEUE_LOCK" - if [ -f "$seen_file" ] || awk -F '\t' -v key="$key" '$3 == "check" && $4 == key { found=1 } END { exit !found }' "$FM_WAKE_QUEUE" 2>/dev/null; then - if [ ! -f "$seen_file" ]; then - (umask 077; mkdir -p "$seen_dir"; printf 'seen\n' > "$seen_file") || status=2 + if fmx_private_artifact_file_valid "$seen_dir" "$id" 600; then + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return 0 + elif [ -e "$seen_file" ] || [ -L "$seen_file" ]; then + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return 2 + fi + if fm_wake_incident_receipt_valid "$receipt" "$id"; then + printf 'seen\n' | fmx_private_artifact_publish_stdin_once "$seen_dir" "$id" 600 >/dev/null 2>&1 + marker_rc=$? + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + [ "$marker_rc" -eq 0 ] || [ "$marker_rc" -eq 1 ] || return 2 + return 0 + elif [ -e "$receipt" ] || [ -L "$receipt" ]; then + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return 2 + fi + if awk -F '\t' -v key="$key" '$3 == "check" && $4 == key { found=1 } END { exit !found }' "$FM_WAKE_QUEUE" 2>/dev/null; then + printf 'fm-better-stack-incident-receipt-v1\n%s\n%s\n' "$id" "$payload" \ + | fmx_private_artifact_publish_stdin_once "$receipt_dir" "$id" 600 >/dev/null 2>&1 + receipt_rc=$? + if [ "$receipt_rc" -ne 0 ] && [ "$receipt_rc" -ne 1 ]; then + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return 2 fi + printf 'seen\n' | fmx_private_artifact_publish_stdin_once "$seen_dir" "$id" 600 >/dev/null 2>&1 + marker_rc=$? fm_lock_release "$FM_WAKE_QUEUE_LOCK" - return "$status" + [ "$marker_rc" -eq 0 ] || [ "$marker_rc" -eq 1 ] || return 2 + return 0 fi epoch=$(date +%s) seq_file="$STATE/.wake-queue.seq" @@ -428,7 +474,15 @@ fm_wake_append_incident_once() { printf '%s\t%s\tcheck\t%s\t%s\n' "$epoch" "$seq" "$key" "$payload" >> "$FM_WAKE_QUEUE" || status=$? fi if [ "$status" -eq 0 ]; then - (umask 077; mkdir -p "$seen_dir"; printf 'seen\n' > "$seen_file") || status=2 + printf 'fm-better-stack-incident-receipt-v1\n%s\n%s\n' "$id" "$payload" \ + | fmx_private_artifact_publish_stdin_once "$receipt_dir" "$id" 600 >/dev/null 2>&1 + receipt_rc=$? + [ "$receipt_rc" -eq 0 ] || [ "$receipt_rc" -eq 1 ] || status=2 + fi + if [ "$status" -eq 0 ]; then + printf 'seen\n' | fmx_private_artifact_publish_stdin_once "$seen_dir" "$id" 600 >/dev/null 2>&1 + marker_rc=$? + [ "$marker_rc" -eq 0 ] || [ "$marker_rc" -eq 1 ] || status=2 fi fm_lock_release "$FM_WAKE_QUEUE_LOCK" return "$status" diff --git a/docs/configuration.md b/docs/configuration.md index 09a123ba136..b89ee94a4dc 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -310,7 +310,7 @@ Skipped items, such as a destination checkout that does not yet gitignore the it Create an ordinary local file at `config/better-stack-incidents` in the one Firstmate home that should receive fleet incident notifications. The file is a presence flag with no secret content, is gitignored, and is deliberately not inherited into secondmate homes so one incident does not alert multiple supervisors. The next locked session-start bootstrap requires `doppler`, `curl`, and `jq`, writes `state/better-stack-incidents.check.sh`, and binds those exact shim bytes in `state/better-stack-incidents.check-trust` through `bin/fm-check-register.sh`. -This uses the existing registered custom-check extension point: the watcher executes only a hash-validated private snapshot, applies `FM_CHECK_TIMEOUT`, and converts the poll's one-line output into a durable `check:` notification. +This uses the existing registered custom-check extension point: the watcher executes only a hash-validated private snapshot, applies `FM_CHECK_TIMEOUT`, and converts each incident line in the poll output into a durable `check:` notification. The registered poll is a home-level supervision need even with no project work in flight and runs on the default `FM_CHECK_INTERVAL=300` slow-check cadence, which keeps normal detection within minutes without adding a second scheduler. `bin/fm-better-stack-incidents-poll.sh` invokes its poll child with `doppler run --silent --no-check-version --no-fallback --project fleet-observability --config prd --only-secrets BETTER_STACK_API_TOKEN`. @@ -318,13 +318,13 @@ The registered poll is a home-level supervision need even with no project work i The token therefore remains runtime-only and must never be added to the flag, repository, logs, task instructions, or another local file. The poll requests unresolved incidents from Better Stack's documented [`GET /api/v3/incidents`](https://betterstack.com/docs/uptime/api/list-all-incidents/) endpoint with a five-second HTTP bound. -A previously unseen incident ID creates `state/better-stack-incidents.seen/<id>` as a private single-link marker and prints one compact identity line. -One response containing several unseen incidents prints one line containing all new IDs, so a check cycle still produces exactly one durable notification. -Already-seen incidents and quiet API responses print nothing. +A poll prints one compact identity line for each unresolved incident; it does not claim delivery state before the watcher appends the wake. +The watcher owns the durable commit point: under the wake-queue lock it appends one `check:` record per incident, publishes the private identity-bound receipt at `state/better-stack-incidents.receipts/<id>`, and then publishes `state/better-stack-incidents.seen/<id>`. +If the watcher stops after queue append, the receipt survives queue drain and the next watcher run completes the seen marker without appending a second wake. Already-seen incidents and quiet API responses print nothing. Missing credentials, Doppler access failure, network failure, non-success HTTP status, and malformed API data print one `better-stack-error ...` diagnostic and record it in `state/better-stack-incidents.diagnostics/error`; the same diagnostic then remains silent until a successful poll clears the marker or a different failure occurs. Remove `config/better-stack-incidents` and rerun locked session start to retire the runnable check and its trust binding. -Bootstrap retains the private seen-ID and diagnostic markers so disabling and later re-enabling the poll cannot re-notify every still-open incident. +Bootstrap retains the private seen-ID, receipt, and diagnostic markers so disabling and later re-enabling the poll cannot re-notify every still-open incident. Better Stack polling from heartbeat handling was an interim practice and is retired; the registered check is the only poll owner, while [`AGENTS.md` section 8](../AGENTS.md#8-supervision-protocol) owns incident triage after a notification arrives. ## X mode (.env) diff --git a/tests/fm-better-stack-incidents.test.sh b/tests/fm-better-stack-incidents.test.sh index 8f569eafda1..4b6eb4061e2 100755 --- a/tests/fm-better-stack-incidents.test.sh +++ b/tests/fm-better-stack-incidents.test.sh @@ -218,6 +218,53 @@ SH pass "registered Better Stack poll delivers exactly one authenticated check wake" } +test_incident_receipt_recovers_after_marker_failure_and_queue_drain() { + local dir state shim body out rc + dir=$(make_case receipt-recovery) + state="$dir/home/state" + shim="$state/better-stack-incidents.check.sh" + body=$(incident_body 92 recovery-production) + cat > "$shim" <<SH +#!/usr/bin/env bash +export FM_HOME=$(printf '%q' "$dir/home") +exec $(printf '%q' "$POLL") +SH + chmod 0700 "$shim" + FM_HOME="$dir/home" "$REGISTER" better-stack-incidents >/dev/null \ + || fail "could not register receipt-recovery custom check" + printf 'not-a-directory\n' > "$state/better-stack-incidents.seen" + out=$(PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$dir/home" FM_STATE_OVERRIDE="$state" \ + FM_TEST_DOPPLER_LOG="$dir/doppler.log" FM_TEST_CURL_LOG="$dir/curl.log" \ + FM_TEST_CURL_COUNT="$dir/curl.count" FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200 \ + FM_CHECK_INTERVAL=0 FM_HEARTBEAT=999999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + "$WATCH"); rc=$? + [ "$rc" -ne 0 ] || fail "marker publication failure must stop the watcher" + assert_present "$state/better-stack-incidents.receipts/92" \ + "queue append failure boundary must leave a private recovery receipt" + [ "$(grep -c 'better-stack-incident opened id=92' "$state/.wake-queue")" -eq 1 ] \ + || fail "failed marker publication must retain exactly one queued wake" + rm -f "$state/.wake-queue" + rm -f "$state/better-stack-incidents.seen" + mkdir "$state/better-stack-incidents.seen" + chmod 700 "$state/better-stack-incidents.seen" + rm -f "$state/.last-check" + out=$(PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$dir/home" FM_STATE_OVERRIDE="$state" \ + FM_TEST_DOPPLER_LOG="$dir/doppler.log" FM_TEST_CURL_LOG="$dir/curl.log" \ + FM_TEST_CURL_COUNT="$dir/curl.count" FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200 \ + FM_CHECK_INTERVAL=0 FM_HEARTBEAT=999999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + "$WATCH"); rc=$? + expect_code 0 "$rc" "receipt recovery watcher exit" + assert_present "$state/better-stack-incidents.seen/92" \ + "receipt recovery must converge the private seen marker after queue drain" + [ ! -e "$state/.wake-queue" ] || [ -z "$(cat "$state/.wake-queue")" ] \ + || fail "receipt recovery after queue drain must not append a duplicate wake" + pass "incident receipt recovers marker publication after queue drain" +} + test_bootstrap_arms_and_retires_home_check() { local dir home out sum1 sum2 dir=$(make_case bootstrap) @@ -274,5 +321,6 @@ test_pagination_and_unsafe_target_rejection test_api_error_reports_once_and_recovers test_missing_token_reports_once test_registered_check_delivers_check_wake +test_incident_receipt_recovers_after_marker_failure_and_queue_drain test_bootstrap_arms_and_retires_home_check test_incident_check_keeps_home_supervised From 9a6c6a6cb21c8575bcdf62422ed15c15147ea35b Mon Sep 17 00:00:00 2001 From: juniorlovestmh <272474227+juniorlovestmh@users.noreply.github.com> Date: Sat, 1 Aug 2026 12:58:50 -0300 Subject: [PATCH 48/52] no-mistakes(review): Captain, documentation now describes repeated poll output, watcher dedupe, and receipt-failure tradeoffs --- docs/configuration.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/configuration.md b/docs/configuration.md index b89ee94a4dc..2663b48cc91 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -320,7 +320,7 @@ The poll requests unresolved incidents from Better Stack's documented [`GET /api A poll prints one compact identity line for each unresolved incident; it does not claim delivery state before the watcher appends the wake. The watcher owns the durable commit point: under the wake-queue lock it appends one `check:` record per incident, publishes the private identity-bound receipt at `state/better-stack-incidents.receipts/<id>`, and then publishes `state/better-stack-incidents.seen/<id>`. -If the watcher stops after queue append, the receipt survives queue drain and the next watcher run completes the seen marker without appending a second wake. Already-seen incidents and quiet API responses print nothing. +If the watcher stops after queue append, the receipt survives queue drain and the next watcher run completes the seen marker without appending a second wake. The poll emits every unresolved incident on each scan; the watcher suppresses already-delivered incident IDs, so repeated poll output normally remains silent at the durable wake boundary. If receipt publication itself fails after queue append and that queue record is drained before retry, a rare duplicate wake can occur; the wake handler deduplicates by incident ID. Missing credentials, Doppler access failure, network failure, non-success HTTP status, and malformed API data print one `better-stack-error ...` diagnostic and record it in `state/better-stack-incidents.diagnostics/error`; the same diagnostic then remains silent until a successful poll clears the marker or a different failure occurs. Remove `config/better-stack-incidents` and rerun locked session start to retire the runnable check and its trust binding. From a37f15af75fe101800e2c08b1c86ae580e41b51a Mon Sep 17 00:00:00 2001 From: juniorlovestmh <272474227+juniorlovestmh@users.noreply.github.com> Date: Sat, 1 Aug 2026 13:06:32 -0300 Subject: [PATCH 49/52] no-mistakes(document): Captain: document Better Stack incident monitoring --- README.md | 1 + docs/configuration.md | 4 +++- docs/documentation-audiences.json | 40 +++++++++++++++++++++++++++++++ 3 files changed, 44 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index 5a03c4b9b1b..ae916431584 100644 --- a/README.md +++ b/README.md @@ -48,6 +48,7 @@ Launching a supported harness inside it instantiates your first mate - and makes - **Explicit project modes** - each project ships via `no-mistakes`, `direct-PR`, or `local-only`, with an optional `+yolo` autonomy flag. - **Optional secondmates** - opt in to persistent second mates that run from isolated firstmate homes with their own `FM_HOME`, state, projects, and session lock, supervising project clones or a project-less firstmate-repo domain, kept on the primary firstmate version by guarded local fast-forwards and checked for live agent processes at session start. - **Event-driven, zero-token supervision** - a bash watcher sleeps on the fleet and wakes the first mate only when something needs you; verified primary harnesses also get a turn-end backstop that blocks or follows up on a blind stop when work is under way and supervision is not live. +- **Optional Better Stack incident monitoring** - opt one home into unresolved-incident polling through the registered custom-check path; runtime-only Doppler injection, private incident dedupe, and one visible diagnostic per failure keep the alert path bounded. - **Optional X mode** - opt in with one local `.env` token so firstmate can answer your public `@myfirstmate` mentions, act on normal reversible mention requests through the same lifecycle as chat requests, acknowledge spawned work, and post up to three public-safe completion follow-ups within seven days for genuine milestones and the final outcome without changing non-X behavior; dry-run preview records would-be replies and dismissals locally before go-live. - **Strict project boundary** - the first mate is read-only over your projects except for the narrow guarded and captain-approved operations authorized by [hard rule 1](AGENTS.md#1-identity-and-prime-directives), including fleet sync's guarded safe branch pruning; crewmates make every other project change behind the configured merge authority. - **Restart-proof** - all state lives on disk and in the active session backend (tmux by hard default, herdr or cmux when selected or auto-detected, zellij/orca when explicitly selected); kill the session anytime and the next one reconciles, including confirmed-dead secondmate agents, and carries on. diff --git a/docs/configuration.md b/docs/configuration.md index 2663b48cc91..37742d05af6 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -320,7 +320,9 @@ The poll requests unresolved incidents from Better Stack's documented [`GET /api A poll prints one compact identity line for each unresolved incident; it does not claim delivery state before the watcher appends the wake. The watcher owns the durable commit point: under the wake-queue lock it appends one `check:` record per incident, publishes the private identity-bound receipt at `state/better-stack-incidents.receipts/<id>`, and then publishes `state/better-stack-incidents.seen/<id>`. -If the watcher stops after queue append, the receipt survives queue drain and the next watcher run completes the seen marker without appending a second wake. The poll emits every unresolved incident on each scan; the watcher suppresses already-delivered incident IDs, so repeated poll output normally remains silent at the durable wake boundary. If receipt publication itself fails after queue append and that queue record is drained before retry, a rare duplicate wake can occur; the wake handler deduplicates by incident ID. +If the watcher stops after queue append, the receipt survives queue drain and the next watcher run completes the seen marker without appending a second wake. +The poll emits every unresolved incident on each scan; the watcher suppresses already-delivered incident IDs, so repeated poll output normally remains silent at the durable wake boundary. +If receipt publication itself fails after queue append and that queue record is drained before retry, a rare duplicate wake can occur; the wake handler deduplicates by incident ID. Missing credentials, Doppler access failure, network failure, non-success HTTP status, and malformed API data print one `better-stack-error ...` diagnostic and record it in `state/better-stack-incidents.diagnostics/error`; the same diagnostic then remains silent until a successful poll clears the marker or a different failure occurs. Remove `config/better-stack-incidents` and rerun locked session start to retire the runnable check and its trust binding. diff --git a/docs/documentation-audiences.json b/docs/documentation-audiences.json index 8f70cd346c2..c9dc1dcb5e9 100644 --- a/docs/documentation-audiences.json +++ b/docs/documentation-audiences.json @@ -151,6 +151,14 @@ "path": ".agents/skills/harness-adapters/SKILL.md", "audience": "agent-runtime" }, + { + "path": ".agents/skills/i-have-adhd/SKILL.md", + "audience": "agent-runtime" + }, + { + "path": ".agents/skills/mobile-mode/SKILL.md", + "audience": "agent-runtime" + }, { "path": ".agents/skills/project-management/SKILL.md", "audience": "agent-runtime" @@ -163,6 +171,10 @@ "path": ".agents/skills/secondmate-provisioning/SKILL.md", "audience": "agent-runtime" }, + { + "path": ".agents/skills/secrets-management/SKILL.md", + "audience": "agent-runtime" + }, { "path": ".agents/skills/stow/SKILL.md", "audience": "agent-runtime" @@ -235,6 +247,18 @@ "path": "docs/examples/crew-dispatch.json", "audience": "operator-example" }, + { + "path": "docs/examples/doppler-oidc-job.yml", + "audience": "operator-example" + }, + { + "path": "docs/examples/doppler-service-token-job.yml", + "audience": "operator-example" + }, + { + "path": "docs/examples/project-secrets-policy.json", + "audience": "operator-example" + }, { "path": "docs/examples/wedge-alarm", "audience": "operator-example" @@ -255,10 +279,18 @@ "path": "docs/herdr-backend.md", "audience": "operator-current" }, + { + "path": "docs/moshi-mobile-review.md", + "audience": "operator-current" + }, { "path": "docs/orca-backend.md", "audience": "operator-current" }, + { + "path": "docs/promotion-ladder.md", + "audience": "operator-current" + }, { "path": "docs/scripts.md", "audience": "operator-current" @@ -303,10 +335,18 @@ "path": "docs/turnend-guard.md", "audience": "operator-current" }, + { + "path": "docs/toolchain-versions.md", + "audience": "maintainer-verification" + }, { "path": "docs/verification/runtime-backends.md", "audience": "maintainer-verification" }, + { + "path": "docs/verification/moshi-mobile-review.md", + "audience": "maintainer-verification" + }, { "path": "docs/verification/stow-memory.md", "audience": "maintainer-verification" From 1a5ac7b7cf716dd59fc681abec847d32c4da471c Mon Sep 17 00:00:00 2001 From: juniorlovestmh <272474227+juniorlovestmh@users.noreply.github.com> Date: Sat, 1 Aug 2026 13:13:21 -0300 Subject: [PATCH 50/52] no-mistakes(lint): Fix ShellCheck warnings in incident wake scripts --- bin/fm-better-stack-incidents-poll.sh | 1 - bin/fm-wake-lib.sh | 7 ++++--- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/bin/fm-better-stack-incidents-poll.sh b/bin/fm-better-stack-incidents-poll.sh index c80c4af16f8..0ac4bc29d89 100755 --- a/bin/fm-better-stack-incidents-poll.sh +++ b/bin/fm-better-stack-incidents-poll.sh @@ -23,7 +23,6 @@ FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" ERROR_DIR="$STATE/better-stack-incidents.diagnostics" ERROR_FILE="$ERROR_DIR/error" -SEEN_DIR="$STATE/better-stack-incidents.seen" # Reuse the watcher's existing private-artifact owner rather than introducing a # second atomic-publication contract for one extension. diff --git a/bin/fm-wake-lib.sh b/bin/fm-wake-lib.sh index 8de6b3b3c5d..16a373b5791 100755 --- a/bin/fm-wake-lib.sh +++ b/bin/fm-wake-lib.sh @@ -407,14 +407,14 @@ fm_wake_append() { } fm_wake_incident_receipt_valid() { - local receipt=$1 id=$2 version receipt_id payload extra + local receipt=$1 id=$2 version receipt_id payload local receipt_dir=${receipt%/*} base=${receipt##*/} fmx_private_artifact_file_valid "$receipt_dir" "$base" 600 || return 1 exec 9< "$receipt" || return 1 IFS= read -r version <&9 || { exec 9<&-; return 1; } IFS= read -r receipt_id <&9 || { exec 9<&-; return 1; } IFS= read -r payload <&9 || { exec 9<&-; return 1; } - if IFS= read -r extra <&9; then + if IFS= read -r <&9; then exec 9<&- return 1 fi @@ -425,7 +425,8 @@ fm_wake_incident_receipt_valid() { } fm_wake_append_incident_once() { - local id=$1 payload=$2 key="better-stack-incident:$id" + local id=$1 payload=$2 + local key="better-stack-incident:$id" local seen_dir="$STATE/better-stack-incidents.seen" seen_file="$STATE/better-stack-incidents.seen/$id" local receipt_dir="$STATE/better-stack-incidents.receipts" receipt="$STATE/better-stack-incidents.receipts/$id" local epoch seq seq_file status=0 receipt_rc marker_rc From f29ab685f5796be542360b5597bad85bd17c4d65 Mon Sep 17 00:00:00 2001 From: juniorlovestmh <272474227+juniorlovestmh@users.noreply.github.com> Date: Sat, 1 Aug 2026 13:25:29 -0300 Subject: [PATCH 51/52] chore: reconcile Better Stack wake onto main --- .agents/skills/mobile-mode/SKILL.md | 54 +++ .no-mistakes.yaml | 6 +- AGENTS.md | 1 + README.md | 2 + bin/fm-better-stack-incidents-poll.sh | 128 +++---- bin/fm-bootstrap.sh | 4 +- bin/fm-brief.sh | 116 +++++- bin/fm-install-shellcheck.sh | 36 +- bin/fm-operational-input.sh | 12 +- bin/fm-test-run.sh | 8 +- bin/fm-toolchain-mirror.sh | 457 +++++++++++++++++++++++ bin/fm-wake-lib.sh | 83 ++++ bin/fm-watch.sh | 19 +- docs/configuration.md | 13 +- docs/documentation-audiences.json | 16 + docs/moshi-mobile-review.md | 92 +++++ docs/scripts.md | 1 + docs/toolchain-versions.md | 104 ++++++ docs/verification/moshi-mobile-review.md | 67 ++++ tests/fm-better-stack-incidents.test.sh | 100 ++++- tests/fm-brief.test.sh | 89 ++++- tests/fm-lint.test.sh | 161 +++++++- tests/fm-operational-input.test.sh | 22 ++ tests/fm-toolchain-mirror.test.sh | 128 +++++++ 24 files changed, 1608 insertions(+), 111 deletions(-) create mode 100644 .agents/skills/mobile-mode/SKILL.md create mode 100755 bin/fm-toolchain-mirror.sh create mode 100644 docs/moshi-mobile-review.md create mode 100644 docs/toolchain-versions.md create mode 100644 docs/verification/moshi-mobile-review.md create mode 100755 tests/fm-toolchain-mirror.test.sh diff --git a/.agents/skills/mobile-mode/SKILL.md b/.agents/skills/mobile-mode/SKILL.md new file mode 100644 index 00000000000..dd316e46926 --- /dev/null +++ b/.agents/skills/mobile-mode/SKILL.md @@ -0,0 +1,54 @@ +--- +name: mobile-mode +description: >- + Shape captain-facing Firstmate messages and review handoffs for a phone, especially when Moshi is the active surface. +user-invocable: false +metadata: + internal: true +--- + +# mobile-mode + +Load this skill when the captain says they are in mobile mode or identifies Moshi as the active surface. +Continue following it until the captain says they are back on desktop or requests normal mode. + +This is a presentation profile over the same host-side Firstmate session. +It does not create a Moshi runtime backend, supervision path, authority channel, or webhook integration. +The always-loaded [`i-have-adhd`](../i-have-adhd/SKILL.md) contract remains the owner of general captain-facing presentation, and `AGENTS.md` section 9 remains the owner of outcome translation and approval escalation. +This skill owns only the mobile delta and the review-surface choice. + +## Message shape + +- Apply the `i-have-adhd` outcome-first rule, then keep the outcome, consequence, evidence, and requested action within one ordinary phone scroll whenever the required facts fit. +- Make choices answerable with one low-typing reply such as `1`, `2`, `yes`, `merge`, or `hold`. +- Number choices, put the recommendation first, and end with the exact short reply that will select it. +- Keep full `https://...` pull-request links under section 9's existing rule so the captain can open the review directly. +- Put long logs and secondary evidence in the existing private report, then summarize the consequence in chat. +- Do not require terminal copy mode, pane navigation, punctuation-heavy commands, or multi-step text entry to answer a decision. + +## Review-surface choice + +Use plain chat for a simple decision or whenever a rich surface is unnecessary or unavailable. +Use a host-local Lavish surface through Moshi Pro Browser Preview when several options or structured feedback benefit from a touch-friendly review. +Follow the operator runbook in [`docs/moshi-mobile-review.md`](../../../docs/moshi-mobile-review.md). + +After Lavish starts on the host, tell the captain to open Browser Preview in Moshi and choose the Lavish server. +Never present a raw `127.0.0.1`, `localhost`, `file://`, or desktop-only LAN URL as if the phone can open it directly. +If Browser Preview is unavailable, restate the complete decision in chat with numbered replies instead of suggesting public sharing. + +Use Moshi Pro Diff to inspect the connected working tree, while keeping a full HTTPS pull-request link in chat for a hosted PR review. +Use Chat View only when Moshi recognizes the active agent and session. For unsupported prompts, incomplete cards, or an unrecognized session, keep the same Moshi/Firstmate session and present the complete fallback as concise numbered Firstmate chat; the terminal remains the source of truth, but is not a separate mobile handoff surface. + +Private fleet reviews stay host-local. +Never invoke or suggest `lavish-axi share` for private fleet state, even with a password. +Public sharing of separately sanitized public material requires an explicit request and the ordinary outward-facing consent boundary. + +## `moshi-hook` authority + +The already-installed `moshi-hook` may surface the running agent's native inbox events and approvals in Moshi or Apple Watch. +An approval button may answer only the exact native agent prompt that Moshi can map safely to the same live session. +It never authorizes a Firstmate merge, product or scope decision, destructive or irreversible action, credential use, permission change, or security-sensitive choice. +Route those decisions through Firstmate chat under the existing authority contract. + +Do not add a Firstmate-to-Moshi webhook, change hook configuration, or place secrets in notification summaries. +When an approval is unavailable, ambiguous, or unsupported, return to the terminal or ask through Firstmate chat without weakening the underlying boundary. diff --git a/.no-mistakes.yaml b/.no-mistakes.yaml index 02e6128f2e9..a26ad4eb08b 100644 --- a/.no-mistakes.yaml +++ b/.no-mistakes.yaml @@ -34,7 +34,11 @@ document: # security, Herdr, tmux, and lifecycle coverage). A full-suite override here # would duplicate CI and defeat the targeted Test contract. commands: - lint: 'bin/fm-lint.sh' + lint: >- + shellcheck_dir=$(mktemp -d "${TMPDIR:-/tmp}/fm-no-mistakes-shellcheck.XXXXXX") && + trap 'rm -rf "$shellcheck_dir"' EXIT && + bin/fm-install-shellcheck.sh "$shellcheck_dir" >/dev/null && + PATH="$shellcheck_dir:$PATH" exec bin/fm-lint.sh # Keep test evidence out of this repo; it stays in a temp dir instead. test: diff --git a/AGENTS.md b/AGENTS.md index 0d9c48c219b..55bd8910bfd 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -395,6 +395,7 @@ Load `stuck-crewmate-recovery` after a stale wake, looping or confused pane, ans ## 9. Escalation and captain etiquette Load `i-have-adhd` before every captain-facing response; the skill owns presentation shape while this section owns outcome translation and internal-vocabulary rewriting. +When the captain says they are in mobile mode or identifies Moshi as the active surface, load `mobile-mode`; it owns the mobile presentation and review handoff delta until the captain returns to desktop or normal mode. **Talk in outcomes, not mechanics.** Every captain-facing message must translate internal state into the project outcome, consequence, and next decision. diff --git a/README.md b/README.md index e4d507c35f0..161c1253fd9 100644 --- a/README.md +++ b/README.md @@ -48,6 +48,7 @@ Launching a supported harness inside it instantiates your first mate - and makes - **Explicit project modes** - each project ships via `no-mistakes`, `direct-PR`, or `local-only`, with an optional `+yolo` autonomy flag. - **Optional secondmates** - opt in to persistent second mates that run from isolated firstmate homes with their own `FM_HOME`, state, projects, and session lock, supervising project clones or a project-less firstmate-repo domain, kept on the primary firstmate version by guarded local fast-forwards and checked for live agent processes at session start. - **Event-driven, zero-token supervision** - a bash watcher sleeps on the fleet and wakes the first mate only when something needs you; verified primary harnesses also get a turn-end backstop that blocks or follows up on a blind stop when work is under way and supervision is not live. +- **Optional Better Stack incident monitoring** - opt one home into unresolved-incident polling through the registered custom-check path; runtime-only Doppler injection, private incident dedupe, and one visible diagnostic per failure keep the alert path bounded. - **Optional X mode** - opt in with one local `.env` token so firstmate can answer your public `@myfirstmate` mentions, act on normal reversible mention requests through the same lifecycle as chat requests, acknowledge spawned work, and post up to three public-safe completion follow-ups within seven days for genuine milestones and the final outcome without changing non-X behavior; dry-run preview records would-be replies and dismissals locally before go-live. - **Strict project boundary** - the first mate is read-only over your projects except for the narrow guarded and captain-approved operations authorized by [hard rule 1](AGENTS.md#1-identity-and-prime-directives), including fleet sync's guarded safe branch pruning; crewmates make every other project change behind the configured merge authority. - **Doppler by default** - the conditional [secrets-management policy](.agents/skills/secrets-management/SKILL.md) prefers secretless provider identity, otherwise scopes Doppler by project and environment, and validates declarations and rollout data through `bin/fm-secrets-check.sh`. @@ -200,6 +201,7 @@ Firstmate's skills live in two separate places with different audiences: - [docs/architecture.md](docs/architecture.md) - maintainer architecture for the crew, supervision, worktrees, secondmates, and project modes. - [docs/configuration.md](docs/configuration.md) - environment variables, `FM_HOME`, runtime backend selection, optional X mode, the files you set, and harness support. - [docs/calm.md](docs/calm.md) - current Pi `/calm` behavior and supported presentation limits. +- [docs/moshi-mobile-review.md](docs/moshi-mobile-review.md) - host-local Moshi Pro Preview, Diff, Chat View, and private mobile-review fallbacks. - [docs/wedge-alarm.md](docs/wedge-alarm.md) - configure the active alert for an away-mode escalation delivery that gets stuck. - [docs/tmux-backend.md](docs/tmux-backend.md) - current setup and limits for the tmux reference backend. - [docs/herdr-backend.md](docs/herdr-backend.md) - current setup, safety boundaries, and limits for the experimental Herdr backend. diff --git a/bin/fm-better-stack-incidents-poll.sh b/bin/fm-better-stack-incidents-poll.sh index e7085bbde4e..0ac4bc29d89 100755 --- a/bin/fm-better-stack-incidents-poll.sh +++ b/bin/fm-better-stack-incidents-poll.sh @@ -23,7 +23,6 @@ FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" ERROR_DIR="$STATE/better-stack-incidents.diagnostics" ERROR_FILE="$ERROR_DIR/error" -SEEN_DIR="$STATE/better-stack-incidents.seen" # Reuse the watcher's existing private-artifact owner rather than introducing a # second atomic-publication contract for one extension. @@ -65,25 +64,21 @@ run_through_doppler() { fi case "$out" in '') return 0 ;; - *$'\n'*) emit_error_once "poll returned invalid multiline output" ;; - better-stack-incident\ opened\ *|better-stack-incidents\ opened\ *|better-stack-error\ *) - printf '%s\n' "$out" + *) + while IFS= read -r line; do + case "$line" in + better-stack-incident\ opened\ *|better-stack-error\ *) printf '%s\n' "$line" ;; + *) emit_error_once "poll returned invalid output"; return 0 ;; + esac + done <<< "$out" ;; - *) emit_error_once "poll returned invalid output" ;; esac } -claim_incident() { - local id=$1 rc - printf 'seen\n' \ - | fmx_private_artifact_publish_stdin_once "$SEEN_DIR" "$id" 600 2>/dev/null - rc=$? - return "$rc" -} - poll_with_injected_token() { - local token=${BETTER_STACK_API_TOKEN:-} raw code body rows id name started - local claim_rc new_count=0 new_ids= first_id= first_name= first_started= + local token=${BETTER_STACK_API_TOKEN:-} raw code body page_rows page_next + local id name started next_url='https://uptime.betterstack.com/api/v3/incidents?resolved=false&per_page=50' + local budget=${FM_CHECK_TIMEOUT:-30} started_at=$SECONDS elapsed remaining curl_timeout [ -n "$token" ] || { emit_error_once "missing BETTER_STACK_API_TOKEN"; return 0; } [[ "$token" =~ ^[A-Za-z0-9._~+/=-]+$ ]] \ @@ -91,67 +86,60 @@ poll_with_injected_token() { command -v curl >/dev/null 2>&1 || { emit_error_once "missing curl"; return 0; } command -v jq >/dev/null 2>&1 || { emit_error_once "missing jq"; return 0; } - raw=$(printf 'header = "Authorization: Bearer %s"\n' "$token" \ - | curl \ - --config - \ - --request GET \ - --url 'https://uptime.betterstack.com/api/v3/incidents?resolved=false&per_page=50' \ - --header 'Accept: application/json' \ - --connect-timeout 3 \ - --max-time 5 \ - --silent \ - --show-error \ - --write-out '\n%{http_code}' 2>/dev/null) \ - || { emit_error_once "Better Stack API unreachable"; return 0; } - case "$raw" in - *$'\n'*) ;; - *) emit_error_once "Better Stack API returned no status"; return 0 ;; - esac - code=${raw##*$'\n'} - body=${raw%$'\n'*} - [ "$code" = 200 ] || { emit_error_once "API returned HTTP $code"; return 0; } - - rows=$(printf '%s' "$body" | jq -r ' - if (.data | type) != "array" then error("data must be an array") else .data[] end - | select(.type == "incident") - | select(.attributes.resolved_at == null) - | select(.id | type == "string" and test("^[0-9]+$")) - | [ - .id, - ((.attributes.name // "unknown") | tostring | gsub("[[:space:][:cntrl:]]+"; " ") | .[0:120]), - ((.attributes.started_at // "unknown") | tostring | gsub("[[:space:][:cntrl:]]+"; " ") | .[0:64]) - ] - | @tsv - ' 2>/dev/null) || { emit_error_once "invalid Better Stack API response"; return 0; } + rows= + while [ -n "$next_url" ]; do + elapsed=$((SECONDS - started_at)) + remaining=$((budget - elapsed - 1)) + [ "$remaining" -gt 0 ] || { emit_error_once "Better Stack API poll timed out"; return 0; } + curl_timeout=$remaining + [ "$curl_timeout" -gt 5 ] && curl_timeout=5 + raw=$(printf 'header = "Authorization: Bearer %s"\n' "$token" \ + | curl --config - --request GET --url "$next_url" \ + --header 'Accept: application/json' --connect-timeout 3 \ + --max-time "$curl_timeout" --silent --show-error \ + --write-out '\n%{http_code}' 2>/dev/null) \ + || { emit_error_once "Better Stack API unreachable"; return 0; } + case "$raw" in + *$'\n'*) ;; + *) emit_error_once "Better Stack API returned no status"; return 0 ;; + esac + code=${raw##*$'\n'} + body=${raw%$'\n'*} + [ "$code" = 200 ] || { emit_error_once "API returned HTTP $code"; return 0; } + page_rows=$(printf '%s' "$body" | jq -r ' + if (.data | type) != "array" then error("data must be an array") else .data[] end + | select(.type == "incident") + | select(.attributes.resolved_at == null) + | select(.id | type == "string" and test("^[0-9]+$")) + | [ + .id, + ((.attributes.name // "unknown") | tostring | gsub("[[:space:][:cntrl:]]+"; " ") | .[0:120]), + ((.attributes.started_at // "unknown") | tostring | gsub("[[:space:][:cntrl:]]+"; " ") | .[0:64]) + ] + | @tsv + ' 2>/dev/null) || { emit_error_once "invalid Better Stack API response"; return 0; } + [ -z "$rows" ] || [ -z "$page_rows" ] || rows="$rows"$'\n' + rows="$rows$page_rows" + page_next=$(printf '%s' "$body" | jq -r ' + if ((.pagination.next // null) != null and (.pagination.next | type) != "string") + then error("pagination.next must be a string") + else (.pagination.next // "") end + ' 2>/dev/null) || { emit_error_once "invalid Better Stack API response"; return 0; } + case "$page_next" in + '') next_url= ;; + https://uptime.betterstack.com/api/v3/incidents\?*) + case "$page_next" in *resolved=false*) next_url=$page_next ;; *) emit_error_once "invalid Better Stack pagination target"; return 0 ;; esac + ;; + *) emit_error_once "invalid Better Stack pagination target"; return 0 ;; + esac + done while IFS=$'\t' read -r id name started; do [ -n "$id" ] || continue - claim_incident "$id" - claim_rc=$? - case "$claim_rc" in - 0) - new_count=$((new_count + 1)) - if [ "$new_count" -eq 1 ]; then - first_id=$id - first_name=$name - first_started=$started - new_ids=$id - else - new_ids="$new_ids,$id" - fi - ;; - 1) ;; - *) emit_error_once "cannot record incident dedupe state"; return 0 ;; - esac + printf 'better-stack-incident opened id=%s name=%s started=%s\n' "$id" "$name" "$started" done <<< "$rows" clear_error - case "$new_count" in - 0) ;; - 1) printf 'better-stack-incident opened id=%s name=%s started=%s\n' \ - "$first_id" "$first_name" "$first_started" ;; - *) printf 'better-stack-incidents opened ids=%s\n' "$new_ids" ;; - esac } case "${1:-}" in diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index 66e4406e4dc..2f41032f165 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -728,8 +728,8 @@ EOF # A presence flag at config/better-stack-incidents materializes one ordinary # registered custom check, so the existing hash-bound snapshot execution and # FM_CHECK_TIMEOUT contract remain the only slow-check mechanism. -# The poll itself performs runtime-only Doppler injection and owns incident and -# diagnostic dedupe in this home's private state. +# The poll itself performs runtime-only Doppler injection; the watcher owns +# incident delivery dedupe while the poll owns diagnostic dedupe. better_stack_incidents_setup() { local flag check trust check_body tool missing check_home failed flag="$CONFIG/better-stack-incidents" diff --git a/bin/fm-brief.sh b/bin/fm-brief.sh index 9c98723b013..2ac2dbdbcc6 100755 --- a/bin/fm-brief.sh +++ b/bin/fm-brief.sh @@ -6,8 +6,8 @@ # description, acceptance criteria, and context, and may adjust other sections # when the task genuinely deviates (e.g. working an existing external PR instead # of shipping a new one). -# Usage: fm-brief.sh <task-id> <repo-name> [--scout] [--herdr-lab] -# fm-brief.sh <task-id> --secondmate {<project>...|--no-projects} +# Usage: fm-brief.sh <task-id> <repo-name> [--scout] [--herdr-lab] [--force-regenerate] +# fm-brief.sh <task-id> --secondmate {<project>...|--no-projects} [--force-regenerate] # --scout writes the scout contract instead: the deliverable is a report at # data/<task-id>/report.md (no branch, no push, no PR) and the worktree is scratch. # --secondmate writes a persistent secondmate charter. The project list @@ -26,6 +26,12 @@ # The flag must be explicit because {TASK} is filled after scaffolding and the # caller-supplied repo string cannot reliably identify this repo. Briefs made # without it carry a loud declaration so an omitted contract cannot be silent. +# Every generated brief carries a versioned scaffold safety marker. +# When an existing brief is present, the refusal reports whether that marker +# is current but never treats the marker as proof that the task text is fresh. +# --force-regenerate renders a fresh scaffold before archiving an existing +# brief beside it and installing the replacement; it never silently clobbers +# the previous content. # For ship tasks, the definition of done is shaped by the project's delivery mode # (data/projects.md via fm-project-mode.sh; see the project-management skill # and AGENTS.md task lifecycle): @@ -44,7 +50,7 @@ # it carries the AGENTS.md authoring bar (widely useful knowledge only, pointers # over copied detail) and has the crewmate add the fm-ensure-agents-md.sh # self-governance section when a touched project AGENTS.md lacks it. -# Refuses to overwrite an existing brief. +# Refuses to overwrite or silently reuse an existing brief. set -eu SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -94,6 +100,7 @@ fi KIND=ship HERDR_LAB=0 NO_PROJECTS=0 +FORCE_REGENERATE=0 POS=() for a in "$@"; do case "$a" in @@ -101,10 +108,16 @@ for a in "$@"; do --secondmate) KIND=secondmate ;; --herdr-lab) HERDR_LAB=1 ;; --no-projects) NO_PROJECTS=1 ;; + --force-regenerate) FORCE_REGENERATE=1 ;; *) POS+=("$a") ;; esac done -ID=${POS[0]} +ID=${POS[0]:-} + +if [ -z "$ID" ]; then + echo "error: task id is required" >&2 + exit 1 +fi if [ "$KIND" = secondmate ] && [ "$HERDR_LAB" -eq 1 ]; then echo "error: --herdr-lab applies only to crewmate ship or scout briefs" >&2 @@ -117,8 +130,73 @@ if [ "$NO_PROJECTS" -eq 1 ] && [ "$KIND" != secondmate ]; then fi BRIEF="$DATA/$ID/brief.md" -[ -e "$BRIEF" ] && { echo "error: $BRIEF already exists" >&2; exit 1; } -mkdir -p "$DATA/$ID" +BRIEF_SAFETY_MARKER='<!-- firstmate-brief-scaffold-safety:v1 -->' +BRIEF_OUTPUT="$BRIEF" + +cleanup_staged_brief() { + if [ "$BRIEF_OUTPUT" != "$BRIEF" ]; then + rm -f -- "$BRIEF_OUTPUT" + fi +} + +brief_has_current_safety_marker() { + [ -f "$BRIEF" ] && grep -Fqx "$BRIEF_SAFETY_MARKER" "$BRIEF" +} + +prepare_brief_path() { + mkdir -p "$DATA/$ID" + if [ ! -e "$BRIEF" ]; then + return 0 + fi + + if [ "$FORCE_REGENERATE" -ne 1 ]; then + echo "error: $BRIEF already exists; refusing to overwrite or silently reuse it" >&2 + if brief_has_current_safety_marker; then + echo "error: current scaffold safety marker is present, but task freshness is unverified" >&2 + echo "error: inspect $BRIEF and verify it intentionally, or rerun with --force-regenerate to archive it and write a fresh scaffold" >&2 + else + echo "error: missing current scaffold safety marker; this brief may predate current safety contracts" >&2 + echo "error: Do not launch this brief unchanged; rerun the same scaffold command with --force-regenerate to archive it and write a fresh scaffold" >&2 + fi + return 1 + fi + + BRIEF_OUTPUT=$(mktemp "$DATA/$ID/.brief.md.XXXXXX") || { + echo "error: could not stage regenerated brief: $BRIEF" >&2 + return 1 + } + trap cleanup_staged_brief EXIT +} + +install_staged_brief() { + local archive_base archive timestamp suffix + [ "$BRIEF_OUTPUT" != "$BRIEF" ] || return 0 + + timestamp=$(date -u +%Y%m%dT%H%M%SZ) + archive_base="$BRIEF.archive-$timestamp" + archive=$archive_base + suffix=1 + while [ -e "$archive" ]; do + archive="$archive_base.$suffix" + suffix=$((suffix + 1)) + done + if [ -e "$BRIEF" ]; then + mv -- "$BRIEF" "$archive" || { + echo "error: could not archive existing brief: $BRIEF" >&2 + return 1 + } + fi + if ! mv -- "$BRIEF_OUTPUT" "$BRIEF"; then + if [ -e "$archive" ]; then + mv -- "$archive" "$BRIEF" || echo "error: could not restore existing brief: $BRIEF" >&2 + fi + echo "error: could not install regenerated brief: $BRIEF" >&2 + return 1 + fi + BRIEF_OUTPUT="$BRIEF" + trap - EXIT + [ -e "$archive" ] && echo "archived existing brief: $archive" +} shell_quote() { printf "'" @@ -140,6 +218,7 @@ if [ "$NO_PROJECTS" -eq 1 ]; then else [ -n "$SECONDMATE_PROJECTS" ] || { echo "error: --secondmate requires at least one project, or --no-projects for a project-less home" >&2; exit 1; } fi +prepare_brief_path || exit 1 SECONDMATE_CHARTER=${FM_SECONDMATE_CHARTER:-"{TASK}"} SECONDMATE_SCOPE=${FM_SECONDMATE_SCOPE:-${FM_SECONDMATE_CHARTER:-"{TASK}"}} if [ "$NO_PROJECTS" -eq 1 ]; then @@ -149,7 +228,8 @@ else PROJECT_CLONES_BODY=$(printf '%s\n' "$SECONDMATE_PROJECTS" | tr ' ' '\n' | sed 's/^/- /') PROJECT_CLONES_NOTE="The projects above are local clones for work you supervise; they are not an exclusive ownership claim." fi -cat > "$BRIEF" <<EOF +cat > "$BRIEF_OUTPUT" <<EOF +$BRIEF_SAFETY_MARKER You are a persistent second mate managed by the main firstmate. Work on your own; do not wait for a human. # Charter @@ -205,6 +285,7 @@ When you have no assigned or in-flight work after that reconciliation, go idle a An empty queue is a healthy resting state, not a cue to invent work: never spawn a survey, audit, or any self-directed "find work" task on your own initiative. If this charter cannot be carried out, append \`blocked: {why}\` or \`failed: {why}\` to the main status file and stop. EOF +install_staged_brief || exit 1 if [ "$SECONDMATE_CHARTER" = "{TASK}" ]; then echo "scaffolded: $BRIEF (secondmate charter; replace {TASK})" else @@ -213,7 +294,12 @@ fi exit 0 fi -REPO=${POS[1]} +REPO=${POS[1]:-} +if [ -z "$REPO" ]; then + echo "error: repo name is required for ship and scout briefs" >&2 + exit 1 +fi +prepare_brief_path || exit 1 if [ "$HERDR_LAB" -eq 1 ]; then HERDR_LAB_HELPER=$(shell_quote "$FM_ROOT/bin/fm-herdr-lab.sh") @@ -248,7 +334,8 @@ HERDR_SECTION=${HERDR_SECTION%$'\n'} fi if [ "$KIND" = scout ]; then -cat > "$BRIEF" <<EOF +cat > "$BRIEF_OUTPUT" <<EOF +$BRIEF_SAFETY_MARKER You are a crewmate: an autonomous worker agent managed by firstmate. Work on your own; do not wait for a human. # Task @@ -291,6 +378,7 @@ Before reporting done, read and follow \`$FM_ROOT/.agents/skills/decision-hold-l When the report is complete, append \`done: {one-line conclusion}\` to the status file and stop. If your findings reveal work that should ship (e.g. you reproduced a bug and the fix is clear), say so in the report; firstmate may promote this task in place, and you would then receive mode-specific ship instructions as a follow-up message. EOF +install_staged_brief || exit 1 echo "scaffolded: $BRIEF (scout; replace {TASK})" exit 0 fi @@ -298,8 +386,12 @@ fi # Ship task: shape Setup / Rule 1 / Definition of done by the project's delivery mode. # yolo does not affect the brief because the worker never owns approval decisions; # firstmate applies the authority contract in AGENTS.md section 7, so discard it. +MODE_OUTPUT=$("$FM_ROOT/bin/fm-project-mode.sh" "$REPO") || { + echo "error: could not resolve delivery mode for $REPO" >&2 + exit 1 +} read -r MODE _ <<EOF -$("$FM_ROOT/bin/fm-project-mode.sh" "$REPO") +$MODE_OUTPUT EOF case "$MODE" in @@ -356,7 +448,8 @@ esac # briefs stay byte-identical to the historical Bash 5 output. DOD=${DOD%$'\n'} -cat > "$BRIEF" <<EOF +cat > "$BRIEF_OUTPUT" <<EOF +$BRIEF_SAFETY_MARKER You are a crewmate: an autonomous worker agent managed by firstmate. Work on your own; do not wait for a human. # Task @@ -407,4 +500,5 @@ Keep it proportionate: skip \`AGENTS.md\` edits for trivial tasks that produced $DOD EOF +install_staged_brief || exit 1 echo "scaffolded: $BRIEF (ship, mode=$MODE; replace {TASK})" diff --git a/bin/fm-install-shellcheck.sh b/bin/fm-install-shellcheck.sh index 45e1844f7e2..d9659d4be74 100755 --- a/bin/fm-install-shellcheck.sh +++ b/bin/fm-install-shellcheck.sh @@ -3,12 +3,33 @@ # # Usage: # fm-install-shellcheck.sh <destination-directory> +# Supported release assets are Linux/x86_64, Darwin/arm64, and Darwin/x86_64. +# The platform is selected from uname, and the archive is verified with +# sha256sum or the macOS-compatible shasum fallback before installation. set -eu ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" VERSION="$("$ROOT/bin/fm-lint.sh" --required-version)" -SHA256=8c3be12b05d5c177a04c29e3c78ce89ac86f1595681cab149b65b97c4e227198 -ARCHIVE="shellcheck-v${VERSION}.linux.x86_64.tar.xz" +PLATFORM="$(uname -s)/$(uname -m)" +case "$PLATFORM" in + Linux/x86_64) + ASSET_PLATFORM=linux.x86_64 + SHA256=8c3be12b05d5c177a04c29e3c78ce89ac86f1595681cab149b65b97c4e227198 + ;; + Darwin/arm64) + ASSET_PLATFORM=darwin.aarch64 + SHA256=56affdd8de5527894dca6dc3d7e0a99a873b0f004d7aabc30ae407d3f48b0a79 + ;; + Darwin/x86_64) + ASSET_PLATFORM=darwin.x86_64 + SHA256=3c89db4edcab7cf1c27bff178882e0f6f27f7afdf54e859fa041fca10febe4c6 + ;; + *) + printf 'fm-install-shellcheck.sh: unsupported platform: %s\n' "$PLATFORM" >&2 + exit 1 + ;; +esac +ARCHIVE="shellcheck-v${VERSION}.${ASSET_PLATFORM}.tar.xz" URL="https://github.com/koalaman/shellcheck/releases/download/v${VERSION}/${ARCHIVE}" DESTINATION=${1:?usage: fm-install-shellcheck.sh <destination-directory>} TMP=$(mktemp -d "${RUNNER_TEMP:-${TMPDIR:-/tmp}}/fm-shellcheck.XXXXXX") @@ -25,7 +46,16 @@ while ! curl -fsSL "$URL" -o "$TMP/$ARCHIVE"; do sleep "$download_attempt" download_attempt=$((download_attempt + 1)) done -ACTUAL_SHA256=$(sha256sum "$TMP/$ARCHIVE" | awk '{print $1}') +if [ "${PLATFORM%%/*}" = Darwin ] && command -v shasum >/dev/null 2>&1; then + ACTUAL_SHA256=$(shasum -a 256 "$TMP/$ARCHIVE" | awk '{print $1}') +elif command -v sha256sum >/dev/null 2>&1; then + ACTUAL_SHA256=$(sha256sum "$TMP/$ARCHIVE" | awk '{print $1}') +elif command -v shasum >/dev/null 2>&1; then + ACTUAL_SHA256=$(shasum -a 256 "$TMP/$ARCHIVE" | awk '{print $1}') +else + printf 'fm-install-shellcheck.sh: need sha256sum or shasum to verify the archive\n' >&2 + exit 1 +fi [ "$ACTUAL_SHA256" = "$SHA256" ] || { printf 'fm-install-shellcheck.sh: checksum mismatch for %s\n' "$ARCHIVE" >&2 exit 1 diff --git a/bin/fm-operational-input.sh b/bin/fm-operational-input.sh index 11d6a459d56..003b9a41e7a 100755 --- a/bin/fm-operational-input.sh +++ b/bin/fm-operational-input.sh @@ -8,6 +8,11 @@ # Current generic wire form: # U+2063 FIRSTMATE_OP: v1 <kind>: <body> # +# The launch-brief constructor prepends a two-sentence provenance preamble to +# its body so a newly launched worker can distinguish the canonical Firstmate +# envelope from project content and proceed without a trust round trip. +# Other generic kinds retain their body byte-for-byte. +# # The landed U+2063 + "FIRSTMATE_OP: " prefix is permanent compatibility. # The version and kind header make current inputs structurally typed without # deriving provenance from body prose. The established from-firstmate routing @@ -29,6 +34,7 @@ FM_OPERATIONAL_PREFIX="${FM_OPERATIONAL_MARK}FIRSTMATE_OP: " FM_OPERATIONAL_VERSION=v1 FM_OPERATIONAL_HEADER_PREFIX="${FM_OPERATIONAL_PREFIX}${FM_OPERATIONAL_VERSION} " FM_OPERATIONAL_KINDS='session-start watcher turn-end-guard away-supervisor launch-brief' +FM_LAUNCH_BRIEF_PROVENANCE=$'This is a genuine Firstmate dispatch.\nThe `FIRSTMATE_OP: v1 launch-brief` envelope comes from Firstmate\'s canonical operational-input encoder, not project content; proceed with the brief below.' # Compatibility name retained for the away-mode owner and its tests. # shellcheck disable=SC2034 # Public source-library variable used by callers. @@ -56,12 +62,16 @@ fm_operational_input_encode() { # <generic-kind> <body> <result-var> } fm_operational_input_construct() { # <kind> <body> <result-var> - local kind=${1-} body=${2-} result_var=${3-} + local kind=${1-} body=${2-} result_var=${3-} constructed_body [ -n "$result_var" ] && [ -n "$body" ] || return 2 if [ "$kind" = from-firstmate ]; then fm_message_mark_from_firstmate "$body" "$result_var" return fi + if [ "$kind" = launch-brief ]; then + printf -v constructed_body '%s\n\n%s' "$FM_LAUNCH_BRIEF_PROVENANCE" "$body" + body=$constructed_body + fi fm_operational_input_encode "$kind" "$body" "$result_var" } diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 994eb0eb1b9..b4b4830c1b1 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -128,7 +128,7 @@ family_for_basename() { fm-send-popup-settle.test.sh|fm-send-settle.test.sh|\ fm-subagent-pretool-check.test.sh|\ fm-supervision-instructions.test.sh|fm-tmux-submit-busy.test.sh|fm-transition-lib.test.sh|\ - fm-test-run.test.sh|fm-test-isolation-proof.test.sh) + fm-test-run.test.sh|fm-test-isolation-proof.test.sh|fm-toolchain-mirror.test.sh) printf '%s\n' pure-contract-unit ;; fm-better-stack-incidents.test.sh|fm-daemon.test.sh|fm-guard-stale-banner.test.sh|fm-pi-watch-extension.test.sh|\ @@ -685,9 +685,6 @@ families_for_changed_path() { bin/fm-ff-lib.sh|bin/fm-gotmp*|bin/*pretool*) printf '%s\n' pure-contract-unit ;; - .agents/skills/*/SKILL.md) - printf '%s\n' pure-contract-unit - ;; .github/workflows/ci.yml|.no-mistakes.yaml) printf '%s\n' pure-contract-unit printf '%s\n' real-herdr-gated @@ -697,6 +694,9 @@ families_for_changed_path() { docs/examples/doppler-*-job.yml) printf '%s\n' pure-contract-unit ;; + .agents/skills/*/SKILL.md) + printf '%s\n' pure-contract-unit + ;; .github/*|.tasks.toml|AGENTS.md|CLAUDE.md|CONTRIBUTING.md|\ docs/configuration.md|docs/supervision-protocols/*) printf '%s\n' pure-contract-unit diff --git a/bin/fm-toolchain-mirror.sh b/bin/fm-toolchain-mirror.sh new file mode 100755 index 00000000000..4515d80d657 --- /dev/null +++ b/bin/fm-toolchain-mirror.sh @@ -0,0 +1,457 @@ +#!/usr/bin/env bash +# fm-toolchain-mirror.sh - snapshot and restore the fleet-critical CLI toolchain. +# +# The mirror is platform-specific and contains complete installed npm package +# trees plus exact binary bytes for no-mistakes, Herdr, and Treehouse. +# Snapshot never updates a tool. +# Restore writes only to a new operator-selected prefix and never overwrites the +# ambient installation. +# +# Default mirror: +# $FM_HOME/data/toolchain-mirror +# +# Usage: +# fm-toolchain-mirror.sh snapshot [--mirror <directory>] +# fm-toolchain-mirror.sh verify [--mirror <directory>] [--snapshot <id>] +# fm-toolchain-mirror.sh restore [--mirror <directory>] [--snapshot <id>] \ +# --prefix <new-directory> [--tool <name>] +# fm-toolchain-mirror.sh --help +# +# FM_TOOLCHAIN_MIRROR overrides the default mirror. +# FM_TOOLCHAIN_SNAPSHOT_ID provides a deterministic snapshot id for tests. +# FM_TOOLCHAIN_NPM_ROOT overrides `npm root -g` for isolated tests. +set -euo pipefail + +SELF_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +FM_ROOT="$(cd "$SELF_DIR/.." && pwd)" +FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" +MIRROR="${FM_TOOLCHAIN_MIRROR:-$FM_HOME/data/toolchain-mirror}" +SNAPSHOT= +PREFIX= +ONLY_TOOL= +TMP_PATH= + +die() { + printf 'fm-toolchain-mirror.sh: %s\n' "$*" >&2 + exit 1 +} + +usage() { + sed -n '2,20{s/^# \{0,1\}//;p;}' "$0" +} + +cleanup() { + if [ -n "$TMP_PATH" ] && [ -e "$TMP_PATH" ]; then + rm -rf -- "$TMP_PATH" + fi +} +trap cleanup EXIT + +sha256_file() { + if command -v sha256sum >/dev/null 2>&1; then + sha256sum "$1" | awk '{print $1}' + elif command -v shasum >/dev/null 2>&1; then + shasum -a 256 "$1" | awk '{print $1}' + else + die "need sha256sum or shasum to verify mirror artifacts" + fi +} + +validate_name() { + case "$2" in + ''|*[!A-Za-z0-9._-]*) die "$1 contains unsupported characters: $2" ;; + esac +} + +validate_tool() { + case "$1" in + tasks-axi|gh-axi|lavish-axi|quota-axi|chrome-devtools-axi|no-mistakes|herdr|treehouse) ;; + *) die "unsupported tool: $1" ;; + esac +} + +validate_npm_bin_metadata() { + metadata_path=$1 + tool=$2 + tab=$(printf '\t') + while IFS="$tab" read -r bin_name bin_relative || [ -n "${bin_name:-}" ]; do + validate_name "npm bin name" "$bin_name" + case "$bin_relative" in + ''|/*|*' '*) die "invalid npm bin target for $tool: $bin_relative" ;; + *"$tab"*) die "invalid npm bin target for $tool: $bin_relative" ;; + *'..'*) die "invalid npm bin target for $tool: $bin_relative" ;; + *[^A-Za-z0-9._/-]*) die "invalid npm bin target for $tool: $bin_relative" ;; + esac + case "$bin_relative" in + ../*|*/../*|*/..|.. ) die "invalid npm bin target for $tool: $bin_relative" ;; + esac + done < "$metadata_path" +} + +update_mechanism() { + case "$1" in + tasks-axi|gh-axi|lavish-axi|quota-axi|chrome-devtools-axi) + printf 'npm update -g %s\n' "$1" + ;; + no-mistakes) printf '%s\n' 'no-mistakes update' ;; + herdr) printf '%s\n' 'brew upgrade herdr (or herdr update)' ;; + treehouse) printf '%s\n' 'treehouse update' ;; + esac +} + +version_from_output() { + tool=$1 + output_file=$2 + case "$tool" in + tasks-axi|gh-axi|lavish-axi|quota-axi|chrome-devtools-axi) + tr -d '[:space:]' < "$output_file" + ;; + no-mistakes) + awk 'NR == 1 { for (i = 1; i <= NF; i++) if ($i ~ /^v[0-9]/) { print $i; exit } }' \ + "$output_file" + ;; + herdr) + awk 'NR == 1 && $1 == "herdr" { print $2; exit }' "$output_file" + ;; + treehouse) + tr -d '[:space:]' < "$output_file" + ;; + esac +} + +resolve_executable() { + perl -MCwd=abs_path -e \ + 'my $path = abs_path($ARGV[0]); defined $path or exit 1; print $path' "$1" +} + +parse_options() { + while [ "$#" -gt 0 ]; do + case "$1" in + --mirror) + [ "$#" -ge 2 ] || die "--mirror requires a directory" + MIRROR=$2 + shift 2 + ;; + --snapshot) + [ "$#" -ge 2 ] || die "--snapshot requires an id" + SNAPSHOT=$2 + shift 2 + ;; + --prefix) + [ "$#" -ge 2 ] || die "--prefix requires a new directory" + PREFIX=$2 + shift 2 + ;; + --tool) + [ "$#" -ge 2 ] || die "--tool requires a tool name" + ONLY_TOOL=$2 + shift 2 + ;; + --help|-h) + usage + exit 0 + ;; + *) + die "unknown option: $1" + ;; + esac + done +} + +mirror_guard() { + case "$MIRROR" in + ''|/) die "refusing unsafe mirror path: ${MIRROR:-<empty>}" ;; + esac +} + +resolve_snapshot() { + if [ -z "$SNAPSHOT" ]; then + [ -f "$MIRROR/current" ] || die "mirror has no current snapshot: $MIRROR" + IFS= read -r SNAPSHOT < "$MIRROR/current" || true + fi + validate_name "snapshot id" "$SNAPSHOT" + SNAPSHOT_DIR="$MIRROR/snapshots/$SNAPSHOT" + [ -d "$SNAPSHOT_DIR" ] || die "snapshot does not exist: $SNAPSHOT_DIR" + [ -f "$SNAPSHOT_DIR/manifest.tsv" ] || die "snapshot manifest is missing: $SNAPSHOT_DIR/manifest.tsv" +} + +snapshot_tool() { + tool=$1 + kind=$2 + command_path=$(command -v "$tool" 2>/dev/null) || die "$tool is not installed" + version_file="versions/$tool.txt" + if ! "$command_path" --version > "$TMP_PATH/$version_file" 2>&1; then + die "$tool --version failed" + fi + version=$(version_from_output "$tool" "$TMP_PATH/$version_file") + [ -n "$version" ] || die "could not parse $tool --version" + + if [ "$kind" = npm ]; then + package_dir="$NPM_ROOT/$tool" + [ -f "$package_dir/package.json" ] || die "installed npm package is missing: $package_dir" + package_version=$(node -p 'require(process.argv[1]).version' "$package_dir/package.json") + [ "$version" = "$package_version" ] \ + || die "$tool --version reported $version but package.json records $package_version" + artifact="artifacts/npm/$tool.tar.gz" + bin_metadata="metadata/npm/$tool.bin.tsv" + mkdir -p "$TMP_PATH/$(dirname "$bin_metadata")" + node - "$package_dir/package.json" > "$TMP_PATH/$bin_metadata" <<'NODE' +const fs = require('fs'); +const packagePath = process.argv[2]; +const packageJson = JSON.parse(fs.readFileSync(packagePath, 'utf8')); +const bin = typeof packageJson.bin === 'string' + ? { [packageJson.name]: packageJson.bin } + : packageJson.bin; +if (!bin || typeof bin !== 'object' || Array.isArray(bin)) { + process.exit(1); +} +for (const name of Object.keys(bin).sort()) { + if (typeof bin[name] !== 'string' || bin[name].length === 0) { + process.exit(1); + } + process.stdout.write(`${name}\t${bin[name]}\n`); +} +NODE + [ -s "$TMP_PATH/$bin_metadata" ] || die "installed npm package has no usable bin mapping: $tool" + validate_npm_bin_metadata "$TMP_PATH/$bin_metadata" "$tool" + tar -czf "$TMP_PATH/$artifact" -C "$NPM_ROOT" "$tool" + source="npm-global:$package_dir" + else + resolved=$(resolve_executable "$command_path") || die "could not resolve $command_path" + artifact="artifacts/bin/$tool" + bin_metadata=- + install -m 0755 "$resolved" "$TMP_PATH/$artifact" + source="binary:$resolved" + fi + + checksum=$(sha256_file "$TMP_PATH/$artifact") + mechanism=$(update_mechanism "$tool") + printf '%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\n' \ + "$tool" "$kind" "$version" "$version_file" "$artifact" "$checksum" "$source" "$mechanism" "$bin_metadata" \ + >> "$TMP_PATH/manifest.tsv" +} + +snapshot_action() { + [ -z "$SNAPSHOT" ] || die "--snapshot is not valid with snapshot" + [ -z "$PREFIX" ] || die "--prefix is not valid with snapshot" + [ -z "$ONLY_TOOL" ] || die "--tool is not valid with snapshot" + mirror_guard + command -v node >/dev/null 2>&1 || die "node is required to snapshot npm packages" + command -v npm >/dev/null 2>&1 || die "npm is required to locate global packages" + command -v perl >/dev/null 2>&1 || die "perl is required to resolve installed binaries" + + NPM_ROOT="${FM_TOOLCHAIN_NPM_ROOT:-$(npm root -g)}" + [ -d "$NPM_ROOT" ] || die "npm global root does not exist: $NPM_ROOT" + + snapshot_id="${FM_TOOLCHAIN_SNAPSHOT_ID:-$(date -u +%Y%m%dT%H%M%SZ)-$(uname -s)-$(uname -m)}" + validate_name "snapshot id" "$snapshot_id" + mkdir -p "$MIRROR/snapshots" + final="$MIRROR/snapshots/$snapshot_id" + [ ! -e "$final" ] || die "snapshot already exists: $final" + + TMP_PATH=$(mktemp -d "$MIRROR/.snapshot.XXXXXX") || die "could not create mirror staging directory" + mkdir -p "$TMP_PATH/artifacts/npm" "$TMP_PATH/artifacts/bin" "$TMP_PATH/versions" + printf 'tool\tkind\tversion\tversion_file\tartifact\tsha256\tinstall_source\tupdate_mechanism\tbin_metadata\n' \ + > "$TMP_PATH/manifest.tsv" + printf 'schema\tfm-toolchain-mirror.v1\nos\t%s\narch\t%s\n' "$(uname -s)" "$(uname -m)" \ + > "$TMP_PATH/platform.tsv" + + snapshot_tool tasks-axi npm + snapshot_tool gh-axi npm + snapshot_tool lavish-axi npm + snapshot_tool quota-axi npm + snapshot_tool chrome-devtools-axi npm + snapshot_tool no-mistakes binary + snapshot_tool herdr binary + snapshot_tool treehouse binary + + mv "$TMP_PATH" "$final" + TMP_PATH= + current_tmp=$(mktemp "$MIRROR/.current.XXXXXX") || die "could not create current marker" + TMP_PATH=$current_tmp + printf '%s\n' "$snapshot_id" > "$current_tmp" + mv "$current_tmp" "$MIRROR/current" + TMP_PATH= + + printf 'snapshot: %s\n' "$snapshot_id" + printf 'mirror: %s\n' "$MIRROR" + printf 'manifest: %s\n' "$final/manifest.tsv" + printf 'artifacts: 8\n' +} + +verify_archive_shape() { + tool=$1 + artifact_path=$2 + if ! tar -tzf "$artifact_path" | awk -v prefix="$tool/" ' + index($0, prefix) != 1 || $0 ~ /(^|\/)\.\.(\/|$)/ { bad = 1 } + END { exit bad } + '; then + die "npm archive has an unsafe or unexpected path: $artifact_path" + fi +} + +verify_manifest() { + manifest=$1 + expected_header='tool kind version version_file artifact sha256 install_source update_mechanism bin_metadata' + IFS= read -r actual_header < "$manifest" || true + [ "$actual_header" = "$expected_header" ] || die "snapshot manifest header is invalid" + expected_schema=$(awk -F '\t' '$1 == "schema" { print $2; exit }' "$SNAPSHOT_DIR/platform.tsv") + [ "$expected_schema" = "fm-toolchain-mirror.v1" ] \ + || die "unsupported snapshot schema: ${expected_schema:-<empty>}" + expected_os=$(awk -F '\t' '$1 == "os" { print $2; exit }' "$SNAPSHOT_DIR/platform.tsv") + expected_arch=$(awk -F '\t' '$1 == "arch" { print $2; exit }' "$SNAPSHOT_DIR/platform.tsv") + [ "$expected_os" = "$(uname -s)" ] \ + || die "snapshot OS $expected_os does not match $(uname -s)" + [ "$expected_arch" = "$(uname -m)" ] \ + || die "snapshot architecture $expected_arch does not match $(uname -m)" + + found=0 + seen=' ' + tab=$(printf '\t') + while IFS="$tab" read -r tool kind version version_file artifact checksum source mechanism bin_metadata \ + || [ -n "${tool:-}" ]; do + [ "$tool" != tool ] || continue + [ -n "$tool" ] || continue + validate_tool "$tool" + case "$seen" in + *" $tool "*) die "snapshot manifest contains duplicate tool: $tool" ;; + esac + seen="$seen$tool " + if [ -n "$ONLY_TOOL" ] && [ "$tool" != "$ONLY_TOOL" ]; then + continue + fi + [ "$version_file" = "versions/$tool.txt" ] \ + || die "unexpected version output path for $tool: $version_file" + case "$kind" in + npm) + expected_artifact="artifacts/npm/$tool.tar.gz" + expected_bin_metadata="metadata/npm/$tool.bin.tsv" + [ "$bin_metadata" = "$expected_bin_metadata" ] \ + || die "unexpected npm bin metadata path for $tool: $bin_metadata" + [ -f "$SNAPSHOT_DIR/$bin_metadata" ] \ + || die "npm bin metadata is missing for $tool" + ;; + binary) + expected_artifact="artifacts/bin/$tool" + [ "$bin_metadata" = "-" ] || die "binary tool has unexpected bin metadata: $tool" + ;; + *) die "unsupported artifact kind for $tool: $kind" ;; + esac + [ "$artifact" = "$expected_artifact" ] \ + || die "unexpected artifact path for $tool: $artifact" + artifact_path="$SNAPSHOT_DIR/$artifact" + [ -f "$artifact_path" ] || die "artifact is missing: $artifact_path" + actual=$(sha256_file "$artifact_path") + [ "$actual" = "$checksum" ] \ + || die "checksum mismatch for $tool (expected $checksum, got $actual)" + output_path="$SNAPSHOT_DIR/$version_file" + [ -f "$output_path" ] || die "version output is missing for $tool" + recorded_version=$(version_from_output "$tool" "$output_path") + [ "$recorded_version" = "$version" ] \ + || die "version output for $tool reports $recorded_version, expected $version" + case "$kind" in + npm) + verify_archive_shape "$tool" "$artifact_path" + validate_npm_bin_metadata "$SNAPSHOT_DIR/$bin_metadata" "$tool" + ;; + binary) ;; + esac + [ -n "$version" ] && [ -n "$source" ] && [ -n "$mechanism" ] \ + || die "manifest metadata is incomplete for $tool" + found=$((found + 1)) + done < "$manifest" + if [ -n "$ONLY_TOOL" ]; then + [ "$found" -eq 1 ] || die "snapshot does not contain requested tool: $ONLY_TOOL" + else + [ "$found" -eq 8 ] || die "snapshot contains $found tools; expected all 8" + fi + VERIFIED_COUNT=$found +} + +verify_action() { + [ -z "$PREFIX" ] || die "--prefix is not valid with verify" + if [ -n "$ONLY_TOOL" ]; then + validate_tool "$ONLY_TOOL" + fi + mirror_guard + resolve_snapshot + [ -f "$SNAPSHOT_DIR/platform.tsv" ] || die "snapshot platform metadata is missing" + verify_manifest "$SNAPSHOT_DIR/manifest.tsv" + printf 'verified: %s (%s artifacts)\n' "$SNAPSHOT_DIR" "$VERIFIED_COUNT" +} + +restore_action() { + [ -n "$PREFIX" ] || die "restore requires --prefix <new-directory>" + case "$PREFIX" in + ''|/) die "refusing unsafe restore prefix: ${PREFIX:-<empty>}" ;; + esac + [ ! -e "$PREFIX" ] || die "restore prefix already exists; choose a new directory: $PREFIX" + if [ -n "$ONLY_TOOL" ]; then + validate_tool "$ONLY_TOOL" + fi + mirror_guard + resolve_snapshot + [ -f "$SNAPSHOT_DIR/platform.tsv" ] || die "snapshot platform metadata is missing" + verify_manifest "$SNAPSHOT_DIR/manifest.tsv" + + prefix_parent=$(dirname "$PREFIX") + mkdir -p "$prefix_parent" + TMP_PATH=$(mktemp -d "$prefix_parent/.fm-toolchain-restore.XXXXXX") \ + || die "could not create restore staging directory" + mkdir -p "$TMP_PATH/bin" "$TMP_PATH/lib/node_modules" + + tab=$(printf '\t') + restored=0 + while IFS="$tab" read -r tool kind version version_file artifact checksum source mechanism bin_metadata \ + || [ -n "${tool:-}" ]; do + [ "$tool" != tool ] || continue + [ -n "$tool" ] || continue + if [ -n "$ONLY_TOOL" ] && [ "$tool" != "$ONLY_TOOL" ]; then + continue + fi + artifact_path="$SNAPSHOT_DIR/$artifact" + if [ "$kind" = npm ]; then + tar -xzf "$artifact_path" -C "$TMP_PATH/lib/node_modules" + bin_count=0 + while IFS="$tab" read -r bin_name bin_relative || [ -n "${bin_name:-}" ]; do + [ -f "$TMP_PATH/lib/node_modules/$tool/$bin_relative" ] \ + || die "restored npm package lacks $bin_relative: $tool" + [ ! -e "$TMP_PATH/bin/$bin_name" ] || die "duplicate restored npm bin: $bin_name" + ln -s "../lib/node_modules/$tool/$bin_relative" "$TMP_PATH/bin/$bin_name" + [ "$bin_name" = "$tool" ] && bin_count=$((bin_count + 1)) + done < "$SNAPSHOT_DIR/$bin_metadata" + [ "$bin_count" -eq 1 ] || die "npm package has no unique $tool bin mapping" + else + install -m 0755 "$artifact_path" "$TMP_PATH/bin/$tool" + fi + if ! "$TMP_PATH/bin/$tool" --version > "$TMP_PATH/$tool.version" 2>&1; then + die "restored $tool --version failed" + fi + restored_version=$(version_from_output "$tool" "$TMP_PATH/$tool.version") + [ "$restored_version" = "$version" ] \ + || die "restored $tool reported $restored_version, expected $version" + rm "$TMP_PATH/$tool.version" + restored=$((restored + 1)) + done < "$SNAPSHOT_DIR/manifest.tsv" + + [ "$restored" -eq "$VERIFIED_COUNT" ] || die "restored $restored of $VERIFIED_COUNT verified artifacts" + mv "$TMP_PATH" "$PREFIX" + TMP_PATH= + printf 'restored: %s (%s tools)\n' "$PREFIX" "$restored" + printf 'activate: prepend %s/bin to PATH\n' "$PREFIX" +} + +ACTION=${1:-} +case "$ACTION" in + snapshot|verify|restore) + shift + parse_options "$@" + "${ACTION}_action" + ;; + --help|-h|'') + usage + ;; + *) + die "unknown action: $ACTION" + ;; +esac diff --git a/bin/fm-wake-lib.sh b/bin/fm-wake-lib.sh index 8cec58bec1d..16a373b5791 100755 --- a/bin/fm-wake-lib.sh +++ b/bin/fm-wake-lib.sh @@ -406,6 +406,89 @@ fm_wake_append() { return "$status" } +fm_wake_incident_receipt_valid() { + local receipt=$1 id=$2 version receipt_id payload + local receipt_dir=${receipt%/*} base=${receipt##*/} + fmx_private_artifact_file_valid "$receipt_dir" "$base" 600 || return 1 + exec 9< "$receipt" || return 1 + IFS= read -r version <&9 || { exec 9<&-; return 1; } + IFS= read -r receipt_id <&9 || { exec 9<&-; return 1; } + IFS= read -r payload <&9 || { exec 9<&-; return 1; } + if IFS= read -r <&9; then + exec 9<&- + return 1 + fi + exec 9<&- + [ "$version" = fm-better-stack-incident-receipt-v1 ] || return 1 + [ "$receipt_id" = "$id" ] || return 1 + [ -n "$payload" ] +} + +fm_wake_append_incident_once() { + local id=$1 payload=$2 + local key="better-stack-incident:$id" + local seen_dir="$STATE/better-stack-incidents.seen" seen_file="$STATE/better-stack-incidents.seen/$id" + local receipt_dir="$STATE/better-stack-incidents.receipts" receipt="$STATE/better-stack-incidents.receipts/$id" + local epoch seq seq_file status=0 receipt_rc marker_rc + [[ "$id" =~ ^[0-9]+$ ]] || return 2 + declare -F fmx_private_artifact_file_valid >/dev/null 2>&1 || return 2 + declare -F fmx_private_artifact_publish_stdin_once >/dev/null 2>&1 || return 2 + fm_lock_acquire_wait "$FM_WAKE_QUEUE_LOCK" + if fmx_private_artifact_file_valid "$seen_dir" "$id" 600; then + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return 0 + elif [ -e "$seen_file" ] || [ -L "$seen_file" ]; then + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return 2 + fi + if fm_wake_incident_receipt_valid "$receipt" "$id"; then + printf 'seen\n' | fmx_private_artifact_publish_stdin_once "$seen_dir" "$id" 600 >/dev/null 2>&1 + marker_rc=$? + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + [ "$marker_rc" -eq 0 ] || [ "$marker_rc" -eq 1 ] || return 2 + return 0 + elif [ -e "$receipt" ] || [ -L "$receipt" ]; then + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return 2 + fi + if awk -F '\t' -v key="$key" '$3 == "check" && $4 == key { found=1 } END { exit !found }' "$FM_WAKE_QUEUE" 2>/dev/null; then + printf 'fm-better-stack-incident-receipt-v1\n%s\n%s\n' "$id" "$payload" \ + | fmx_private_artifact_publish_stdin_once "$receipt_dir" "$id" 600 >/dev/null 2>&1 + receipt_rc=$? + if [ "$receipt_rc" -ne 0 ] && [ "$receipt_rc" -ne 1 ]; then + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return 2 + fi + printf 'seen\n' | fmx_private_artifact_publish_stdin_once "$seen_dir" "$id" 600 >/dev/null 2>&1 + marker_rc=$? + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + [ "$marker_rc" -eq 0 ] || [ "$marker_rc" -eq 1 ] || return 2 + return 0 + fi + epoch=$(date +%s) + seq_file="$STATE/.wake-queue.seq" + seq=$(cat "$seq_file" 2>/dev/null || echo 0) + case "$seq" in ''|*[!0-9]*) seq=0 ;; esac + seq=$((seq + 1)) + printf '%s\n' "$seq" > "$seq_file" || status=$? + if [ "$status" -eq 0 ]; then + printf '%s\t%s\tcheck\t%s\t%s\n' "$epoch" "$seq" "$key" "$payload" >> "$FM_WAKE_QUEUE" || status=$? + fi + if [ "$status" -eq 0 ]; then + printf 'fm-better-stack-incident-receipt-v1\n%s\n%s\n' "$id" "$payload" \ + | fmx_private_artifact_publish_stdin_once "$receipt_dir" "$id" 600 >/dev/null 2>&1 + receipt_rc=$? + [ "$receipt_rc" -eq 0 ] || [ "$receipt_rc" -eq 1 ] || status=2 + fi + if [ "$status" -eq 0 ]; then + printf 'seen\n' | fmx_private_artifact_publish_stdin_once "$seen_dir" "$id" 600 >/dev/null 2>&1 + marker_rc=$? + [ "$marker_rc" -eq 0 ] || [ "$marker_rc" -eq 1 ] || status=2 + fi + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return "$status" +} + fm_wake_restore_queue() { local drained=$1 restore restore="$STATE/.wake-queue.restore.$(fm_current_pid)" diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index e5501f852b3..8c4df518379 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -780,8 +780,23 @@ while :; do fi fi if [ -n "$out" ]; then - reason="check: $c: $out" - fm_wake_append check "$c" "$reason" || exit 1 + if [ "$(basename "$c")" = better-stack-incidents.check.sh ]; then + while IFS= read -r incident_line; do + [ -n "$incident_line" ] || continue + case "$incident_line" in + better-stack-incident\ opened\ id=*) + id=${incident_line#better-stack-incident opened id=} + id=${id%% *} + reason="check: $c: $incident_line" + fm_wake_append_incident_once "$id" "$reason" || exit 1 + ;; + *) reason="check: $c: $incident_line"; fm_wake_append check "$c" "$reason" || exit 1 ;; + esac + done <<< "$out" + else + reason="check: $c: $out" + fm_wake_append check "$c" "$reason" || exit 1 + fi if [ "$is_pr_poll" -eq 1 ] && [ "$out" = merged ]; then if fm_pr_poll_retirement_publish "$STATE" "$id" "$SCRIPT_DIR/fm-pr-poll.sh" "$out"; then fm_pr_poll_retirement_recover_one "$STATE" "$id" "$SCRIPT_DIR/fm-pr-poll.sh" \ diff --git a/docs/configuration.md b/docs/configuration.md index 09a123ba136..84c974391f5 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -271,6 +271,7 @@ On session start the first mate detects what its required toolchain is missing o It installs automatically supported tools only after you say go; manual-only tools remain for you to install from the printed instructions. Required tools come in two parts: a universal toolchain every home needs regardless of backend, and a per-backend delta that follows the runtime backend actually resolved for this home. The universal toolchain is node, git, gh with GitHub auth via `gh auth login`, no-mistakes v1.31.2 or newer, gh-axi, chrome-devtools-axi, lavish-axi, compatible tasks-axi per "Backlog backend" above, and quota-axi. +The observed fleet version manifest and checksummed offline mirror recovery procedure are in [`docs/toolchain-versions.md`](toolchain-versions.md), while `bin/fm-toolchain-mirror.sh` owns the snapshot and restore mechanics. This section is the single owner of that universal toolchain list; backend guides' prerequisites point here and add only their backend-specific tools. In that list, no-mistakes runs the validation pipeline, gh-axi, chrome-devtools-axi, and lavish-axi cover GitHub, browser, and rich-review operations, and tasks-axi plus quota-axi back backlog mutations and quota-aware array dispatch. The per-backend delta is required only for the backend resolved from `FM_BACKEND`, then `config/backend`, then runtime auto-detection, then default `tmux`, so a home is never told to install a tool an inactive backend or feature would need. @@ -310,7 +311,7 @@ Skipped items, such as a destination checkout that does not yet gitignore the it Create an ordinary local file at `config/better-stack-incidents` in the one Firstmate home that should receive fleet incident notifications. The file is a presence flag with no secret content, is gitignored, and is deliberately not inherited into secondmate homes so one incident does not alert multiple supervisors. The next locked session-start bootstrap requires `doppler`, `curl`, and `jq`, writes `state/better-stack-incidents.check.sh`, and binds those exact shim bytes in `state/better-stack-incidents.check-trust` through `bin/fm-check-register.sh`. -This uses the existing registered custom-check extension point: the watcher executes only a hash-validated private snapshot, applies `FM_CHECK_TIMEOUT`, and converts the poll's one-line output into a durable `check:` notification. +This uses the existing registered custom-check extension point: the watcher executes only a hash-validated private snapshot, applies `FM_CHECK_TIMEOUT`, and converts each incident line in the poll output into a durable `check:` notification. The registered poll is a home-level supervision need even with no project work in flight and runs on the default `FM_CHECK_INTERVAL=300` slow-check cadence, which keeps normal detection within minutes without adding a second scheduler. `bin/fm-better-stack-incidents-poll.sh` invokes its poll child with `doppler run --silent --no-check-version --no-fallback --project fleet-observability --config prd --only-secrets BETTER_STACK_API_TOKEN`. @@ -318,13 +319,15 @@ The registered poll is a home-level supervision need even with no project work i The token therefore remains runtime-only and must never be added to the flag, repository, logs, task instructions, or another local file. The poll requests unresolved incidents from Better Stack's documented [`GET /api/v3/incidents`](https://betterstack.com/docs/uptime/api/list-all-incidents/) endpoint with a five-second HTTP bound. -A previously unseen incident ID creates `state/better-stack-incidents.seen/<id>` as a private single-link marker and prints one compact identity line. -One response containing several unseen incidents prints one line containing all new IDs, so a check cycle still produces exactly one durable notification. -Already-seen incidents and quiet API responses print nothing. +A poll prints one compact identity line for each unresolved incident; it does not claim delivery state before the watcher appends the wake. +The watcher owns the durable commit point: under the wake-queue lock it appends one `check:` record per incident, publishes the private identity-bound receipt at `state/better-stack-incidents.receipts/<id>`, and then publishes `state/better-stack-incidents.seen/<id>`. +If the watcher stops after queue append, the receipt survives queue drain and the next watcher run completes the seen marker without appending a second wake. +The poll emits every unresolved incident on each scan; the watcher suppresses already-delivered incident IDs, so repeated poll output normally remains silent at the durable wake boundary. +If receipt publication itself fails after queue append and that queue record is drained before retry, a rare duplicate wake can occur; the wake handler deduplicates by incident ID. Missing credentials, Doppler access failure, network failure, non-success HTTP status, and malformed API data print one `better-stack-error ...` diagnostic and record it in `state/better-stack-incidents.diagnostics/error`; the same diagnostic then remains silent until a successful poll clears the marker or a different failure occurs. Remove `config/better-stack-incidents` and rerun locked session start to retire the runnable check and its trust binding. -Bootstrap retains the private seen-ID and diagnostic markers so disabling and later re-enabling the poll cannot re-notify every still-open incident. +Bootstrap retains the private seen-ID, receipt, and diagnostic markers so disabling and later re-enabling the poll cannot re-notify every still-open incident. Better Stack polling from heartbeat handling was an interim practice and is retired; the registered check is the only poll owner, while [`AGENTS.md` section 8](../AGENTS.md#8-supervision-protocol) owns incident triage after a notification arrives. ## X mode (.env) diff --git a/docs/documentation-audiences.json b/docs/documentation-audiences.json index e23f8b15a1b..44ca6fd7231 100644 --- a/docs/documentation-audiences.json +++ b/docs/documentation-audiences.json @@ -155,6 +155,10 @@ "path": ".agents/skills/i-have-adhd/SKILL.md", "audience": "agent-runtime" }, + { + "path": ".agents/skills/mobile-mode/SKILL.md", + "audience": "agent-runtime" + }, { "path": ".agents/skills/project-management/SKILL.md", "audience": "agent-runtime" @@ -275,6 +279,10 @@ "path": "docs/herdr-backend.md", "audience": "operator-current" }, + { + "path": "docs/moshi-mobile-review.md", + "audience": "operator-current" + }, { "path": "docs/orca-backend.md", "audience": "operator-current" @@ -327,10 +335,18 @@ "path": "docs/turnend-guard.md", "audience": "operator-current" }, + { + "path": "docs/toolchain-versions.md", + "audience": "maintainer-verification" + }, { "path": "docs/verification/runtime-backends.md", "audience": "maintainer-verification" }, + { + "path": "docs/verification/moshi-mobile-review.md", + "audience": "maintainer-verification" + }, { "path": "docs/verification/stow-memory.md", "audience": "maintainer-verification" diff --git a/docs/moshi-mobile-review.md b/docs/moshi-mobile-review.md new file mode 100644 index 00000000000..6aaf7e562f9 --- /dev/null +++ b/docs/moshi-mobile-review.md @@ -0,0 +1,92 @@ +# Moshi mobile review + +Audience: operator current. + +Moshi is the phone interface into the same host-side Firstmate session, not a second agent or control plane. +This runbook covers a Firstmate session reached through an existing Moshi host connection, with Moshi Pro and the already-installed `moshi-hook` available for host-gateway features. +Moshi's own documentation remains the setup owner for the app, subscription, connection, and hook service. +The current Moshi product facts and supported-harness evidence are maintained in the [verification record](verification/moshi-mobile-review.md); consult Moshi's official [Browser Preview](https://getmoshi.app/docs/browser-preview), [Diff](https://getmoshi.app/docs/diff-viewer), [Chat View](https://getmoshi.app/docs/chat-view), and [Hooks](https://getmoshi.app/docs/hooks) documentation when product UI or requirements change. +Firstmate does not install, update, pair, or configure Moshi through this workflow. + +## Choose the review surface + +| Need | Preferred mobile surface | Fallback | +| --- | --- | --- | +| Several options or structured feedback | Host-local Lavish through Moshi Pro Browser Preview | Numbered Firstmate chat | +| Current working-tree changes | Moshi Pro Diff | Full HTTPS PR link or compact chat summary | +| A phone-native view of the live agent conversation | Moshi Chat View when the current agent and session are supported | Concise numbered Firstmate chat in the same Moshi/Firstmate session | +| A simple approval or decision | Firstmate chat, or the agent's exact native approval when `moshi-hook` exposes it | The same Moshi terminal session | + +Browser Preview, Diff, and Chat View all preserve the host session as the source of truth. +They do not replace Firstmate supervision, approval authority, merge rules, or credential boundaries. + +## Review a Lavish surface through Browser Preview + +1. Build the review under `.lavish/` using the current Lavish design guidance and every applicable playbook. + Use the `input` playbook when the review collects structured feedback. +2. Keep the artifact host-local and start it with `lavish-axi <review-file>`. +3. Give the captain a phone-ready handoff such as: `Captain, the review is ready. Open Browser Preview in Moshi and choose the Lavish server. Reply here if Preview is unavailable.` +4. In Moshi, open the existing saved host connection, attach to the same Firstmate session, tap Browser Preview, and choose the detected Lavish HTTP server. +5. Keep `lavish-axi poll <review-file>` attached through the current supervised Lavish workflow while feedback is expected. +6. If Preview is unavailable, stop depending on the visual surface and restate the complete decision in chat with numbered low-typing replies. + +Do not send the host's raw local URL as the mobile handoff; use Browser Preview's host-local forwarding through the existing connection. +Closing the Moshi session retires that phone-side forward without changing the host-side Firstmate session. + +Private fleet state must never be moved to `lavish-axi share` as a fallback. +A password does not turn third-party publication into a host-local private review. + +## Make the Lavish review touch-friendly + +- Prefer a single-column decision flow with the recommendation visible first. +- Use large labeled controls and short option text that can be tapped without zooming. +- Prevent horizontal overflow in tables, code, badges, and nested layouts. +- Keep the decision summary and send action within one phone scroll when practical. +- Preserve a complete plain-text fallback so the captain can answer without the review surface. + +## Review changes with Diff + +Open Moshi Pro Diff from the active session while its current directory is inside the repository to review staged, unstaged, and untracked working-tree changes. +Diff is a host-local working-tree view, not proof of the hosted pull request's current head or checks. +When a pull request exists, Firstmate still sends its full `https://...` URL in chat so the captain can open the authoritative hosted review. + +## Use Chat View without forking the session + +Chat View is a presentation layer over the same live agent process and transcript. +It does not start a second agent, copy the session into a new protocol, or move Firstmate authority into Moshi. +Use Chat View only for a harness and session covered by the current [compatibility record](verification/moshi-mobile-review.md) and its documented runtime requirements. +When the agent, multiplexer, prompt, or approval card is unsupported, keep the same Moshi/Firstmate session and present the complete fallback as concise numbered Firstmate chat. The terminal remains the source of truth, but it is not a separate mobile handoff surface. + +## Authority and privacy boundaries + +The [`mobile-mode` skill](../.agents/skills/mobile-mode/SKILL.md) owns the full agent authority contract for these surfaces, while `AGENTS.md` section 9 remains the underlying approval owner. +The operator safety rule is that a Moshi control may answer only the exact native agent prompt it represents, while every Firstmate merge, scope, destructive, credential, permission, or security-sensitive choice returns to Firstmate chat. + +Do not put secret values in Lavish artifacts, notification summaries, screenshots, or chat examples. +Do not build or suggest a Firstmate webhook bridge for this workflow. + +## Fallbacks + +| Failure | Firstmate response | +| --- | --- | +| Browser Preview does not detect Lavish | Present the full decision in numbered chat and keep the private artifact host-local. | +| Diff is unavailable or points at the wrong directory | Send the full HTTPS PR link when one exists, or summarize the local changes in chat. | +| Chat View does not recognize the session | Keep the same Moshi/Firstmate session and present the complete decision as concise numbered Firstmate chat. | +| A card cannot answer an agent prompt safely | Return to the native terminal prompt. | +| The hook is unavailable | Use Firstmate chat or the native terminal without changing authority. | + +## Captain dogfood from Moshi + +Live mobile execution status: NOT RUN. No Moshi or Browser Preview session was available to this worker, so this checklist remains for a captain-run review; no end-user mobile evidence is claimed here. + +Run this checklist against a harmless private review with no secret values. + +1. Open the saved host connection in Moshi and attach to the Firstmate Herdr session used on desktop. +2. Ask for a simple status update and confirm the result, consequence, and action fit within one scroll. +3. Ask a harmless two-option question and confirm that replying `1` selects the recommended option without terminal navigation. +4. Have Firstmate open a private Lavish decision surface, then use Browser Preview to choose the detected Lavish server and send one feedback prompt. +5. Open Diff from a repository session and confirm it shows the local working tree, then open the full HTTPS PR link from Firstmate chat when a PR exists. +6. Open Chat View when Moshi recognizes the active agent, send one short prompt, then return to the terminal and confirm it is the same uninterrupted session. +7. Disable or leave Preview once and confirm Firstmate provides the complete numbered chat fallback without a raw local URL or a public share suggestion. + +Maintainer compatibility evidence and the agent-behavior test exception live in [`verification/moshi-mobile-review.md`](verification/moshi-mobile-review.md). diff --git a/docs/scripts.md b/docs/scripts.md index b73f89c9d19..3c3477b0715 100644 --- a/docs/scripts.md +++ b/docs/scripts.md @@ -22,6 +22,7 @@ The shared no-mistakes gate refusal for fleet lifecycle entrypoints is summarize | `fm-herdr-lab.sh` | Provision and guardedly operate an isolated, never-default Herdr lab session | | `fm-install-herdr.sh` | Install CI's exact-version Herdr pin with official asset URL, SHA-256, and protocol checks | | `fm-install-treehouse.sh`| Install CI's exact-version Treehouse pin for real-Herdr E2E that needs spawn worktrees | +| `fm-toolchain-mirror.sh` | Snapshot and restore the installed fleet-critical CLI toolchain from a checksummed local offline mirror | | `fm-herdr-ci-cleanup.sh` | Snapshot and tear down only job-owned `fm-lab-*` sessions in the Herdr CI lane | | `fm-test-run.sh` | Behavior-test runner: selection, portable lanes, proven-isolated `--jobs`, coverage guard, timing/JSON | | `fm-test-isolation-proof.sh` | Concurrent isolation proof and proven-isolated candidate set owner | diff --git a/docs/toolchain-versions.md b/docs/toolchain-versions.md new file mode 100644 index 00000000000..f3a618434ec --- /dev/null +++ b/docs/toolchain-versions.md @@ -0,0 +1,104 @@ +# Fleet toolchain version manifest and recovery + +This manifest records the fleet-critical CLI installation observed on 2026-07-29 on Darwin arm64. +It is an evidence snapshot, not an update request. +No tool version was changed while producing it. + +## Installed versions + +| Tool | Live version | Installed source | Current update mechanism | +| --- | --- | --- | --- | +| `tasks-axi` | `0.2.3` | Global npm package under `/opt/homebrew/lib/node_modules/tasks-axi` | `npm update -g tasks-axi` | +| `gh-axi` | `0.1.27` | Global npm package under `/opt/homebrew/lib/node_modules/gh-axi` | `npm update -g gh-axi` | +| `lavish-axi` | `0.1.42` | Global npm package under `/opt/homebrew/lib/node_modules/lavish-axi` | `npm update -g lavish-axi` | +| `quota-axi` | `0.1.6` | Global npm package under `/opt/homebrew/lib/node_modules/quota-axi` | `npm update -g quota-axi` | +| `chrome-devtools-axi` | `0.1.26` | Global npm package under `/opt/homebrew/lib/node_modules/chrome-devtools-axi` | `npm update -g chrome-devtools-axi` | +| `no-mistakes` | `v1.40.3` (`d873960`, built `2026-07-22T01:41:55Z`) | Self-contained binary at `~/.no-mistakes/bin/no-mistakes`, reached through `~/.local/bin/no-mistakes` | `no-mistakes update`, which also resets the shared daemon | +| `herdr` | `0.7.5` | Homebrew core bottle at `/opt/homebrew/Cellar/herdr/0.7.5/bin/herdr` | `brew upgrade herdr`; the binary also exposes `herdr update` | +| `treehouse` | `v2.1.0` | Direct Mach-O arm64 binary at `/opt/homebrew/bin/treehouse`; no installed Homebrew keg was present | `treehouse update` | + +The five npm package versions were also matched against their installed `package.json` files. +Their package metadata points to the corresponding `kunchenguid/*-axi` repositories. + +## Live evidence commands + +The following commands were run from the Firstmate task worktree. + +```sh +for tool in tasks-axi gh-axi lavish-axi quota-axi chrome-devtools-axi no-mistakes herdr treehouse; do + command -v "$tool" + "$tool" --version +done +npm prefix -g +npm root -g +npm list -g --depth=0 +brew list --versions herdr treehouse +brew info --json=v2 herdr treehouse +``` + +The exact `--version` outputs were: + +```text +tasks-axi: 0.2.3 +gh-axi: 0.1.27 +lavish-axi: 0.1.42 +quota-axi: 0.1.6 +chrome-devtools-axi: 0.1.26 +no-mistakes: no-mistakes version v1.40.3 (d873960) 2026-07-22T01:41:55Z +herdr: herdr 0.7.5 +treehouse: v2.1.0 +``` + +The live mirror acceptance run completed on 2026-07-29 local time, or 2026-07-30 UTC. +`bin/fm-toolchain-mirror.sh snapshot` created snapshot `20260730T030202Z-Darwin-arm64` with eight artifacts. +`bin/fm-toolchain-mirror.sh verify` checked all eight artifacts successfully. +An isolated-prefix restore then reproduced every version output above. +The local mirror occupied 67 MiB, and its `manifest.tsv` SHA-256 was `34985504ab51d5eee5b36ba4582dbc9204d816c3eebee5a2976ff5e60e1d422d`. + +## Local offline mirror + +[`bin/fm-toolchain-mirror.sh`](../bin/fm-toolchain-mirror.sh) snapshots the eight installed tools without contacting an upstream or changing their versions. +Its default destination is `$FM_HOME/data/toolchain-mirror`. +The repository already ignores `data/`, so the large platform-specific package trees and binaries remain local. +Set `FM_TOOLCHAIN_MIRROR` or pass `--mirror` to use another operator-controlled volume. + +Each snapshot contains: + +- The complete installed directory for each npm package, including its installed dependencies. +- Exact binary bytes for `no-mistakes`, `herdr`, and `treehouse`. +- Raw `--version` output for every tool. +- A TSV manifest with versions, install sources, update mechanisms, and SHA-256 checksums. +- The operating system and architecture required by the captured binaries. + +Create and verify a snapshot: + +```sh +bin/fm-toolchain-mirror.sh snapshot +bin/fm-toolchain-mirror.sh verify +``` + +Restore the current snapshot offline into a new, isolated prefix: + +```sh +recovery_prefix="$FM_HOME/data/toolchain-recovery/$(date -u +%Y%m%dT%H%M%SZ)" +bin/fm-toolchain-mirror.sh restore --prefix "$recovery_prefix" +export PATH="$recovery_prefix/bin:$PATH" +``` + +Restore one tool by adding `--tool <name>`. +The restore refuses an existing prefix, verifies every selected checksum before writing, and verifies the restored `--version` before publishing the new prefix. + +## Upstream break or disappearance recovery + +1. Stop issuing update commands and preserve the broken installation for diagnosis. +2. Run `bin/fm-toolchain-mirror.sh verify` against the last known-good local snapshot. +3. Restore that snapshot to a new prefix and place its `bin/` first on `PATH`. +4. Re-run the live evidence commands above and the affected Firstmate bootstrap or task path. +5. Resume fleet work only after the pinned commands match the mirror manifest. + +Do not restart or update the shared `no-mistakes` daemon while any lane has an active pipeline. +If the restored CLI and running daemon are incompatible, drain active lanes and let the Firstmate operator own the daemon recovery. + +The mirror does not disable any updater. +That preserves normal update notices and operator choice, but it also means a later manual update can still introduce breaking bytes. +After every deliberate, verified tool upgrade, create a new snapshot and retain the previous known-good snapshot until fleet validation passes. diff --git a/docs/verification/moshi-mobile-review.md b/docs/verification/moshi-mobile-review.md new file mode 100644 index 00000000000..d604c440447 --- /dev/null +++ b/docs/verification/moshi-mobile-review.md @@ -0,0 +1,67 @@ +# Moshi mobile review verification + +Audience: maintainer verification. + +This record covers the current Firstmate-owned mobile presentation and review handoff boundary. +Moshi product behavior was checked against its official Browser Preview, Diff, Chat View, and Hooks documentation on 2026-07-31. +The operator contract is [`docs/moshi-mobile-review.md`](../moshi-mobile-review.md), and the agent contract is [`mobile-mode`](../../.agents/skills/mobile-mode/SKILL.md). + +## Test boundary + +The natural-language trigger and captain-facing message shape are agent behavior, not a shell executable. +That is an explicit agent-behavior test exception: do not add a test that parses or asserts instruction source bytes. +Deterministic validation covers the maintained-prose inventory, local links, repository lint surface, and changed-file-selected behavior suite. +Fresh-context dogfood covers whether an agent loads the public `AGENTS.md` trigger and produces the required mobile handoff. +Live mobile execution status: NOT RUN. No Moshi or Browser Preview session was available to this worker, so no Preview, Diff, Chat View, or fallback end-user evidence is claimed; the complete captain-run checklist remains in [`docs/moshi-mobile-review.md`](../moshi-mobile-review.md). + +## Moshi facts in scope + +The official [Browser Preview documentation](https://getmoshi.app/docs/browser-preview) says host-local HTTP servers are detected by `moshi-hook` and reached in-app through the active SSH-capable session without a public tunnel. +The official [Diff documentation](https://getmoshi.app/docs/diff-viewer) says Diff reads the connected host's staged, unstaged, and untracked working-tree state and keeps diff contents host-local. +The official [Chat View documentation](https://getmoshi.app/docs/chat-view) says Chat View renders the same live agent session, currently lists Claude Code, Codex CLI, OpenCode, and Pi, and requires tmux or Herdr. +The official [Hooks documentation](https://getmoshi.app/docs/hooks) says hook support is broader than Chat View support and that inbox summaries and approval routing are separate from host-local transcript, diff, source-file, and terminal traffic. + +## Supported harness review + +The supported Firstmate harness list comes from [`harness-adapters`](../../.agents/skills/harness-adapters/SKILL.md). + +| Firstmate harness | Mobile presentation | Existing Moshi hook surface | Chat View | Firstmate adapter change | +| --- | --- | --- | --- | --- | +| Claude | Applies through the shared agent contract. | Official Moshi hooks support exists; configuration is external and unchanged. | Listed by Moshi. | None. | +| Codex | Applies through the shared agent contract. | Official Moshi hooks support exists; configuration is external and unchanged. | Listed by Moshi. | None. | +| OpenCode | Applies through the shared agent contract. | Official Moshi hooks support exists; configuration is external and unchanged. | Listed by Moshi. | None. | +| Pi | Applies through the shared agent contract. | Official Moshi hooks support exists, but Firstmate's Pi has no permission system, so approval authority is not applicable. | Listed by Moshi. | None. | +| pi-signed | Applies through the shared agent contract. | It uses the Pi engine, but wrapper-specific Moshi detection is not independently verified. | Use concise numbered Firstmate chat in the same Moshi/Firstmate session unless Moshi recognizes it as Pi. | None. | +| Grok | Applies through the shared agent contract. | Official Moshi hooks support exists; configuration is external and unchanged. | Not in the current official Chat View list, so use concise numbered Firstmate chat in the same Moshi/Firstmate session. | None. | +| Kimi | Applies through the shared agent contract. | Official Moshi hooks support exists; configuration is external and unchanged. | Not in the current official Chat View list, so use concise numbered Firstmate chat in the same Moshi/Firstmate session. | None. | + +The repository's Claude, Codex, OpenCode, Pi, Grok, and Kimi hook or extension surfaces were inspected for ownership overlap. +This slice changes none of them and makes no compatibility claim about the captain's external Moshi-managed hook configuration. + +## Supported runtime backend review + +The supported spawn backend list comes from `FM_BACKEND_SPAWN` in [`bin/fm-backend.sh`](../../bin/fm-backend.sh). + +| Firstmate backend | Moshi mobile path | Applicability to this slice | +| --- | --- | --- | +| tmux | Moshi Chat View currently supports tmux; Browser Preview and Diff use the host gateway. | No backend behavior changes. | +| Herdr | Selected Firstmate mobile path; Moshi Chat View currently supports Herdr; Browser Preview and Diff use the host gateway. | No backend behavior changes. | +| Zellij | Moshi can detect Zellij for terminal context, but current Chat View requirements exclude it. | Terminal, Browser Preview, and Diff may remain usable; Chat View falls back to concise numbered Firstmate chat in the same Moshi/Firstmate session; no backend behavior changes. | +| Orca | No Firstmate-to-Moshi session integration is claimed. | Not applicable to the selected Herdr workflow; no backend behavior changes. | +| cmux | No Firstmate-to-Moshi session integration is claimed. | Not applicable to the selected Herdr workflow; no backend behavior changes. | +| Codex App | Firstmate does not accept it as a runtime backend. | Not applicable. | + +Moshi remains absent from the backend registry by design. +The mobile contract changes presentation and review handoff only, so spawn, supervision, recovery, cleanup, and backend metadata need no new branch. + +## Verification entry points + +Run the focused maintained-prose check and the repository's changed-file-selected canonical behavior entrypoint: + +```sh +bin/fm-doc-audience-check.sh +bin/fm-test-run.sh --changed +``` + +If shell surfaces change in a future extension, also run `bin/fm-lint.sh` and the relevant focused test script. +This slice changes no shell surface, but the delivery pipeline still owns its canonical lint gate. diff --git a/tests/fm-better-stack-incidents.test.sh b/tests/fm-better-stack-incidents.test.sh index 86320bea0f4..4b6eb4061e2 100755 --- a/tests/fm-better-stack-incidents.test.sh +++ b/tests/fm-better-stack-incidents.test.sh @@ -36,16 +36,22 @@ shift exec env BETTER_STACK_API_TOKEN="${FM_TEST_BETTER_STACK_TOKEN:-}" "$@" SH - cat > "$fakebin/curl" <<'SH' +cat > "$fakebin/curl" <<'SH' #!/usr/bin/env bash cat >/dev/null printf '%s\n' "$*" >> "$FM_TEST_CURL_LOG" [ "${FM_TEST_CURL_FAIL:-0}" = 0 ] || exit 7 -printf '%s\n%s' "${FM_TEST_API_BODY:-}" "${FM_TEST_API_CODE:-200}" +count=$(cat "$FM_TEST_CURL_COUNT" 2>/dev/null || echo 0) +count=$((count + 1)) +printf '%s\n' "$count" > "$FM_TEST_CURL_COUNT" +body=${FM_TEST_API_BODY:-} +[ "$count" -eq 2 ] && body=${FM_TEST_API_BODY_2:-$body} +printf '%s\n%s' "$body" "${FM_TEST_API_CODE:-200}" SH chmod +x "$fakebin/doppler" "$fakebin/curl" : > "$dir/doppler.log" : > "$dir/curl.log" + : > "$dir/curl.count" printf '%s\n' "$dir" } @@ -56,6 +62,7 @@ run_poll() { FM_HOME="$dir/home" \ FM_TEST_DOPPLER_LOG="$dir/doppler.log" \ FM_TEST_CURL_LOG="$dir/curl.log" \ + FM_TEST_CURL_COUNT="$dir/curl.count" \ "$@" "$POLL" } @@ -76,14 +83,6 @@ test_new_incident_and_duplicate_suppression() { expect_code 0 "$rc" "new incident poll exit" [ "$out" = 'better-stack-incident opened id=25 name=api-production started=2026-08-01T12:00:00.000Z' ] \ || fail "new incident must print one compact identity line (got: $out)" - assert_present "$dir/home/state/better-stack-incidents.seen/25" \ - "new incident must claim a private seen marker" - - out=$(run_poll "$dir" env \ - FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ - FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200); rc=$? - expect_code 0 "$rc" "duplicate incident poll exit" - [ -z "$out" ] || fail "an already-seen incident must stay silent (got: $out)" assert_grep '--project fleet-observability --config prd' "$dir/doppler.log" \ "poll must select the fleet-observability/prd Doppler scope" @@ -99,6 +98,25 @@ test_new_incident_and_duplicate_suppression() { pass "new Better Stack incident wakes once and duplicate observations stay silent" } +test_pagination_and_unsafe_target_rejection() { + local dir body next_body out rc + dir=$(make_case pagination) + body=$(jq -cn --arg next 'https://uptime.betterstack.com/api/v3/incidents?page=2&resolved=false' \ + '{data: [{id:"25", type:"incident", attributes:{name:"page-one", started_at:"t1", resolved_at:null}}], pagination:{next:$next}}') + next_body=$(incident_body 26 page-two t2) + out=$(run_poll "$dir" env FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_BODY_2="$next_body" FM_TEST_API_CODE=200) + [ "$out" = $'better-stack-incident opened id=25 name=page-one started=t1\nbetter-stack-incident opened id=26 name=page-two started=t2' ] \ + || fail "poll must include unresolved incidents from later pages (got: $out)" + body=$(jq -cn '{data: [], pagination:{next:"https://evil.example/api/v3/incidents?resolved=false"}}') + out=$(run_poll "$dir" env FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200); rc=$? + expect_code 0 "$rc" "unsafe pagination target poll exit" + [ "$out" = 'better-stack-error invalid Better Stack pagination target' ] \ + || fail "unsafe pagination target must produce one diagnostic (got: $out)" + pass "Better Stack pagination reaches later pages and rejects unsafe targets" +} + test_api_error_reports_once_and_recovers() { local dir out rc dir=$(make_case api-error) @@ -175,6 +193,7 @@ SH out=$(PATH="$dir/fakebin:$BASE_PATH" \ FM_HOME="$dir/home" FM_STATE_OVERRIDE="$state" \ FM_TEST_DOPPLER_LOG="$dir/doppler.log" FM_TEST_CURL_LOG="$dir/curl.log" \ + FM_TEST_CURL_COUNT="$dir/curl.count" \ FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200 \ FM_CHECK_INTERVAL=0 FM_HEARTBEAT=999999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ @@ -184,9 +203,68 @@ SH "registered poll output must become a check wake with the incident identity" [ "$(grep -c 'better-stack-incident opened id=91' "$state/.wake-queue")" -eq 1 ] \ || fail "new incident must create exactly one durable wake record" + rm -f "$state/.last-check" + out=$(PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$dir/home" FM_STATE_OVERRIDE="$state" \ + FM_TEST_DOPPLER_LOG="$dir/doppler.log" FM_TEST_CURL_LOG="$dir/curl.log" \ + FM_TEST_CURL_COUNT="$dir/curl.count" \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200 \ + FM_CHECK_INTERVAL=0 FM_HEARTBEAT=999999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + "$WATCH"); rc=$? + expect_code 0 "$rc" "duplicate watcher incident delivery exit" + [ "$(grep -c 'better-stack-incident opened id=91' "$state/.wake-queue")" -eq 1 ] \ + || fail "replayed incident must not create a duplicate durable wake" pass "registered Better Stack poll delivers exactly one authenticated check wake" } +test_incident_receipt_recovers_after_marker_failure_and_queue_drain() { + local dir state shim body out rc + dir=$(make_case receipt-recovery) + state="$dir/home/state" + shim="$state/better-stack-incidents.check.sh" + body=$(incident_body 92 recovery-production) + cat > "$shim" <<SH +#!/usr/bin/env bash +export FM_HOME=$(printf '%q' "$dir/home") +exec $(printf '%q' "$POLL") +SH + chmod 0700 "$shim" + FM_HOME="$dir/home" "$REGISTER" better-stack-incidents >/dev/null \ + || fail "could not register receipt-recovery custom check" + printf 'not-a-directory\n' > "$state/better-stack-incidents.seen" + out=$(PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$dir/home" FM_STATE_OVERRIDE="$state" \ + FM_TEST_DOPPLER_LOG="$dir/doppler.log" FM_TEST_CURL_LOG="$dir/curl.log" \ + FM_TEST_CURL_COUNT="$dir/curl.count" FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200 \ + FM_CHECK_INTERVAL=0 FM_HEARTBEAT=999999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + "$WATCH"); rc=$? + [ "$rc" -ne 0 ] || fail "marker publication failure must stop the watcher" + assert_present "$state/better-stack-incidents.receipts/92" \ + "queue append failure boundary must leave a private recovery receipt" + [ "$(grep -c 'better-stack-incident opened id=92' "$state/.wake-queue")" -eq 1 ] \ + || fail "failed marker publication must retain exactly one queued wake" + rm -f "$state/.wake-queue" + rm -f "$state/better-stack-incidents.seen" + mkdir "$state/better-stack-incidents.seen" + chmod 700 "$state/better-stack-incidents.seen" + rm -f "$state/.last-check" + out=$(PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$dir/home" FM_STATE_OVERRIDE="$state" \ + FM_TEST_DOPPLER_LOG="$dir/doppler.log" FM_TEST_CURL_LOG="$dir/curl.log" \ + FM_TEST_CURL_COUNT="$dir/curl.count" FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200 \ + FM_CHECK_INTERVAL=0 FM_HEARTBEAT=999999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + "$WATCH"); rc=$? + expect_code 0 "$rc" "receipt recovery watcher exit" + assert_present "$state/better-stack-incidents.seen/92" \ + "receipt recovery must converge the private seen marker after queue drain" + [ ! -e "$state/.wake-queue" ] || [ -z "$(cat "$state/.wake-queue")" ] \ + || fail "receipt recovery after queue drain must not append a duplicate wake" + pass "incident receipt recovers marker publication after queue drain" +} + test_bootstrap_arms_and_retires_home_check() { local dir home out sum1 sum2 dir=$(make_case bootstrap) @@ -239,8 +317,10 @@ test_incident_check_keeps_home_supervised() { } test_new_incident_and_duplicate_suppression +test_pagination_and_unsafe_target_rejection test_api_error_reports_once_and_recovers test_missing_token_reports_once test_registered_check_delivers_check_wake +test_incident_receipt_recovers_after_marker_failure_and_queue_drain test_bootstrap_arms_and_retires_home_check test_incident_check_keeps_home_supervised diff --git a/tests/fm-brief.test.sh b/tests/fm-brief.test.sh index 0199311824b..2d7fbba6b7b 100755 --- a/tests/fm-brief.test.sh +++ b/tests/fm-brief.test.sh @@ -173,10 +173,94 @@ PERL test_help_includes_entire_header() { local help help=$("$ROOT/bin/fm-brief.sh" --help) - assert_contains "$help" "Refuses to overwrite an existing brief." "fm-brief.sh --help omitted its header terminator" + assert_contains "$help" "--force-regenerate renders a fresh scaffold before archiving" \ + "fm-brief.sh --help omitted archive-and-regenerate mechanics" pass "fm-brief.sh: --help renders the complete header" } +test_existing_brief_refusal_detects_staleness_and_force_regenerates() { + local home id brief err out status archive_count archive + home="$TMP_ROOT/stale-brief-home" + id=stale-brief-guard + brief="$home/data/$id/brief.md" + err="$home/refusal.err" + out="$home/regenerate.out" + mkdir -p "$(dirname "$brief")" + printf '%s\n' 'months-old draft without current safety contracts' > "$brief" + + status=0 + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" "$id" some-proj > /dev/null 2>"$err" || status=$? + expect_code 1 "$status" "scaffolding over an existing stale brief must fail" + assert_grep "missing current scaffold safety marker" "$err" \ + "existing stale brief refusal did not detect its missing safety marker" + assert_grep "Do not launch this brief unchanged" "$err" \ + "existing stale brief refusal did not prevent unchanged launch" + assert_grep "--force-regenerate" "$err" \ + "existing stale brief refusal did not name the archive-and-regenerate recovery" + assert_grep "months-old draft" "$brief" \ + "ordinary refusal changed the existing stale brief" + + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" "$id" some-proj --force-regenerate >"$out" 2>&1 \ + || fail "--force-regenerate did not replace a stale brief safely" + assert_grep "firstmate-brief-scaffold-safety:v1" "$brief" \ + "regenerated brief is missing the current scaffold safety marker" + assert_grep "archived existing brief:" "$out" \ + "--force-regenerate did not report the archive path" + archive_count=$(find "$(dirname "$brief")" -maxdepth 1 -type f -name 'brief.md.archive-*' | wc -l | tr -d ' ') + [ "$archive_count" = 1 ] \ + || fail "--force-regenerate must create exactly one archive, found $archive_count" + archive=$(find "$(dirname "$brief")" -maxdepth 1 -type f -name 'brief.md.archive-*' -print) + assert_grep "months-old draft without current safety contracts" "$archive" \ + "--force-regenerate archive did not preserve the stale brief" + pass "fm-brief.sh: stale existing briefs fail loudly and force regeneration archives before replacing" +} + +test_existing_current_brief_still_requires_freshness_verification() { + local home id brief err status + home="$TMP_ROOT/current-brief-home" + id=current-brief-guard + err="$home/refusal.err" + mkdir -p "$home/data" + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" "$id" some-proj >/dev/null 2>&1 \ + || fail "current brief fixture did not scaffold" + brief="$home/data/$id/brief.md" + assert_grep "firstmate-brief-scaffold-safety:v1" "$brief" \ + "fresh brief fixture is missing the current scaffold safety marker" + + status=0 + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" "$id" some-proj >/dev/null 2>"$err" || status=$? + expect_code 1 "$status" "scaffolding over an existing current brief must still fail" + assert_grep "current scaffold safety marker is present" "$err" \ + "existing current brief refusal did not report marker status" + assert_grep "task freshness is unverified" "$err" \ + "existing current brief refusal falsely treated its marker as freshness proof" + assert_grep "verify it intentionally, or rerun with --force-regenerate" "$err" \ + "existing current brief refusal did not give both safe recovery choices" + pass "fm-brief.sh: a current safety marker never substitutes for task freshness verification" +} + +test_force_regeneration_preserves_brief_when_rendering_fails() { + local home id brief fake_root status archive_count + home="$TMP_ROOT/failed-regeneration-home" + id=failed-regeneration-guard + brief="$home/data/$id/brief.md" + fake_root="$home/fake-root" + mkdir -p "$(dirname "$brief")" "$fake_root/bin" + printf '%s\n' 'original brief must survive a failed regeneration' > "$brief" + printf '%s\n' '#!/usr/bin/env bash' 'exit 1' > "$fake_root/bin/fm-project-mode.sh" + chmod +x "$fake_root/bin/fm-project-mode.sh" + + status=0 + FM_HOME="$home" FM_ROOT_OVERRIDE="$fake_root" \ + "$ROOT/bin/fm-brief.sh" "$id" some-proj --force-regenerate >/dev/null 2>&1 || status=$? + expect_code 1 "$status" "failed regeneration must return the render failure" + assert_grep "original brief must survive" "$brief" \ + "failed regeneration removed or changed the live brief" + archive_count=$(find "$(dirname "$brief")" -maxdepth 1 -type f -name 'brief.md.archive-*' | wc -l | tr -d ' ') + [ "$archive_count" = 0 ] || fail "failed regeneration archived the live brief before rendering" + pass "fm-brief.sh: failed force regeneration preserves the live brief" +} + # Registry with one project per delivery mode, so each ship-mode DOD branch is # exercised. A project absent from the registry defaults to no-mistakes. write_registry() { @@ -621,6 +705,9 @@ test_scout_and_secondmate_scaffold() { test_script_parses test_no_heredoc_in_command_substitution test_help_includes_entire_header +test_existing_brief_refusal_detects_staleness_and_force_regenerates +test_existing_current_brief_still_requires_freshness_verification +test_force_regeneration_preserves_brief_when_rendering_fails test_ship_modes_generate_clean_briefs test_faster_paths_use_configured_authority_without_stacked_review test_no_mistakes_dod_wording diff --git a/tests/fm-lint.test.sh b/tests/fm-lint.test.sh index 17fb097f758..35d3bad419a 100755 --- a/tests/fm-lint.test.sh +++ b/tests/fm-lint.test.sh @@ -46,6 +46,154 @@ test_pins_an_explicit_version() { pass "fm-lint.sh pins an explicit ShellCheck version ($REQUIRED)" } +test_installer_selects_supported_platform_assets() { + local tmp fakebin destination out system machine expected_asset expected_sha expected_archive actual_url + tmp=$(fm_test_tmproot fm-shellcheck-platform) + fakebin=$(fm_fakebin "$tmp") + + cat > "$fakebin/uname" <<'SH' +#!/usr/bin/env bash +case "$1" in + -s) printf '%s\n' "$TEST_UNAME_S" ;; + -m) printf '%s\n' "$TEST_UNAME_M" ;; + *) exit 2 ;; +esac +SH + cat > "$fakebin/curl" <<'SH' +#!/usr/bin/env bash +printf '%s\n' "$2" > "$CURL_URL" +: > "$4" +SH + cat > "$fakebin/sha256sum" <<'SH' +#!/usr/bin/env bash +printf '%s %s\n' "$EXPECTED_SHA" "$1" +SH + cat > "$fakebin/shasum" <<'SH' +#!/usr/bin/env bash +[ "$1" = "-a" ] && [ "$2" = "256" ] || exit 2 +printf '%s %s\n' "$EXPECTED_SHA" "$3" +SH + cat > "$fakebin/tar" <<'SH' +#!/usr/bin/env bash +while [ "$#" -gt 0 ]; do + if [ "$1" = "-C" ]; then + mkdir -p "$2/shellcheck-v${TEST_VERSION}" + cat > "$2/shellcheck-v${TEST_VERSION}/shellcheck" <<EOF +#!/usr/bin/env bash +printf 'ShellCheck - shell script analysis tool\nversion: ${TEST_VERSION}\n' +EOF + chmod +x "$2/shellcheck-v${TEST_VERSION}/shellcheck" + exit 0 + fi + shift +done +exit 2 +SH + chmod +x "$fakebin/uname" "$fakebin/curl" "$fakebin/sha256sum" "$fakebin/shasum" "$fakebin/tar" + + while IFS='|' read -r system machine expected_asset expected_sha; do + destination="$tmp/bin-$system-$machine" + expected_archive="shellcheck-v${REQUIRED}.${expected_asset}.tar.xz" + out=$(TEST_UNAME_S="$system" TEST_UNAME_M="$machine" \ + TEST_VERSION="$REQUIRED" EXPECTED_SHA="$expected_sha" CURL_URL="$tmp/curl-url" \ + PATH="$fakebin:$PATH" "$INSTALLER" "$destination" 2>&1) \ + || fail "installer rejected supported platform $system/$machine"$'\n'"$out" + actual_url=$(cat "$tmp/curl-url") + assert_contains "$actual_url" "/$expected_archive" \ + "installer selected the wrong asset for $system/$machine" + done <<'EOF' +Linux|x86_64|linux.x86_64|8c3be12b05d5c177a04c29e3c78ce89ac86f1595681cab149b65b97c4e227198 +Darwin|arm64|darwin.aarch64|56affdd8de5527894dca6dc3d7e0a99a873b0f004d7aabc30ae407d3f48b0a79 +Darwin|x86_64|darwin.x86_64|3c89db4edcab7cf1c27bff178882e0f6f27f7afdf54e859fa041fca10febe4c6 +EOF + pass "ShellCheck installer selects each supported platform asset" +} + +test_installer_uses_shasum_fallback() { + local tmp fakebin destination out + tmp=$(fm_test_tmproot fm-shellcheck-shasum) + fakebin=$(fm_fakebin "$tmp") + destination="$tmp/bin" + + cat > "$fakebin/uname" <<'SH' +#!/usr/bin/env bash +case "$1" in + -s) printf 'Darwin\n' ;; + -m) printf 'arm64\n' ;; + *) exit 2 ;; +esac +SH + cat > "$fakebin/curl" <<'SH' +#!/usr/bin/env bash +while [ "$#" -gt 0 ]; do + if [ "$1" = "-o" ]; then + : > "$2" + exit 0 + fi + shift +done +exit 2 +SH + cat > "$fakebin/shasum" <<'SH' +#!/usr/bin/env bash +[ "$1" = "-a" ] && [ "$2" = "256" ] || exit 2 +printf '56affdd8de5527894dca6dc3d7e0a99a873b0f004d7aabc30ae407d3f48b0a79 %s\n' "$3" +SH + cat > "$fakebin/tar" <<'SH' +#!/usr/bin/env bash +while [ "$#" -gt 0 ]; do + if [ "$1" = "-C" ]; then + mkdir -p "$2/shellcheck-v0.11.0" + cat > "$2/shellcheck-v0.11.0/shellcheck" <<'EOF' +#!/usr/bin/env bash +printf 'ShellCheck - shell script analysis tool\nversion: 0.11.0\n' +EOF + chmod +x "$2/shellcheck-v0.11.0/shellcheck" + exit 0 + fi + shift +done +exit 2 +SH + chmod +x "$fakebin/uname" "$fakebin/curl" "$fakebin/shasum" "$fakebin/tar" + + out=$(PATH="$fakebin:/usr/bin:/bin" "$INSTALLER" "$destination" 2>&1) \ + || fail "installer did not use the shasum fallback on Darwin"$'\n'"$out" + [ -x "$destination/shellcheck" ] || fail "installer did not install ShellCheck after shasum verification" + pass "ShellCheck installer uses shasum when sha256sum is unavailable" +} + +test_installer_rejects_unsupported_platform() { + local tmp fakebin destination out rc + tmp=$(fm_test_tmproot fm-shellcheck-platform-unsupported) + fakebin=$(fm_fakebin "$tmp") + destination="$tmp/bin" + + cat > "$fakebin/uname" <<'SH' +#!/usr/bin/env bash +case "$1" in + -s) printf 'Plan9\n' ;; + -m) printf 'mips64\n' ;; + *) exit 2 ;; +esac +SH + cat > "$fakebin/curl" <<'SH' +#!/usr/bin/env bash +printf 'curl must not run for an unsupported platform\n' >&2 +exit 99 +SH + chmod +x "$fakebin/uname" "$fakebin/curl" + + rc=0 + out=$(PATH="$fakebin:$PATH" "$INSTALLER" "$destination" 2>&1) || rc=$? + [ "$rc" -ne 0 ] || fail "installer accepted unsupported platform Plan9/mips64" + assert_contains "$out" "fm-install-shellcheck.sh: unsupported platform: Plan9/mips64" \ + "installer did not report the exact unsupported platform" + assert_not_contains "$out" "curl must not run" \ + "installer attempted a download for an unsupported platform" + pass "ShellCheck installer rejects unsupported platforms before download" +} + test_installer_retries_transient_download_failure() { local tmp fakebin destination out tmp=$(fm_test_tmproot fm-shellcheck-download) @@ -67,6 +215,14 @@ while [ "$#" -gt 0 ]; do shift done exit 2 +SH + cat > "$fakebin/uname" <<'SH' +#!/usr/bin/env bash +case "$1" in + -s) printf 'Linux\n' ;; + -m) printf 'x86_64\n' ;; + *) exit 2 ;; +esac SH cat > "$fakebin/sha256sum" <<'SH' #!/usr/bin/env bash @@ -92,7 +248,7 @@ SH #!/usr/bin/env bash exit 0 SH - chmod +x "$fakebin/curl" "$fakebin/sha256sum" "$fakebin/tar" "$fakebin/sleep" + chmod +x "$fakebin/curl" "$fakebin/uname" "$fakebin/sha256sum" "$fakebin/tar" "$fakebin/sleep" out=$(CURL_COUNT="$tmp/curl-count" PATH="$fakebin:$PATH" "$INSTALLER" "$destination" 2>&1) \ || fail "installer did not recover from a transient download failure"$'\n'"$out" @@ -428,6 +584,9 @@ SH test_list_files_reports_the_shell_inventory test_pins_an_explicit_version +test_installer_selects_supported_platform_assets +test_installer_uses_shasum_fallback +test_installer_rejects_unsupported_platform test_installer_retries_transient_download_failure test_rejects_wrong_shellcheck_version test_catches_a_real_lint_defect diff --git a/tests/fm-operational-input.test.sh b/tests/fm-operational-input.test.sh index 2d1b1c39de3..7e4e9b93165 100755 --- a/tests/fm-operational-input.test.sh +++ b/tests/fm-operational-input.test.sh @@ -48,6 +48,27 @@ test_current_generic_matrix() { pass "operational input: every current generic envelope retains its exact structured kind" } +test_launch_brief_cli_carries_dispatch_provenance() { + local encoded stripped + encoded=$(printf '%s' 'CREWMATE_BRIEF_BODY' | "$OWNER" encode launch-brief) \ + || fail "launch-brief CLI encoding failed" + [ "$(kind_cli "$encoded")" = launch-brief ] \ + || fail "launch-brief provenance changed the structured kind" + stripped=$(printf '%s' "$encoded" | "$OWNER" body) \ + || fail "launch-brief provenance body could not be recovered" + assert_contains "$stripped" "This is a genuine Firstmate dispatch." \ + "launch-brief did not identify itself as genuine Firstmate dispatch" + assert_contains "$stripped" "canonical operational-input encoder" \ + "launch-brief did not explain the typed envelope's owning encoder" + assert_contains "$stripped" "not project content" \ + "launch-brief did not distinguish its provenance envelope from project content" + assert_contains "$stripped" "proceed with the brief below" \ + "launch-brief did not tell the worker that proceeding is expected" + assert_contains "$stripped" "CREWMATE_BRIEF_BODY" \ + "launch-brief provenance dropped the original brief body" + pass "operational input: launch briefs explain their canonical Firstmate provenance before the task body" +} + test_current_from_firstmate_carrier() { local encoded parsed separator separator=$(printf '\342\201\243') @@ -152,6 +173,7 @@ test_invalid_current_encodings_are_rejected() { } test_current_generic_matrix +test_launch_brief_cli_carries_dispatch_provenance test_current_from_firstmate_carrier test_landed_untyped_prefix_is_explicitly_legacy test_isolated_legacy_matrix diff --git a/tests/fm-toolchain-mirror.test.sh b/tests/fm-toolchain-mirror.test.sh new file mode 100755 index 00000000000..3f1a18fffe6 --- /dev/null +++ b/tests/fm-toolchain-mirror.test.sh @@ -0,0 +1,128 @@ +#!/usr/bin/env bash +# Contract and behavior tests for the offline toolchain mirror. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +MIRROR_SCRIPT="$ROOT/bin/fm-toolchain-mirror.sh" +assert_present "$MIRROR_SCRIPT" "bin/fm-toolchain-mirror.sh is missing" +[ -x "$MIRROR_SCRIPT" ] || fail "fm-toolchain-mirror.sh must be executable" +TMP_ROOT=$(fm_test_tmproot fm-toolchain-mirror) +trap 'rm -rf "$TMP_ROOT"' EXIT + +make_version_tool() { + path=$1 + output=$2 + mkdir -p "$(dirname "$path")" + { + printf '%s\n' '#!/usr/bin/env bash' + printf 'printf '\''%%s\\n'\'' '\''%s'\''\n' "$output" + } > "$path" + chmod +x "$path" +} + +make_npm_tool() { + tool=$1 + version=$2 + bin_relative=$3 + package_dir="$CASE_DIR/npm-root/$tool" + mkdir -p "$package_dir/$(dirname "$bin_relative")" "$package_dir/node_modules/example-dependency" + cat > "$package_dir/package.json" <<EOF +{"name":"$tool","version":"$version","bin":{"$tool":"$bin_relative"}} +EOF + cat > "$package_dir/$bin_relative" <<EOF +#!/usr/bin/env node +console.log("$version"); +EOF + chmod +x "$package_dir/$bin_relative" + printf '%s\n' 'offline dependency proof' > "$package_dir/node_modules/example-dependency/proof.txt" + ln -s "$package_dir/$bin_relative" "$CASE_DIR/fakebin/$tool" +} + +setup_case() { + CASE_DIR="$TMP_ROOT/$1" + mkdir -p "$CASE_DIR/fakebin" "$CASE_DIR/npm-root" + make_npm_tool tasks-axi 0.2.3 dist/bin/tasks-axi.js + make_npm_tool gh-axi 0.1.27 lib/custom-gh-entry.js + make_npm_tool lavish-axi 0.1.42 dist/cli.mjs + make_npm_tool quota-axi 0.1.6 dist/bin/quota-axi.js + make_npm_tool chrome-devtools-axi 0.1.26 dist/bin/chrome-devtools-axi.js + make_version_tool "$CASE_DIR/fakebin/no-mistakes" \ + 'no-mistakes version v1.40.3 (test) 2026-07-22T01:41:55Z' + make_version_tool "$CASE_DIR/fakebin/herdr" 'herdr 0.7.5' + make_version_tool "$CASE_DIR/fakebin/treehouse" 'v2.1.0' +} + +test_snapshot_and_offline_restore() { + setup_case snapshot-restore + mirror="$CASE_DIR/mirror" + restore="$CASE_DIR/restored" + out=$(PATH="$CASE_DIR/fakebin:$PATH" \ + FM_TOOLCHAIN_NPM_ROOT="$CASE_DIR/npm-root" \ + FM_TOOLCHAIN_SNAPSHOT_ID=test-snapshot \ + "$MIRROR_SCRIPT" snapshot --mirror "$mirror") || fail "snapshot command failed" + + assert_contains "$out" "snapshot: test-snapshot" "snapshot must report its id" + assert_contains "$out" "artifacts: 8" "snapshot must contain all eight tools" + assert_present "$mirror/current" "snapshot must publish the current marker" + assert_present "$mirror/snapshots/test-snapshot/manifest.tsv" "snapshot must publish a manifest" + assert_present "$mirror/snapshots/test-snapshot/versions/no-mistakes.txt" \ + "snapshot must retain raw version output" + assert_contains "$(cat "$mirror/snapshots/test-snapshot/versions/no-mistakes.txt")" \ + "no-mistakes version v1.40.3" "raw version output must be preserved" + + verify_out=$("$MIRROR_SCRIPT" verify --mirror "$mirror") || fail "verify command failed" + assert_contains "$verify_out" "(8 artifacts)" "verify must check every artifact" + + restore_out=$("$MIRROR_SCRIPT" restore --mirror "$mirror" --prefix "$restore") \ + || fail "restore command failed" + assert_contains "$restore_out" "(8 tools)" "restore must reinstall every tool" + assert_present "$restore/bin/no-mistakes" "restore must install binary tools" + assert_present "$restore/lib/node_modules/lavish-axi/node_modules/example-dependency/proof.txt" \ + "restore must retain installed npm dependencies for offline use" + [ "$("$restore/bin/tasks-axi" --version)" = "0.2.3" ] \ + || fail "restored tasks-axi must report the pinned version" + [ "$("$restore/bin/treehouse" --version)" = "v2.1.0" ] \ + || fail "restored treehouse must report the pinned version" + pass "snapshot and restore preserve all pinned tools offline" +} + +test_restore_refuses_existing_prefix() { + setup_case existing-prefix + mirror="$CASE_DIR/mirror" + PATH="$CASE_DIR/fakebin:$PATH" \ + FM_TOOLCHAIN_NPM_ROOT="$CASE_DIR/npm-root" \ + FM_TOOLCHAIN_SNAPSHOT_ID=test-snapshot \ + "$MIRROR_SCRIPT" snapshot --mirror "$mirror" >/dev/null \ + || fail "snapshot setup failed" + mkdir -p "$CASE_DIR/existing" + if "$MIRROR_SCRIPT" restore --mirror "$mirror" --prefix "$CASE_DIR/existing" \ + >"$CASE_DIR/restore.out" 2>"$CASE_DIR/restore.err"; then + fail "restore must refuse an existing prefix" + fi + assert_contains "$(cat "$CASE_DIR/restore.err")" "restore prefix already exists" \ + "restore refusal must explain how to recover safely" + pass "restore refuses to overwrite an existing installation" +} + +test_verify_rejects_tampered_artifact() { + setup_case tampered-artifact + mirror="$CASE_DIR/mirror" + PATH="$CASE_DIR/fakebin:$PATH" \ + FM_TOOLCHAIN_NPM_ROOT="$CASE_DIR/npm-root" \ + FM_TOOLCHAIN_SNAPSHOT_ID=test-snapshot \ + "$MIRROR_SCRIPT" snapshot --mirror "$mirror" >/dev/null \ + || fail "snapshot setup failed" + printf '%s\n' 'tamper' >> "$mirror/snapshots/test-snapshot/artifacts/bin/treehouse" + if "$MIRROR_SCRIPT" verify --mirror "$mirror" >"$CASE_DIR/verify.out" 2>"$CASE_DIR/verify.err"; then + fail "verify must reject a tampered artifact" + fi + assert_contains "$(cat "$CASE_DIR/verify.err")" "checksum mismatch for treehouse" \ + "tamper refusal must name the affected tool" + pass "verify rejects changed mirror bytes" +} + +test_snapshot_and_offline_restore +test_restore_refuses_existing_prefix +test_verify_rejects_tampered_artifact From cf316da28eec66d0a749ec4570104bf32fa39e00 Mon Sep 17 00:00:00 2001 From: juniorlovestmh <272474227+juniorlovestmh@users.noreply.github.com> Date: Sat, 1 Aug 2026 13:52:03 -0300 Subject: [PATCH 52/52] test: keep Better Stack bootstrap hermetic --- tests/fm-better-stack-incidents.test.sh | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/tests/fm-better-stack-incidents.test.sh b/tests/fm-better-stack-incidents.test.sh index 4b6eb4061e2..bf43fe0c3ee 100755 --- a/tests/fm-better-stack-incidents.test.sh +++ b/tests/fm-better-stack-incidents.test.sh @@ -271,7 +271,8 @@ test_bootstrap_arms_and_retires_home_check() { home="$dir/home" : > "$home/config/better-stack-incidents" - out=$(FM_HOME="$home" "$ROOT/bin/fm-bootstrap.sh" 2>/dev/null) + out=$(PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$home" "$ROOT/bin/fm-bootstrap.sh" 2>/dev/null) assert_contains "$out" 'BETTER_STACK: incident monitoring on' \ "bootstrap must announce Better Stack incident monitoring" assert_present "$home/state/better-stack-incidents.check.sh" \ @@ -285,13 +286,15 @@ test_bootstrap_arms_and_retires_home_check() { sum1=$(cat "$home/state/better-stack-incidents.check.sh" \ "$home/state/better-stack-incidents.check-trust" | shasum) - FM_HOME="$home" "$ROOT/bin/fm-bootstrap.sh" >/dev/null 2>&1 + PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$home" "$ROOT/bin/fm-bootstrap.sh" >/dev/null 2>&1 sum2=$(cat "$home/state/better-stack-incidents.check.sh" \ "$home/state/better-stack-incidents.check-trust" | shasum) [ "$sum1" = "$sum2" ] || fail "bootstrap incident activation must be idempotent" rm "$home/config/better-stack-incidents" - out=$(FM_HOME="$home" "$ROOT/bin/fm-bootstrap.sh" 2>/dev/null) + out=$(PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$home" "$ROOT/bin/fm-bootstrap.sh" 2>/dev/null) assert_contains "$out" 'BETTER_STACK: incident monitoring off' \ "bootstrap must announce removal of an armed incident poll" assert_absent "$home/state/better-stack-incidents.check.sh" \