diff --git a/.agents/skills/bootstrap-diagnostics/SKILL.md b/.agents/skills/bootstrap-diagnostics/SKILL.md index 477980b8df1..e9a57058a55 100644 --- a/.agents/skills/bootstrap-diagnostics/SKILL.md +++ b/.agents/skills/bootstrap-diagnostics/SKILL.md @@ -32,6 +32,9 @@ When any diagnostic needs captain attention, report the plain consequence and re - `FLEET_SYNC: : skipped: ` - a benign one-off skip (offline, no origin, local-only); bootstrap continued, investigate only if it blocks work. A skip can also report the bounded fleet-refresh timeout (`FM_FLEET_SYNC_BOOTSTRAP_TIMEOUT`, or a fleet-size-aware default with a 20 second floor); a timeout never blocks startup. - `FLEET_SYNC: : recovered: ` - the clone had drifted onto a clean detached HEAD holding no unique commits and the sync self-healed it (re-attached the default branch and fast-forwarded); no action needed, it is reported only so the self-heal is visible. +- `BETTER_STACK: incident monitoring on ...` - the home-scoped poll is registered at the default check cadence; no action is needed. +- `BETTER_STACK: incident monitoring off - removed ...` - the local presence flag was removed and bootstrap retired the runnable check while retaining incident dedupe state; no action is needed. +- Any other `BETTER_STACK:` line - follow its concrete dependency, unsafe flag, activation, or cleanup diagnostic before relying on incident monitoring. - `FLEET_SYNC: : STUCK: on , N commits behind - needs attention` - the clone is dirty, on a non-default branch, detached with unique commits, or diverged, so the sync left it untouched (never forcing or discarding); it will keep falling behind until you look. A loud STUCK, especially a growing N across bootstraps, means that clone needs hands-on attention; dispatch a crewmate or resolve it before it strands work. - `PR_CHECK_MIGRATION: canonical polls rebuilt and armed; resume supervision for this home` - the non-executing migration rebuilt canonical task polls from validated metadata, and those polls are already armed. diff --git a/.opencode/plugins/fm-primary-watch-arm.js b/.opencode/plugins/fm-primary-watch-arm.js index 433edb80ab4..a8ae265ed2e 100644 --- a/.opencode/plugins/fm-primary-watch-arm.js +++ b/.opencode/plugins/fm-primary-watch-arm.js @@ -103,6 +103,7 @@ async function isPrimaryRoot(root, home) { function shouldArm(paths) { if (existsSync(`${paths.state}/.afk`)) return false; if (existsSync(`${paths.config}/x-mode.env`)) return true; + if (existsSync(`${paths.state}/better-stack-incidents.check.sh`)) return true; try { return readdirSync(paths.state).some((name) => name.endsWith(".meta")); } catch { diff --git a/AGENTS.md b/AGENTS.md index 1a1ad8e17d0..55bd8910bfd 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -74,6 +74,7 @@ config/startup-memory-budget primary-authoritative per-home startup-memory b config/herdr-presentation-spaces optional presence flag for Herdr's default-off disposable single-task visual projection; LOCAL, gitignored; inherited by secondmate homes; see docs/herdr-backend.md "Optional presentation spaces" config/cmux-socket-password optional cmux control-socket password; LOCAL, gitignored; read fresh on every cmux CLI call and passed through without ever overriding an operator's own ambient CMUX_SOCKET_PASSWORD when absent (docs/cmux-backend.md "Setup") config/wedge-alarm optional away-mode wedge-alarm active-alert directives; LOCAL, gitignored; absent means auto (macOS Notification Center when available); see docs/wedge-alarm.md +config/better-stack-incidents optional presence flag for the home-scoped Better Stack incident poll; LOCAL, gitignored, and not inherited; see docs/configuration.md "Better Stack incident monitoring" config/x-mode.env generated X-mode watcher cadence; LOCAL, gitignored; source before arming watcher when present data/ personal fleet records; LOCAL, gitignored as a whole backlog.md task queue, dependencies, history @@ -101,6 +102,8 @@ state/ volatile runtime signals; gitignored .pr-check-migration.log private per-task outcomes distinguishing rebuilt or canonically registered replacement polls, quarantined unarmed polls, and incomplete migrations .pr-check-migration-scan-v1 private marker proving the non-executing scan disabled every unsafe legacy check; .pr-check-migration-v1 separately records completed private repairs x-watch.check.sh generated X-mode relay poll shim; present only when opted in (section 14) + better-stack-incidents.check.sh better-stack-incidents.check-trust generated and registered home-scoped Better Stack incident poll; present only when opted in + better-stack-incidents.seen/ better-stack-incidents.diagnostics/ private incident-ID and diagnostic dedupe state retained across poll disable/re-enable pending-replies/ parent-owned secondmate pending-reply records (correlation id, delivery vs reply, recovery, escalation); fm-pending-reply-lib.sh x-inbox/ generated X-mode pending mention payloads; fmx-respond drains it (section 14) x-context/ generated X-mode durable per-request reply context and one-wake offer markers, keyed by request_id; survives inbox cleanup and expires within seven days (section 14; bin/fm-x-lib.sh) @@ -138,7 +141,7 @@ A lock-refused session must not spawn, steer, merge, drain the wake queue, repai 1. **Lock** - acquires the per-home session lock first, before anything mutates shared state. 2. **Bootstrap** - detect-only checks (tool/version problems, GitHub auth, the worktree-tangle check, harness override, dispatch-profile validation, backlog-backend status) always run, but routine confirmations stay silent by default. When the lock could not be acquired, the worktree-tangle check uses read-only advisory wording without a checkout repair command. - Home-local stale Herdr projection cleanup and the five bootstrap MUTATING sweeps - non-executing legacy PR-check migration, fleet sync, the local secondmate fast-forward sweep, the secondmate liveness sweep, and X-mode artifact writes - run only when this session actually holds the lock from step 1. + Home-local stale Herdr projection cleanup and the six bootstrap MUTATING sweeps - non-executing legacy PR-check migration, fleet sync, the local secondmate fast-forward sweep, the secondmate liveness sweep, X-mode artifact writes, and Better Stack incident-poll registration - run only when this session actually holds the lock from step 1. The secondmate liveness sweep deterministically accounts for every registered secondmate: it relaunches only from the recovery-grade `dead` or `missing` states, preserves ambiguous or unreadable targets, and reports skipped or failed guarantees as `SECONDMATE_LIVENESS:` lines (`bin/fm-bootstrap.sh`; `bin/fm-backend.sh`'s `fm_backend_agent_state`). 3. **Wake queue** - when locked, drains the durable wake queue and prints the raw records prominently as this turn's first work queue; a bounded, clearly labeled historical status-event annotation may follow a valid `signal` record but never replaces it or current-state reconciliation, and a lapsed watcher chain still surfaces here via the same guard alarm. When the lock could not be acquired and verified, the queue is left untouched because no session mutation is authorized, and the guard's tangle/watcher-liveness alarms still print in read-only advisory mode without drain, supervision repair, or checkout repair commands. @@ -147,7 +150,7 @@ A lock-refused session must not spawn, steer, merge, drain the wake queue, repai 5. **Fleet-state digest** - the compact backlog listing owned by `bin/fm-session-start.sh`; every `state/.meta`; a bounded tail of each task's `state/.status` (labeled as wake-EVENT history, not current state, with the full log path printed for a deeper read); the `state/.afk` flag; and one cheap alive/dead read of each task's recorded backend endpoint. That liveness line is a fast presence check only, not a full state read - when you need a crew's actual current state (a run-step, not just "is the pane there"), read it with `bin/fm-crew-state.sh ` as before; the digest deliberately skips that deeper, slower read for every task so it stays fast and bounded. 6. **Supervision operating instructions and next step** - after the wake queue and before context, the digest emits exactly one operating block for the detected primary harness. - The closing reminder points back to that emitted block and preserves only the lock, afk, X-mode, and read-once reminders. + The closing reminder points back to that emitted block and preserves only the lock, afk, home-monitoring, and read-once reminders. The script itself never starts supervision; the emitted harness protocol owns the exact wait or wake mechanism. Bootstrap detects first, asks for consent, and installs only after the captain approves in the current session. @@ -337,7 +340,7 @@ The promoted worker must inventory scratch state, return to a clean default-bran Fleet supervision is an always-loaded operational contract; `docs/architecture.md`, `docs/turnend-guard.md`, the emitted session-start block, and script help own mechanisms and harness-specific recipes. Whenever work is under way, keep exactly one live supervision cycle using the emitted protocol for this primary harness. -X mode may require that same live cycle with no fleet work. +X mode or Better Stack incident monitoring may require that same live cycle with no fleet work. Do not substitute another harness's wait shape, use shell `&`, or create a second cycle when a healthy one already exists. For every actionable wake, follow the ordinary-wake continuation in the emitted protocol; use its repair action only when the live cycle is missing or failed. No turn ends blind while work is under way, including turns described as holding or waiting. @@ -351,9 +354,14 @@ Handle actionable wakes as follows: 1. For `signal:`, read the listed event lines first, then reconcile current state only where action depends on it. 2. For `stale:`, inspect the recorded endpoint and load `stuck-crewmate-recovery` for a stopped, looping, confused, or unresponsive worker; a deep-inspection reason also requires current-state and validation-log inspection. -3. For `check:`, act on the named poll result, including merges and X-mode events. +3. For `check:`, act on the named poll result, including merges, X-mode events, and Better Stack incidents or diagnostics. 4. For `heartbeat:`, review the whole fleet from the structured fleet view, reconcile suspicious tasks and PR state, update the backlog, and never report an unchanged fleet as progress. +For a `better-stack-incident opened ...` or `better-stack-incidents opened ...` result, load `diagnostic-reasoning` before scoping the response. +When the delivery ladder caused the breakage, restore service through the available rollback path first and investigate after recovery. +Page the captain immediately only for security-shaped, irreversible, or product-affecting incidents; otherwise carry the result and resolution in the next outcome digest. +Better Stack polling during `heartbeat:` handling was an interim practice and is retired; the registered home check is its only poll owner. + When any wake reports a merged PR for a project cloned in this home, refresh that clone through the guarded fleet-sync path. When X-linked work reaches a milestone or terminal state, load `fmx-respond`; before terminal teardown, always post the final completion follow-up so the link clears even if earlier follow-ups were spent. @@ -477,7 +485,7 @@ It performs guarded fast-forward updates of firstmate and registered secondmate These skills are not captain-invocable; load them only at their precise triggers. -- `bootstrap-diagnostics` - load whenever the session-start digest's bootstrap section prints an actionable diagnostic line (`MISSING:`, `MISSING_MANUAL:`, `BACKEND_INVALID:`, `NEEDS_GH_AUTH`, `TANGLE:`, `STARTUP_MEMORY_BUDGET:`, `CREW_DISPATCH: invalid`, `FLEET_SYNC:`, `PR_CHECK_MIGRATION:`, `SECONDMATE_SYNC:`, `SECONDMATE_LIVENESS:`, `NUDGE_SECONDMATES:`, or `FMX:`); silence and `BOOTSTRAP_INFO:` need no load. +- `bootstrap-diagnostics` - load whenever the session-start digest's bootstrap section prints an actionable diagnostic line (`MISSING:`, `MISSING_MANUAL:`, `BACKEND_INVALID:`, `NEEDS_GH_AUTH`, `TANGLE:`, `STARTUP_MEMORY_BUDGET:`, `CREW_DISPATCH: invalid`, `FLEET_SYNC:`, `PR_CHECK_MIGRATION:`, `SECONDMATE_SYNC:`, `SECONDMATE_LIVENESS:`, `NUDGE_SECONDMATES:`, `FMX:`, or `BETTER_STACK:`); silence and `BOOTSTRAP_INFO:` need no load. - `diagnostic-reasoning` - load before scoping a reported bug and before acting on a diagnostic report. - `ask-user-authority` - load before deciding any ask-user finding, regardless of the project's `yolo` posture. - `quota-array-dispatch` - load before choosing among a matched crew-dispatch profile array from current quota-axi output. @@ -499,7 +507,7 @@ X mode ships inert and causes no behavior change until the home opts in by placi That token is consent for public replies and normal reversible lifecycle actions from eligible mentions, not authority for destructive, irreversible, or security-sensitive action; those still require trusted-channel confirmation. `docs/configuration.md` owns activation, generated state, cadence, wire protocol, and opt-out mechanics. -An X-only home still requires the live supervision cycle so mentions can wake it without fleet work. +A home with X mode or Better Stack incident monitoring still requires the live supervision cycle without fleet work. On an `x-mention ` or `x-mode-error ...` check wake, load `fmx-respond`, which owns classification, public-safety policy, reply or dismissal, task linking, and follow-ups. For every X-linked terminal outcome, load that owner and post the final completion follow-up before teardown, regardless of earlier milestone follow-ups. diff --git a/README.md b/README.md index 726891f188d..161c1253fd9 100644 --- a/README.md +++ b/README.md @@ -48,6 +48,7 @@ Launching a supported harness inside it instantiates your first mate - and makes - **Explicit project modes** - each project ships via `no-mistakes`, `direct-PR`, or `local-only`, with an optional `+yolo` autonomy flag. - **Optional secondmates** - opt in to persistent second mates that run from isolated firstmate homes with their own `FM_HOME`, state, projects, and session lock, supervising project clones or a project-less firstmate-repo domain, kept on the primary firstmate version by guarded local fast-forwards and checked for live agent processes at session start. - **Event-driven, zero-token supervision** - a bash watcher sleeps on the fleet and wakes the first mate only when something needs you; verified primary harnesses also get a turn-end backstop that blocks or follows up on a blind stop when work is under way and supervision is not live. +- **Optional Better Stack incident monitoring** - opt one home into unresolved-incident polling through the registered custom-check path; runtime-only Doppler injection, private incident dedupe, and one visible diagnostic per failure keep the alert path bounded. - **Optional X mode** - opt in with one local `.env` token so firstmate can answer your public `@myfirstmate` mentions, act on normal reversible mention requests through the same lifecycle as chat requests, acknowledge spawned work, and post up to three public-safe completion follow-ups within seven days for genuine milestones and the final outcome without changing non-X behavior; dry-run preview records would-be replies and dismissals locally before go-live. - **Strict project boundary** - the first mate is read-only over your projects except for the narrow guarded and captain-approved operations authorized by [hard rule 1](AGENTS.md#1-identity-and-prime-directives), including fleet sync's guarded safe branch pruning; crewmates make every other project change behind the configured merge authority. - **Doppler by default** - the conditional [secrets-management policy](.agents/skills/secrets-management/SKILL.md) prefers secretless provider identity, otherwise scopes Doppler by project and environment, and validates declarations and rollout data through `bin/fm-secrets-check.sh`. diff --git a/bin/fm-better-stack-incidents-poll.sh b/bin/fm-better-stack-incidents-poll.sh new file mode 100755 index 00000000000..0ac4bc29d89 --- /dev/null +++ b/bin/fm-better-stack-incidents-poll.sh @@ -0,0 +1,152 @@ +#!/usr/bin/env bash +# Poll Better Stack for unresolved incidents through the fleet-observability/prd +# Doppler config. +# +# Usage: fm-better-stack-incidents-poll.sh +# +# The public entrypoint always launches itself through Doppler with fallback +# files disabled and only BETTER_STACK_API_TOKEN injected. +# The internal --from-doppler mode is used only by that child process and tests. +# +# Output is the authenticated custom-check contract consumed by fm-watch.sh: +# better-stack-incident opened id= name= started= +# better-stack-incidents opened ids= +# better-stack-error +# A quiet or already-seen result prints nothing. +# The watcher provides the outer FM_CHECK_TIMEOUT; curl stays within five +# seconds so the check finishes with margin. +set -u + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" +FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" +STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" +ERROR_DIR="$STATE/better-stack-incidents.diagnostics" +ERROR_FILE="$ERROR_DIR/error" + +# Reuse the watcher's existing private-artifact owner rather than introducing a +# second atomic-publication contract for one extension. +# shellcheck source=bin/fm-x-lib.sh disable=SC1091 +. "$SCRIPT_DIR/fm-x-lib.sh" + +emit_error_once() { + local msg=$1 + if fmx_private_artifact_file_valid "$ERROR_DIR" error 600 \ + && [ "$(cat "$ERROR_FILE" 2>/dev/null)" = "$msg" ]; then + return 0 + fi + printf '%s\n' "$msg" \ + | fmx_private_artifact_publish_stdin "$ERROR_DIR" error 600 2>/dev/null || true + printf 'better-stack-error %s\n' "$msg" +} + +clear_error() { + fmx_private_artifact_file_valid "$ERROR_DIR" error 600 || return 0 + rm -f -- "$ERROR_FILE" 2>/dev/null || true +} + +run_through_doppler() { + local out rc + command -v doppler >/dev/null 2>&1 \ + || { emit_error_once "missing doppler"; return 0; } + out=$(doppler run \ + --silent \ + --no-check-version \ + --no-fallback \ + --project fleet-observability \ + --config prd \ + --only-secrets BETTER_STACK_API_TOKEN \ + -- "$SCRIPT_DIR/fm-better-stack-incidents-poll.sh" --from-doppler 2>/dev/null) + rc=$? + if [ "$rc" -ne 0 ]; then + emit_error_once "Doppler access unavailable for fleet-observability/prd" + return 0 + fi + case "$out" in + '') return 0 ;; + *) + while IFS= read -r line; do + case "$line" in + better-stack-incident\ opened\ *|better-stack-error\ *) printf '%s\n' "$line" ;; + *) emit_error_once "poll returned invalid output"; return 0 ;; + esac + done <<< "$out" + ;; + esac +} + +poll_with_injected_token() { + local token=${BETTER_STACK_API_TOKEN:-} raw code body page_rows page_next + local id name started next_url='https://uptime.betterstack.com/api/v3/incidents?resolved=false&per_page=50' + local budget=${FM_CHECK_TIMEOUT:-30} started_at=$SECONDS elapsed remaining curl_timeout + + [ -n "$token" ] || { emit_error_once "missing BETTER_STACK_API_TOKEN"; return 0; } + [[ "$token" =~ ^[A-Za-z0-9._~+/=-]+$ ]] \ + || { emit_error_once "invalid BETTER_STACK_API_TOKEN"; return 0; } + command -v curl >/dev/null 2>&1 || { emit_error_once "missing curl"; return 0; } + command -v jq >/dev/null 2>&1 || { emit_error_once "missing jq"; return 0; } + + rows= + while [ -n "$next_url" ]; do + elapsed=$((SECONDS - started_at)) + remaining=$((budget - elapsed - 1)) + [ "$remaining" -gt 0 ] || { emit_error_once "Better Stack API poll timed out"; return 0; } + curl_timeout=$remaining + [ "$curl_timeout" -gt 5 ] && curl_timeout=5 + raw=$(printf 'header = "Authorization: Bearer %s"\n' "$token" \ + | curl --config - --request GET --url "$next_url" \ + --header 'Accept: application/json' --connect-timeout 3 \ + --max-time "$curl_timeout" --silent --show-error \ + --write-out '\n%{http_code}' 2>/dev/null) \ + || { emit_error_once "Better Stack API unreachable"; return 0; } + case "$raw" in + *$'\n'*) ;; + *) emit_error_once "Better Stack API returned no status"; return 0 ;; + esac + code=${raw##*$'\n'} + body=${raw%$'\n'*} + [ "$code" = 200 ] || { emit_error_once "API returned HTTP $code"; return 0; } + page_rows=$(printf '%s' "$body" | jq -r ' + if (.data | type) != "array" then error("data must be an array") else .data[] end + | select(.type == "incident") + | select(.attributes.resolved_at == null) + | select(.id | type == "string" and test("^[0-9]+$")) + | [ + .id, + ((.attributes.name // "unknown") | tostring | gsub("[[:space:][:cntrl:]]+"; " ") | .[0:120]), + ((.attributes.started_at // "unknown") | tostring | gsub("[[:space:][:cntrl:]]+"; " ") | .[0:64]) + ] + | @tsv + ' 2>/dev/null) || { emit_error_once "invalid Better Stack API response"; return 0; } + [ -z "$rows" ] || [ -z "$page_rows" ] || rows="$rows"$'\n' + rows="$rows$page_rows" + page_next=$(printf '%s' "$body" | jq -r ' + if ((.pagination.next // null) != null and (.pagination.next | type) != "string") + then error("pagination.next must be a string") + else (.pagination.next // "") end + ' 2>/dev/null) || { emit_error_once "invalid Better Stack API response"; return 0; } + case "$page_next" in + '') next_url= ;; + https://uptime.betterstack.com/api/v3/incidents\?*) + case "$page_next" in *resolved=false*) next_url=$page_next ;; *) emit_error_once "invalid Better Stack pagination target"; return 0 ;; esac + ;; + *) emit_error_once "invalid Better Stack pagination target"; return 0 ;; + esac + done + + while IFS=$'\t' read -r id name started; do + [ -n "$id" ] || continue + printf 'better-stack-incident opened id=%s name=%s started=%s\n' "$id" "$name" "$started" + done <<< "$rows" + + clear_error +} + +case "${1:-}" in + '') run_through_doppler ;; + --from-doppler) + [ "$#" -eq 1 ] || { emit_error_once "invalid poll invocation"; exit 0; } + poll_with_injected_token + ;; + *) emit_error_once "invalid poll invocation" ;; +esac diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index 16102adfa45..2f41032f165 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -17,7 +17,9 @@ # "NUDGE_SECONDMATES: secondmate : send failed: ", # "BOOTSTRAP_INFO: nudged fm- with ''", # "SECONDMATE_LIVENESS: secondmate : skipped: |respawn failed after : ", -# "FMX: X mode on ..." or "FMX: X mode off ...". +# "FMX: X mode on ..." or "FMX: X mode off ...", +# "BETTER_STACK: incident monitoring on ..." or +# "BETTER_STACK: incident monitoring off ...". # When a RUNNING secondmate worktree is fast-forwarded to firstmate's # own current default-branch commit (a purely LOCAL fast-forward, never # an origin fetch) AND its loaded instruction surface (AGENTS.md, bin/, @@ -73,15 +75,16 @@ # refresh relays any completed fm-fleet-sync.sh output before the # aggregate timeout skip line with timeout and elapsed seconds. # Set FM_FLEET_PRUNE=0 to skip branch pruning during that refresh. -# Set FM_BOOTSTRAP_DETECT_ONLY=1 to skip the five MUTATING sweeps +# Set FM_BOOTSTRAP_DETECT_ONLY=1 to skip the six MUTATING sweeps # (PR-check migration, secondmate_sync, secondmate_liveness_sweep, -# x_mode_setup, fleet_sync) while still printing every read-only detect line +# x_mode_setup, better_stack_incidents_setup, fleet_sync) while still +# printing every read-only detect line # above; the TANGLE line switches to advisory-only wording with no # checkout command. Used by # fm-session-start.sh's read-only path when another live session holds # the fleet lock, so a second concurrent session never race-mutates -# PR-check artifacts, secondmate homes, X-mode artifacts, project -# clones, or repair instructions. +# PR-check artifacts, secondmate homes, X-mode or Better Stack poll +# artifacts, project clones, or repair instructions. # Unset/0 (the default) runs every sweep exactly as before - this flag # is purely additive. # fm-bootstrap.sh install ... @@ -107,6 +110,10 @@ DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" . "$SCRIPT_DIR/fm-startup-memory-budget-lib.sh" # shellcheck source=bin/fm-x-lib.sh disable=SC1091 . "$SCRIPT_DIR/fm-x-lib.sh" +# shellcheck source=bin/fm-pr-lib.sh disable=SC1091 +. "$SCRIPT_DIR/fm-pr-lib.sh" +# shellcheck source=bin/fm-check-lib.sh disable=SC1091 +. "$SCRIPT_DIR/fm-check-lib.sh" # shellcheck source=bin/fm-backend.sh disable=SC1091 . "$SCRIPT_DIR/fm-backend.sh" @@ -503,6 +510,7 @@ install_cmd() { manual_install_url() { case "$1" in + doppler) echo "https://docs.doppler.com/docs/install-cli" ;; herdr) echo "https://herdr.dev" ;; *) return 1 ;; esac @@ -557,7 +565,7 @@ no_mistakes_compatible() { [ "$patch" -ge "$NO_MISTAKES_MIN_PATCH" ] } -x_mode_write_if_changed() { +bootstrap_write_if_changed() { local dest=$1 content=$2 mode=$3 parent tmp parent_device current_mode parent=${dest%/*} [ "$parent" != "$dest" ] || return 1 @@ -578,7 +586,7 @@ x_mode_write_if_changed() { return 0 fi fi - tmp=$(umask 077; mktemp "$parent/.fm-x-mode.XXXXXX" 2>/dev/null) || return 1 + tmp=$(umask 077; mktemp "$parent/.fm-bootstrap-artifact.XXXXXX" 2>/dev/null) || return 1 if ! printf '%s\n' "$content" > "$tmp" \ || ! chmod "$mode" "$tmp" \ || ! fmx_single_link_file_mode_valid "$tmp" "$mode" "$parent_device"; then @@ -699,7 +707,7 @@ x_mode_setup() { ;; esac shim_body=$(fmx_poll_shim_content "$shim_home" "$FM_ROOT") - x_mode_write_if_changed "$shim" "$shim_body" 700 || { fmx_arm_failed; return 0; } + bootstrap_write_if_changed "$shim" "$shim_body" 700 || { fmx_arm_failed; return 0; } fmx_poll_shim_valid "$shim" "$shim_home" "$FM_ROOT" \ || { fmx_arm_failed; return 0; } @@ -711,11 +719,97 @@ x_mode_setup() { export FM_CHECK_INTERVAL=30 EOF ) - x_mode_write_if_changed "$cadence" "$cadence_body" 600 || { fmx_arm_failed; return 0; } + bootstrap_write_if_changed "$cadence" "$cadence_body" 600 || { fmx_arm_failed; return 0; } echo "FMX: X mode on - relay poll armed via state/x-watch.check.sh; 30s watcher cadence in config/x-mode.env" } +# Better Stack incident monitoring is an explicit home-local opt-in. +# A presence flag at config/better-stack-incidents materializes one ordinary +# registered custom check, so the existing hash-bound snapshot execution and +# FM_CHECK_TIMEOUT contract remain the only slow-check mechanism. +# The poll itself performs runtime-only Doppler injection; the watcher owns +# incident delivery dedupe while the poll owns diagnostic dedupe. +better_stack_incidents_setup() { + local flag check trust check_body tool missing check_home failed + flag="$CONFIG/better-stack-incidents" + check="$STATE/better-stack-incidents.check.sh" + trust="$STATE/better-stack-incidents.check-trust" + + better_stack_remove_artifacts() { + local remove_failed=0 + x_mode_remove_artifact "$check" || remove_failed=1 + x_mode_remove_artifact "$trust" || remove_failed=1 + [ "$remove_failed" -eq 0 ] + } + + if [ ! -e "$flag" ] && [ ! -L "$flag" ]; then + if x_mode_artifact_present "$check" || x_mode_artifact_present "$trust"; then + if better_stack_remove_artifacts; then + echo "BETTER_STACK: incident monitoring off - removed the home-scoped poll" + else + echo "BETTER_STACK: incident monitoring off - failed to remove the home-scoped poll" + fi + fi + return 0 + fi + if [ ! -f "$flag" ] || [ -L "$flag" ] || [ "$(fm_pr_file_link_count "$flag")" != 1 ]; then + better_stack_remove_artifacts || true + echo "BETTER_STACK: incident monitoring off - config/better-stack-incidents must be an ordinary file" + return 0 + fi + + missing=0 + for tool in doppler curl jq; do + if ! command -v "$tool" >/dev/null 2>&1; then + missing_tool_diagnostic "$tool" + missing=1 + fi + done + if [ "$missing" -ne 0 ]; then + better_stack_remove_artifacts || true + echo "BETTER_STACK: incident monitoring off - install the reported poll dependencies and rerun bootstrap" + return 0 + fi + + failed=0 + mkdir -p "$STATE" 2>/dev/null || failed=1 + if [ "$failed" -eq 0 ]; then + case "$FM_HOME" in + /*) check_home=$FM_HOME ;; + *) + check_home=$(CDPATH='' cd -- "$FM_HOME" 2>/dev/null && pwd -P) || failed=1 + ;; + esac + fi + if [ "$failed" -eq 0 ]; then + check_body=$(printf '%s\n' \ + '#!/usr/bin/env bash' \ + '# Auto-generated by fm-bootstrap.sh - Better Stack incident custom check.' \ + '# Registered bytes call the tracked poll; output becomes a check: wake.' \ + "export FM_HOME=$(printf '%q' "$check_home")" \ + "exec $(printf '%q' "$FM_ROOT/bin/fm-better-stack-incidents-poll.sh")") + bootstrap_write_if_changed "$check" "$check_body" 700 || failed=1 + fi + if [ "$failed" -eq 0 ] && ! fm_custom_check_registered "$STATE" better-stack-incidents; then + FM_HOME="$FM_HOME" FM_STATE_OVERRIDE="$STATE" FM_ROOT_OVERRIDE="$FM_ROOT" \ + "$SCRIPT_DIR/fm-check-register.sh" better-stack-incidents >/dev/null 2>&1 || failed=1 + fi + if [ "$failed" -eq 0 ]; then + fm_custom_check_registered "$STATE" better-stack-incidents || failed=1 + fi + if [ "$failed" -ne 0 ]; then + if better_stack_remove_artifacts; then + echo "BETTER_STACK: incident monitoring off - failed to arm the home-scoped poll" + else + echo "BETTER_STACK: incident monitoring off - failed to arm the home-scoped poll; stale artifacts remain" + fi + return 0 + fi + + echo "BETTER_STACK: incident monitoring on - registered state/better-stack-incidents.check.sh at the default 300s check cadence" +} + crew_dispatch_validate() { local file err file="$CONFIG/crew-dispatch.json" @@ -900,6 +994,7 @@ if [ "${FM_BOOTSTRAP_DETECT_ONLY:-0}" != 1 ]; then secondmate_liveness_sweep secondmate_sync x_mode_setup + better_stack_incidents_setup fleet_sync fi exit 0 diff --git a/bin/fm-claude-stop-autoarm.sh b/bin/fm-claude-stop-autoarm.sh index df9ee1128fc..0c2c4fbf528 100755 --- a/bin/fm-claude-stop-autoarm.sh +++ b/bin/fm-claude-stop-autoarm.sh @@ -18,8 +18,8 @@ # - AFK: while state/.afk exists the away daemon owns the watcher and triage; # this hook exits 0 and NEVER rewakes the primary (checked again at # translation time so a mid-cycle AFK transition is honored). -# - Need: arms only while work is in flight (state/*.meta) or X mode has a -# relay poll to run (state/x-watch.check.sh); an idle home exits 0. +# - Need: arms only while work is in flight (state/*.meta) or a home-level +# X-mode or Better Stack poll is active; an idle home exits 0. # - Single-flight: Claude does not dedupe async hooks, so a home-scoped owner # lock (state/.claude-autoarm.lock) admits exactly one owner; every other # concurrent firing exits 0 without translating, which keeps one event @@ -89,7 +89,7 @@ fi # --- AFK: the away daemon owns the watcher and triage; never rewake ---------- [ -e "$STATE/.afk" ] && exit 0 -# --- need: in-flight work or an X-mode relay poll ---------------------------- +# --- need: in-flight work or a home-level poll ------------------------------- need_supervision() { fm_supervision_needed "$STATE" "$GRACE" } diff --git a/bin/fm-session-start.sh b/bin/fm-session-start.sh index 1abbace4bf1..85ace7fa29f 100755 --- a/bin/fm-session-start.sh +++ b/bin/fm-session-start.sh @@ -18,7 +18,7 @@ # standalone with unchanged default behavior - other flows (fm-bootstrap.sh # install after consent, /updatefirstmate, the afk daemon, existing # tests) still call them directly. The one seam this script needed - -# bootstrap running its detect-only diagnostics without its five mutating +# bootstrap running its detect-only diagnostics without its six mutating # sweeps - is an opt-in FM_BOOTSTRAP_DETECT_ONLY=1 flag on fm-bootstrap.sh # itself (default unset/0 = unchanged behavior), not a fork. # @@ -29,9 +29,10 @@ # mutating step runs. # 2. bootstrap - home-local stale Herdr projection cleanup runs only # when this session actually holds the lock. Detect-only -# diagnostics always run. Bootstrap's five MUTATING sweeps +# diagnostics always run. Bootstrap's six MUTATING sweeps # (legacy PR-check migration, secondmate fast-forward, -# secondmate liveness, X-mode artifact writes, fleet sync) +# secondmate liveness, X-mode artifact writes, Better Stack +# incident-poll registration, fleet sync) # also run only when locked. # 3. wake-drain - mutates the durable wake queue, so it also only runs # when locked. @@ -52,7 +53,8 @@ # # Why lock first: the old documented order (bootstrap, THEN lock) let a # SECOND concurrent session run bootstrap's mutating sweeps - fast-forwarding -# secondmate homes, writing X-mode artifacts, fetching/fast-forwarding every +# secondmate homes, writing X-mode and Better Stack poll artifacts, +# fetching/fast-forwarding every # project clone - before ever discovering another session already holds the # lock. Two sessions racing those sweeps is exactly the hazard the lock # exists to prevent, so locking first closes the hole outright: only the @@ -65,7 +67,7 @@ # tasks-axi and quota-axi tool checks, and tasks-axi availability - none of # which mutate shared state and all of which are safe to compute without # verified lock ownership. -# Only projection cleanup, the five bootstrap mutating sweeps, and the +# Only projection cleanup, the six bootstrap mutating sweeps, and the # wake-queue drain are skipped. # The context and fleet-state digests # below are always read-only, so they run unconditionally in both modes. @@ -257,7 +259,7 @@ if [ "$LOCK_RC" -ne 0 ]; then printf '● READ-ONLY SESSION - FLEET LOCK OWNERSHIP WAS NOT VERIFIED\n' printf '● %s\n' "$LOCK_OUT" printf '● Skipping every mutating step: PR-check migration, stale Herdr child cleanup,\n' - printf '● secondmate sync, X-mode artifacts, fleet sync, and wake-queue drain. Detect-only bootstrap\n' + printf '● secondmate sync, X-mode and Better Stack poll artifacts, fleet sync, and wake-queue drain. Detect-only bootstrap\n' printf '● diagnostics and the rest of this read-only-safe digest still ran below.\n' printf '● Operate read-only until this resolves - do not spawn, steer, merge, or\n' printf '● otherwise mutate fleet state from this session.\n' diff --git a/bin/fm-subagent-pretool-check.sh b/bin/fm-subagent-pretool-check.sh index 8edb507218b..95a0c6be434 100755 --- a/bin/fm-subagent-pretool-check.sh +++ b/bin/fm-subagent-pretool-check.sh @@ -6,7 +6,7 @@ # no `data//brief.md`. Only `bin/fm-spawn.sh` writes that metadata, and # untracked project work contributes nothing to the in-flight branch of # bin/fm-supervision-lib.sh or bin/fm-turnend-guard.sh. So such work is not -# merely unsupervised: absent an independent X-mode need, it makes the whole +# merely unsupervised: absent an independent home-monitoring need, it makes the whole # guard stack structurally inert, and it dies with the primary session instead # of living in its own backend session. # diff --git a/bin/fm-supervision-lib.sh b/bin/fm-supervision-lib.sh index 1930700d2af..b7ff73d0158 100644 --- a/bin/fm-supervision-lib.sh +++ b/bin/fm-supervision-lib.sh @@ -3,9 +3,9 @@ # Usage: . bin/fm-supervision-lib.sh # # Reports whether a firstmate home needs supervision because it has in-flight -# work (a state/.meta exists) or an X-mode relay poll -# (state/x-watch.check.sh), and whether its watcher has a fresh liveness beacon -# (state/.last-watcher-beat, touched every poll cycle, within the grace window). +# work (a state/.meta exists) or a home-level poll (X mode or Better Stack), +# and whether its watcher has a fresh liveness beacon (state/.last-watcher-beat, +# touched every poll cycle, within the grace window). # bin/fm-guard.sh keeps its task-specific grace-based warning predicate; # bin/fm-turnend-guard.sh uses the status fields here for its banner but performs # its end-of-turn block decision with the live watcher lock check in @@ -23,7 +23,7 @@ fm_sup_stat_mtime() { # fm_supervision_status [grace-seconds] # Populates, for the state dir at $1: # FM_SUP_IN_FLIGHT count of state/*.meta (in-flight tasks) -# FM_SUP_NEEDED true/false - in-flight work or an X-mode relay poll +# FM_SUP_NEEDED true/false - in-flight work or a home-level poll # FM_SUP_WATCHER_FRESH true/false - a watcher beacon within the grace window # FM_SUP_BEACON_DESC human-readable beacon age, for banners ("never" if absent) # FM_SUP_QUEUE_PENDING true/false - state/.wake-queue has unread records @@ -41,7 +41,9 @@ fm_supervision_status() { [ -e "$meta" ] || continue FM_SUP_IN_FLIGHT=$((FM_SUP_IN_FLIGHT + 1)) done - if [ "$FM_SUP_IN_FLIGHT" -gt 0 ] || [ -f "$state/x-watch.check.sh" ]; then + if [ "$FM_SUP_IN_FLIGHT" -gt 0 ] \ + || [ -f "$state/x-watch.check.sh" ] \ + || [ -f "$state/better-stack-incidents.check.sh" ]; then FM_SUP_NEEDED=true fi @@ -64,8 +66,8 @@ fm_supervision_status() { } # fm_supervision_needed [grace-seconds] -# Exit 0 (true) exactly when in-flight work or an X-mode relay poll needs a -# watcher. Exit 1 (false) for an idle home. +# Exit 0 (true) exactly when in-flight work or a home-level poll needs a watcher. +# Exit 1 (false) for an idle home. fm_supervision_needed() { fm_supervision_status "$@" [ "$FM_SUP_NEEDED" = true ] diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 14bc0748260..b4b4830c1b1 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -131,7 +131,7 @@ family_for_basename() { fm-test-run.test.sh|fm-test-isolation-proof.test.sh|fm-toolchain-mirror.test.sh) printf '%s\n' pure-contract-unit ;; - fm-daemon.test.sh|fm-guard-stale-banner.test.sh|fm-pi-watch-extension.test.sh|\ + fm-better-stack-incidents.test.sh|fm-daemon.test.sh|fm-guard-stale-banner.test.sh|fm-pi-watch-extension.test.sh|\ fm-supervision-events.test.sh|fm-turnend-guard.test.sh|fm-wake-daemon-lifecycle-e2e.test.sh|\ fm-wake-queue.test.sh|fm-watch-checkpoint.test.sh|fm-watch-triage.test.sh|\ fm-watcher-lock.test.sh) diff --git a/bin/fm-turnend-guard.sh b/bin/fm-turnend-guard.sh index 2e96fb33e48..d9bdc72fff6 100755 --- a/bin/fm-turnend-guard.sh +++ b/bin/fm-turnend-guard.sh @@ -164,6 +164,10 @@ block_stop() { printf '● TURN WOULD END BLIND - SUPERVISION IS OFF\n' if [ "$FM_SUP_IN_FLIGHT" -gt 0 ]; then printf '● %s task(s) in flight, but no live watcher holds this home lock (last beat: %s).\n' "$FM_SUP_IN_FLIGHT" "$FM_SUP_BEACON_DESC" + elif [ -f "$STATE/better-stack-incidents.check.sh" ] && [ -f "$STATE/x-watch.check.sh" ]; then + printf '● X-mode relay and Better Stack incident monitoring need supervision, but no live watcher holds this home lock (last beat: %s).\n' "$FM_SUP_BEACON_DESC" + elif [ -f "$STATE/better-stack-incidents.check.sh" ]; then + printf '● Better Stack incident monitoring needs supervision, but no live watcher holds this home lock (last beat: %s).\n' "$FM_SUP_BEACON_DESC" else printf '● X-mode relay polling needs supervision, but no live watcher holds this home lock (last beat: %s).\n' "$FM_SUP_BEACON_DESC" fi @@ -229,6 +233,10 @@ if [ "$COUNT" -gt "$BLOCK_BUDGET" ]; then budget_reset if [ "$FM_SUP_IN_FLIGHT" -gt 0 ]; then NEED_DESC="$FM_SUP_IN_FLIGHT task(s) in flight" + elif [ -f "$STATE/better-stack-incidents.check.sh" ] && [ -f "$STATE/x-watch.check.sh" ]; then + NEED_DESC="X-mode relay and Better Stack incident monitoring active" + elif [ -f "$STATE/better-stack-incidents.check.sh" ]; then + NEED_DESC="Better Stack incident monitoring active" else NEED_DESC="X-mode relay polling active" fi diff --git a/bin/fm-wake-lib.sh b/bin/fm-wake-lib.sh index 8cec58bec1d..16a373b5791 100755 --- a/bin/fm-wake-lib.sh +++ b/bin/fm-wake-lib.sh @@ -406,6 +406,89 @@ fm_wake_append() { return "$status" } +fm_wake_incident_receipt_valid() { + local receipt=$1 id=$2 version receipt_id payload + local receipt_dir=${receipt%/*} base=${receipt##*/} + fmx_private_artifact_file_valid "$receipt_dir" "$base" 600 || return 1 + exec 9< "$receipt" || return 1 + IFS= read -r version <&9 || { exec 9<&-; return 1; } + IFS= read -r receipt_id <&9 || { exec 9<&-; return 1; } + IFS= read -r payload <&9 || { exec 9<&-; return 1; } + if IFS= read -r <&9; then + exec 9<&- + return 1 + fi + exec 9<&- + [ "$version" = fm-better-stack-incident-receipt-v1 ] || return 1 + [ "$receipt_id" = "$id" ] || return 1 + [ -n "$payload" ] +} + +fm_wake_append_incident_once() { + local id=$1 payload=$2 + local key="better-stack-incident:$id" + local seen_dir="$STATE/better-stack-incidents.seen" seen_file="$STATE/better-stack-incidents.seen/$id" + local receipt_dir="$STATE/better-stack-incidents.receipts" receipt="$STATE/better-stack-incidents.receipts/$id" + local epoch seq seq_file status=0 receipt_rc marker_rc + [[ "$id" =~ ^[0-9]+$ ]] || return 2 + declare -F fmx_private_artifact_file_valid >/dev/null 2>&1 || return 2 + declare -F fmx_private_artifact_publish_stdin_once >/dev/null 2>&1 || return 2 + fm_lock_acquire_wait "$FM_WAKE_QUEUE_LOCK" + if fmx_private_artifact_file_valid "$seen_dir" "$id" 600; then + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return 0 + elif [ -e "$seen_file" ] || [ -L "$seen_file" ]; then + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return 2 + fi + if fm_wake_incident_receipt_valid "$receipt" "$id"; then + printf 'seen\n' | fmx_private_artifact_publish_stdin_once "$seen_dir" "$id" 600 >/dev/null 2>&1 + marker_rc=$? + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + [ "$marker_rc" -eq 0 ] || [ "$marker_rc" -eq 1 ] || return 2 + return 0 + elif [ -e "$receipt" ] || [ -L "$receipt" ]; then + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return 2 + fi + if awk -F '\t' -v key="$key" '$3 == "check" && $4 == key { found=1 } END { exit !found }' "$FM_WAKE_QUEUE" 2>/dev/null; then + printf 'fm-better-stack-incident-receipt-v1\n%s\n%s\n' "$id" "$payload" \ + | fmx_private_artifact_publish_stdin_once "$receipt_dir" "$id" 600 >/dev/null 2>&1 + receipt_rc=$? + if [ "$receipt_rc" -ne 0 ] && [ "$receipt_rc" -ne 1 ]; then + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return 2 + fi + printf 'seen\n' | fmx_private_artifact_publish_stdin_once "$seen_dir" "$id" 600 >/dev/null 2>&1 + marker_rc=$? + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + [ "$marker_rc" -eq 0 ] || [ "$marker_rc" -eq 1 ] || return 2 + return 0 + fi + epoch=$(date +%s) + seq_file="$STATE/.wake-queue.seq" + seq=$(cat "$seq_file" 2>/dev/null || echo 0) + case "$seq" in ''|*[!0-9]*) seq=0 ;; esac + seq=$((seq + 1)) + printf '%s\n' "$seq" > "$seq_file" || status=$? + if [ "$status" -eq 0 ]; then + printf '%s\t%s\tcheck\t%s\t%s\n' "$epoch" "$seq" "$key" "$payload" >> "$FM_WAKE_QUEUE" || status=$? + fi + if [ "$status" -eq 0 ]; then + printf 'fm-better-stack-incident-receipt-v1\n%s\n%s\n' "$id" "$payload" \ + | fmx_private_artifact_publish_stdin_once "$receipt_dir" "$id" 600 >/dev/null 2>&1 + receipt_rc=$? + [ "$receipt_rc" -eq 0 ] || [ "$receipt_rc" -eq 1 ] || status=2 + fi + if [ "$status" -eq 0 ]; then + printf 'seen\n' | fmx_private_artifact_publish_stdin_once "$seen_dir" "$id" 600 >/dev/null 2>&1 + marker_rc=$? + [ "$marker_rc" -eq 0 ] || [ "$marker_rc" -eq 1 ] || status=2 + fi + fm_lock_release "$FM_WAKE_QUEUE_LOCK" + return "$status" +} + fm_wake_restore_queue() { local drained=$1 restore restore="$STATE/.wake-queue.restore.$(fm_current_pid)" diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index e5501f852b3..8c4df518379 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -780,8 +780,23 @@ while :; do fi fi if [ -n "$out" ]; then - reason="check: $c: $out" - fm_wake_append check "$c" "$reason" || exit 1 + if [ "$(basename "$c")" = better-stack-incidents.check.sh ]; then + while IFS= read -r incident_line; do + [ -n "$incident_line" ] || continue + case "$incident_line" in + better-stack-incident\ opened\ id=*) + id=${incident_line#better-stack-incident opened id=} + id=${id%% *} + reason="check: $c: $incident_line" + fm_wake_append_incident_once "$id" "$reason" || exit 1 + ;; + *) reason="check: $c: $incident_line"; fm_wake_append check "$c" "$reason" || exit 1 ;; + esac + done <<< "$out" + else + reason="check: $c: $out" + fm_wake_append check "$c" "$reason" || exit 1 + fi if [ "$is_pr_poll" -eq 1 ] && [ "$out" = merged ]; then if fm_pr_poll_retirement_publish "$STATE" "$id" "$SCRIPT_DIR/fm-pr-poll.sh" "$out"; then fm_pr_poll_retirement_recover_one "$STATE" "$id" "$SCRIPT_DIR/fm-pr-poll.sh" \ diff --git a/docs/architecture.md b/docs/architecture.md index bf8b5cb3ec1..4e7e82e661c 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -10,6 +10,9 @@ firstmate's always-loaded operating contract and routing index for conditional p A zero-token bash watcher (`bin/fm-watch.sh`) sleeps on the fleet, classifies detected wakes in bash, and wakes the first mate only when something is actionable. Actionable wakes include captain-relevant status signals, no-verb signals whose crew is not provably working, authenticated check output such as PR merge polling or an X-mode mention, stale panes whose crew is not provably working whether their status log looks terminal or non-terminal, provably-working stale panes that persist past `FM_STALE_ESCALATE_SECS`, declared external waits that remain paused past `FM_PAUSE_RESURFACE_SECS`, and heartbeat backstop hits. +Better Stack incident monitoring uses the registered custom-check extension rather than a new watcher scheduler because that existing path already supplies home scoping, hash-bound private snapshot execution, the slow-check timeout, and durable `check:` delivery. +The locked bootstrap materializes the check only for a home carrying `config/better-stack-incidents`, and [`configuration.md`](configuration.md#better-stack-incident-monitoring-configbetter-stack-incidents) owns runtime Doppler injection, incident-ID dedupe, diagnostic suppression, cadence, and opt-out mechanics. +The registered check remains a supervision need when the project fleet is idle, and it replaces the retired interim practice of querying Better Stack during heartbeat reviews. Repeated provably-working stale escalations on the same unchanged pane add an escalation count to the wake reason and, at `FM_WEDGE_DEMAND_INSPECT_COUNT`, a `demand-deep-inspection` marker. A busy pane is otherwise exempt from staleness, but only until its latest `state/.turn-ended` marker reaches `FM_BUSY_TURN_MAX_SECS`, or its `state/.meta` spawn record reaches that age before any turn completes; past that bound it is routed through the same wedge escalation, with the identical reason, escalation count, and `demand-deep-inspection` marker, for inspection only - never an automatic interrupt, signal, or restart. Those actionable wakes are written to a durable local queue (`state/.wake-queue`) before detector state advances, so a missed process exit can be recovered by draining the queue. diff --git a/docs/configuration.md b/docs/configuration.md index 94799d0e8ea..84c974391f5 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -11,7 +11,7 @@ The shared orchestrator behavior lives in [`AGENTS.md`](../AGENTS.md) - edit it This section is the single owner of the top-level operational-home layout; producer script headers and their help own exact child-file fields and mutation contracts. The tracked code root contains the shared instruction, skill, documentation, workflow, and `bin/` surfaces, while each effective `FM_HOME` contains private operational directories. `data/` holds durable private fleet records such as the project and secondmate registries, captain preferences, optional shared captain preferences, learnings, backlog, briefs, and scout reports. -`state/` holds volatile runtime records such as task metadata, append-only status events, endpoint signals, watcher and wake-queue coordination, away-mode state, generated X-mode artifacts, private secondmate config-reread generations with their retry and quarantine state, and parent-owned secondmate pending-reply records under `state/pending-replies/` (`bin/fm-pending-reply-lib.sh`). +`state/` holds volatile runtime records such as task metadata, append-only status events, endpoint signals, watcher and wake-queue coordination, away-mode state, generated X-mode and Better Stack poll artifacts, private secondmate config-reread generations with their retry and quarantine state, and parent-owned secondmate pending-reply records under `state/pending-replies/` (`bin/fm-pending-reply-lib.sh`). `config/` holds local gitignored operating choices, and `projects/` holds the local project clones that Firstmate reads but changes only through the narrow guarded and concrete captain-approved exceptions in `AGENTS.md`. `bin/fm-spawn.sh` owns the base task-metadata fields it emits, while the runtime-backend section below owns backend-specific fields and selector interpretation. @@ -306,6 +306,30 @@ The locked bootstrap inheritance pass uses the same per-home changed-set and rer That live discovery starts from `state/*.meta` records with `kind=secondmate`; `data/secondmates.md` only backfills `home=` for older or incomplete meta records. Skipped items, such as a destination checkout that does not yet gitignore the item, are visible warnings but not hard failures. +## Better Stack incident monitoring (config/better-stack-incidents) + +Create an ordinary local file at `config/better-stack-incidents` in the one Firstmate home that should receive fleet incident notifications. +The file is a presence flag with no secret content, is gitignored, and is deliberately not inherited into secondmate homes so one incident does not alert multiple supervisors. +The next locked session-start bootstrap requires `doppler`, `curl`, and `jq`, writes `state/better-stack-incidents.check.sh`, and binds those exact shim bytes in `state/better-stack-incidents.check-trust` through `bin/fm-check-register.sh`. +This uses the existing registered custom-check extension point: the watcher executes only a hash-validated private snapshot, applies `FM_CHECK_TIMEOUT`, and converts each incident line in the poll output into a durable `check:` notification. +The registered poll is a home-level supervision need even with no project work in flight and runs on the default `FM_CHECK_INTERVAL=300` slow-check cadence, which keeps normal detection within minutes without adding a second scheduler. + +`bin/fm-better-stack-incidents-poll.sh` invokes its poll child with `doppler run --silent --no-check-version --no-fallback --project fleet-observability --config prd --only-secrets BETTER_STACK_API_TOKEN`. +`--no-fallback` prevents Doppler from reading or writing a fallback secret file, and the child passes the Better Stack bearer header to `curl` through standard input rather than a command argument or temporary header file. +The token therefore remains runtime-only and must never be added to the flag, repository, logs, task instructions, or another local file. +The poll requests unresolved incidents from Better Stack's documented [`GET /api/v3/incidents`](https://betterstack.com/docs/uptime/api/list-all-incidents/) endpoint with a five-second HTTP bound. + +A poll prints one compact identity line for each unresolved incident; it does not claim delivery state before the watcher appends the wake. +The watcher owns the durable commit point: under the wake-queue lock it appends one `check:` record per incident, publishes the private identity-bound receipt at `state/better-stack-incidents.receipts/`, and then publishes `state/better-stack-incidents.seen/`. +If the watcher stops after queue append, the receipt survives queue drain and the next watcher run completes the seen marker without appending a second wake. +The poll emits every unresolved incident on each scan; the watcher suppresses already-delivered incident IDs, so repeated poll output normally remains silent at the durable wake boundary. +If receipt publication itself fails after queue append and that queue record is drained before retry, a rare duplicate wake can occur; the wake handler deduplicates by incident ID. +Missing credentials, Doppler access failure, network failure, non-success HTTP status, and malformed API data print one `better-stack-error ...` diagnostic and record it in `state/better-stack-incidents.diagnostics/error`; the same diagnostic then remains silent until a successful poll clears the marker or a different failure occurs. + +Remove `config/better-stack-incidents` and rerun locked session start to retire the runnable check and its trust binding. +Bootstrap retains the private seen-ID, receipt, and diagnostic markers so disabling and later re-enabling the poll cannot re-notify every still-open incident. +Better Stack polling from heartbeat handling was an interim practice and is retired; the registered check is the only poll owner, while [`AGENTS.md` section 8](../AGENTS.md#8-supervision-protocol) owns incident triage after a notification arrives. + ## X mode (.env) X mode lets a firstmate instance answer public `@myfirstmate` mentions and act on normal reversible mention requests through firstmate's normal lifecycle. diff --git a/docs/subagent-guard.md b/docs/subagent-guard.md index 47aaf10e0f3..0e48f2f4715 100644 --- a/docs/subagent-guard.md +++ b/docs/subagent-guard.md @@ -367,8 +367,8 @@ tests/fm-subagent-pretool-check.test.sh This change does not close the deeper harness-agnostic defect. Every firstmate guard's in-flight-work branch keys off `state/.meta`, and only `bin/fm-spawn.sh` writes that record. -`bin/fm-supervision-lib.sh` also recognizes an X-mode relay poll as supervision need, but unaccounted primary work still contributes nothing to that predicate. -Without an independent X-mode need, unaccounted primary work therefore reads as idle rather than suspicious. +`bin/fm-supervision-lib.sh` also recognizes X-mode and Better Stack home-level polls as supervision needs, but unaccounted primary work still contributes nothing to that predicate. +Without an independent home-monitoring need, unaccounted primary work therefore reads as idle rather than suspicious. The durable fix for that class is to make the guards treat "the primary is doing project-shaped work with zero `state/*.meta` files" as a suspicious state rather than an idle one. That would catch this class on any harness, including work created through `Bash`. diff --git a/docs/supervision-protocols/grok.md b/docs/supervision-protocols/grok.md index 22444b2bd7f..b4255df5971 100644 --- a/docs/supervision-protocols/grok.md +++ b/docs/supervision-protocols/grok.md @@ -24,7 +24,7 @@ When you see a background-task-completed system reminder for the arm: 1. Run `bin/fm-wake-drain.sh` first. 2. Optionally fetch arm output with `get_command_or_subagent_output()` for the reason line. 3. Handle `signal`, `stale`, `check`, or `heartbeat` using the harness-neutral contract in `AGENTS.md`. -4. Ordinary wake: re-arm the next cycle with the same background `bin/fm-watch-arm.sh` call if work remains in flight or X mode still needs polling. +4. Ordinary wake: re-arm the next cycle with the same background `bin/fm-watch-arm.sh` call if work remains in flight or a home-level X-mode or Better Stack poll remains active. 5. Do not invent a wake from an attach-status line alone. Drain the queue and act only on real wake records or a real watcher reason line. Re-arm attaches to an existing healthy cycle when one is already present and follows its verified successor chain. diff --git a/docs/turnend-guard.md b/docs/turnend-guard.md index 8ee750de397..43ac978675c 100644 --- a/docs/turnend-guard.md +++ b/docs/turnend-guard.md @@ -13,7 +13,7 @@ Do not infer this guard's scope, loop safety, or compatibility tradeoffs for tho `bin/fm-guard.sh` is a pull-based warning that runs only when another supervision command invokes it. The turn-end guard closes the remaining gap at the primary's own turn boundary. -When work is in flight and no identity-matched watcher has a fresh beacon, the harness integration must either block the turn end or force one bounded follow-up that uses the recovery instruction from the emitted session-start protocol. +When work is in flight or a home-level poll is active and no identity-matched watcher has a fresh beacon, the harness integration must either block the turn end or force one bounded follow-up that uses the recovery instruction from the emitted session-start protocol. The guard remains a backstop; [`watcher-continuity.md`](watcher-continuity.md) owns normal continuity. ## Shared predicate @@ -25,9 +25,9 @@ An unmarked checkout or invalid marker falls through to the git-dir check. That check keeps crewmate and scout linked worktrees inert because their git dir differs from their git common dir. It also requires `AGENTS.md`, `bin/`, and the effective state directory. -For an in-scope primary, the guard counts in-flight work from `state/*.meta`. +For an in-scope primary, the guard counts in-flight work from `state/*.meta` and recognizes the generated X-mode and Better Stack checks as independent home-level supervision needs. The default cross-harness mode exits silently with no work in flight. -Claude's `--claude` mode also treats `state/x-watch.check.sh` as supervision need, so X-mode relay polling remains guarded without an in-flight task. +Claude's `--claude` mode uses that full supervision predicate, so either home-level poll remains guarded without an in-flight task. Otherwise it calls `fm_watcher_healthy [grace-seconds] [home]` from `bin/fm-wake-lib.sh`, the same identity-matched lock and fresh-beacon check used by `bin/fm-watch-arm.sh`. A stale beacon blocks even when a watcher pid is live. A fresh leftover beacon blocks when the lock is missing, dead, or identity-mismatched. @@ -76,7 +76,7 @@ That warning uses `bin/fm-supervision-instructions.sh --repair-line`, so it alwa ## Compatibility limits - Child crewmate and scout worktrees are outside scope. -- A valid secondmate home is in scope; an idle secondmate endpoint with no X-mode relay poll remains healthy because it has no supervision need. +- A valid secondmate home is in scope; an idle secondmate endpoint with no home-level poll remains healthy because it has no supervision need. - The direct-blocking and bounded passive-follow-up split is limited to the primary integrations listed above. - OpenCode headless mode and untrusted Grok project hooks remain fail-open at the host boundary. - Kimi Code CLI 0.29.1 exposes only global `[[hooks]]` configuration in `~/.kimi-code/config.toml`, including a `Stop` event with snake_case payload fields `hook_event_name`, `session_id`, `cwd`, and `stop_hook_active`. diff --git a/tests/fm-better-stack-incidents.test.sh b/tests/fm-better-stack-incidents.test.sh new file mode 100755 index 00000000000..bf43fe0c3ee --- /dev/null +++ b/tests/fm-better-stack-incidents.test.sh @@ -0,0 +1,329 @@ +#!/usr/bin/env bash +# Behavior tests for the home-scoped Better Stack incident poll. +# +# The API and Doppler boundary are both mocked. +# These tests make no live network or secret-store calls. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" +# shellcheck source=bin/fm-supervision-lib.sh disable=SC1091 +. "$ROOT/bin/fm-supervision-lib.sh" + +POLL="$ROOT/bin/fm-better-stack-incidents-poll.sh" +REGISTER="$ROOT/bin/fm-check-register.sh" +WATCH="$ROOT/bin/fm-watch.sh" +BASE_PATH=${FM_TEST_BASE_PATH:-/usr/bin:/bin:/usr/sbin:/sbin} +JQ_DIR=$(command -v jq 2>/dev/null) && JQ_DIR=$(dirname "$JQ_DIR") || JQ_DIR= +[ -n "$JQ_DIR" ] && BASE_PATH="$JQ_DIR:$BASE_PATH" +TMP_ROOT=$(fm_test_tmproot fm-better-stack-incidents) + +make_case() { + local name=$1 dir fakebin + dir="$TMP_ROOT/$name" + fakebin=$(fm_fakebin "$dir") + mkdir -p "$dir/home/state" "$dir/home/config" + chmod 0700 "$dir/home/state" + + cat > "$fakebin/doppler" <<'SH' +#!/usr/bin/env bash +printf '%s\n' "$*" >> "$FM_TEST_DOPPLER_LOG" +while [ "$#" -gt 0 ] && [ "$1" != -- ]; do + shift +done +[ "${1:-}" = -- ] || exit 2 +shift +exec env BETTER_STACK_API_TOKEN="${FM_TEST_BETTER_STACK_TOKEN:-}" "$@" +SH + +cat > "$fakebin/curl" <<'SH' +#!/usr/bin/env bash +cat >/dev/null +printf '%s\n' "$*" >> "$FM_TEST_CURL_LOG" +[ "${FM_TEST_CURL_FAIL:-0}" = 0 ] || exit 7 +count=$(cat "$FM_TEST_CURL_COUNT" 2>/dev/null || echo 0) +count=$((count + 1)) +printf '%s\n' "$count" > "$FM_TEST_CURL_COUNT" +body=${FM_TEST_API_BODY:-} +[ "$count" -eq 2 ] && body=${FM_TEST_API_BODY_2:-$body} +printf '%s\n%s' "$body" "${FM_TEST_API_CODE:-200}" +SH + chmod +x "$fakebin/doppler" "$fakebin/curl" + : > "$dir/doppler.log" + : > "$dir/curl.log" + : > "$dir/curl.count" + printf '%s\n' "$dir" +} + +run_poll() { + local dir=$1 + shift + PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$dir/home" \ + FM_TEST_DOPPLER_LOG="$dir/doppler.log" \ + FM_TEST_CURL_LOG="$dir/curl.log" \ + FM_TEST_CURL_COUNT="$dir/curl.count" \ + "$@" "$POLL" +} + +incident_body() { + local id=$1 name=${2:-api-production} started=${3:-2026-08-01T12:00:00.000Z} + jq -cn --arg id "$id" --arg name "$name" --arg started "$started" \ + '{data: [{id: $id, type: "incident", attributes: {name: $name, started_at: $started, resolved_at: null, status: "Started"}}]}' +} + +test_new_incident_and_duplicate_suppression() { + local dir body out rc + dir=$(make_case new-and-duplicate) + body=$(incident_body 25 api-production) + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200); rc=$? + expect_code 0 "$rc" "new incident poll exit" + [ "$out" = 'better-stack-incident opened id=25 name=api-production started=2026-08-01T12:00:00.000Z' ] \ + || fail "new incident must print one compact identity line (got: $out)" + + assert_grep '--project fleet-observability --config prd' "$dir/doppler.log" \ + "poll must select the fleet-observability/prd Doppler scope" + assert_grep '--no-fallback' "$dir/doppler.log" \ + "poll must prohibit Doppler fallback files" + assert_grep '--only-secrets BETTER_STACK_API_TOKEN' "$dir/doppler.log" \ + "poll must inject only the Better Stack token" + assert_grep 'https://uptime.betterstack.com/api/v3/incidents?resolved=false&per_page=50' \ + "$dir/curl.log" "poll must query unresolved Better Stack incidents" + if grep -R -F 'synthetic-test-token' "$dir/home/state" "$dir/doppler.log" "$dir/curl.log" >/dev/null 2>&1; then + fail "the Better Stack token reached private state or command logs" + fi + pass "new Better Stack incident wakes once and duplicate observations stay silent" +} + +test_pagination_and_unsafe_target_rejection() { + local dir body next_body out rc + dir=$(make_case pagination) + body=$(jq -cn --arg next 'https://uptime.betterstack.com/api/v3/incidents?page=2&resolved=false' \ + '{data: [{id:"25", type:"incident", attributes:{name:"page-one", started_at:"t1", resolved_at:null}}], pagination:{next:$next}}') + next_body=$(incident_body 26 page-two t2) + out=$(run_poll "$dir" env FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_BODY_2="$next_body" FM_TEST_API_CODE=200) + [ "$out" = $'better-stack-incident opened id=25 name=page-one started=t1\nbetter-stack-incident opened id=26 name=page-two started=t2' ] \ + || fail "poll must include unresolved incidents from later pages (got: $out)" + body=$(jq -cn '{data: [], pagination:{next:"https://evil.example/api/v3/incidents?resolved=false"}}') + out=$(run_poll "$dir" env FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200); rc=$? + expect_code 0 "$rc" "unsafe pagination target poll exit" + [ "$out" = 'better-stack-error invalid Better Stack pagination target' ] \ + || fail "unsafe pagination target must produce one diagnostic (got: $out)" + pass "Better Stack pagination reaches later pages and rejects unsafe targets" +} + +test_api_error_reports_once_and_recovers() { + local dir out rc + dir=$(make_case api-error) + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY='{"error":"unavailable"}' FM_TEST_API_CODE=503); rc=$? + expect_code 0 "$rc" "API error poll exit" + [ "$out" = 'better-stack-error API returned HTTP 503' ] \ + || fail "API error must produce one visible diagnostic (got: $out)" + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY='{"error":"unavailable"}' FM_TEST_API_CODE=503); rc=$? + expect_code 0 "$rc" "repeated API error poll exit" + [ -z "$out" ] || fail "repeated API failure must not produce a wake storm (got: $out)" + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY='{"data":[]}' FM_TEST_API_CODE=200); rc=$? + expect_code 0 "$rc" "recovered API poll exit" + [ -z "$out" ] || fail "successful recovery must stay silent (got: $out)" + assert_absent "$dir/home/state/better-stack-incidents.diagnostics/error" \ + "successful API access must clear the diagnostic marker" + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_CURL_FAIL=1); rc=$? + expect_code 0 "$rc" "unreachable API poll exit" + [ "$out" = 'better-stack-error Better Stack API unreachable' ] \ + || fail "unreachable API must produce one visible diagnostic (got: $out)" + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_CURL_FAIL=1); rc=$? + expect_code 0 "$rc" "repeated unreachable API poll exit" + [ -z "$out" ] || fail "repeated unreachable API failure must stay quiet (got: $out)" + pass "Better Stack API failures surface once and recovery clears the diagnostic" +} + +test_missing_token_reports_once() { + local dir out rc + dir=$(make_case missing-token) + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN='' \ + FM_TEST_API_BODY='{"data":[]}' FM_TEST_API_CODE=200); rc=$? + expect_code 0 "$rc" "missing-token poll exit" + [ "$out" = 'better-stack-error missing BETTER_STACK_API_TOKEN' ] \ + || fail "missing token must produce one visible diagnostic (got: $out)" + + out=$(run_poll "$dir" env \ + FM_TEST_BETTER_STACK_TOKEN='' \ + FM_TEST_API_BODY='{"data":[]}' FM_TEST_API_CODE=200); rc=$? + expect_code 0 "$rc" "repeated missing-token poll exit" + [ -z "$out" ] || fail "repeated missing-token failure must stay quiet (got: $out)" + pass "missing Better Stack token produces one diagnostic without a wake storm" +} + +test_registered_check_delivers_check_wake() { + local dir state shim body out rc + dir=$(make_case watcher-delivery) + state="$dir/home/state" + shim="$state/better-stack-incidents.check.sh" + body=$(incident_body 91 web-production) + cat > "$shim" </dev/null \ + || fail "could not register Better Stack custom check" + + out=$(PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$dir/home" FM_STATE_OVERRIDE="$state" \ + FM_TEST_DOPPLER_LOG="$dir/doppler.log" FM_TEST_CURL_LOG="$dir/curl.log" \ + FM_TEST_CURL_COUNT="$dir/curl.count" \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200 \ + FM_CHECK_INTERVAL=0 FM_HEARTBEAT=999999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + "$WATCH"); rc=$? + expect_code 0 "$rc" "watcher incident delivery exit" + assert_contains "$out" "check: $shim: better-stack-incident opened id=91" \ + "registered poll output must become a check wake with the incident identity" + [ "$(grep -c 'better-stack-incident opened id=91' "$state/.wake-queue")" -eq 1 ] \ + || fail "new incident must create exactly one durable wake record" + rm -f "$state/.last-check" + out=$(PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$dir/home" FM_STATE_OVERRIDE="$state" \ + FM_TEST_DOPPLER_LOG="$dir/doppler.log" FM_TEST_CURL_LOG="$dir/curl.log" \ + FM_TEST_CURL_COUNT="$dir/curl.count" \ + FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200 \ + FM_CHECK_INTERVAL=0 FM_HEARTBEAT=999999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + "$WATCH"); rc=$? + expect_code 0 "$rc" "duplicate watcher incident delivery exit" + [ "$(grep -c 'better-stack-incident opened id=91' "$state/.wake-queue")" -eq 1 ] \ + || fail "replayed incident must not create a duplicate durable wake" + pass "registered Better Stack poll delivers exactly one authenticated check wake" +} + +test_incident_receipt_recovers_after_marker_failure_and_queue_drain() { + local dir state shim body out rc + dir=$(make_case receipt-recovery) + state="$dir/home/state" + shim="$state/better-stack-incidents.check.sh" + body=$(incident_body 92 recovery-production) + cat > "$shim" </dev/null \ + || fail "could not register receipt-recovery custom check" + printf 'not-a-directory\n' > "$state/better-stack-incidents.seen" + out=$(PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$dir/home" FM_STATE_OVERRIDE="$state" \ + FM_TEST_DOPPLER_LOG="$dir/doppler.log" FM_TEST_CURL_LOG="$dir/curl.log" \ + FM_TEST_CURL_COUNT="$dir/curl.count" FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200 \ + FM_CHECK_INTERVAL=0 FM_HEARTBEAT=999999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + "$WATCH"); rc=$? + [ "$rc" -ne 0 ] || fail "marker publication failure must stop the watcher" + assert_present "$state/better-stack-incidents.receipts/92" \ + "queue append failure boundary must leave a private recovery receipt" + [ "$(grep -c 'better-stack-incident opened id=92' "$state/.wake-queue")" -eq 1 ] \ + || fail "failed marker publication must retain exactly one queued wake" + rm -f "$state/.wake-queue" + rm -f "$state/better-stack-incidents.seen" + mkdir "$state/better-stack-incidents.seen" + chmod 700 "$state/better-stack-incidents.seen" + rm -f "$state/.last-check" + out=$(PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$dir/home" FM_STATE_OVERRIDE="$state" \ + FM_TEST_DOPPLER_LOG="$dir/doppler.log" FM_TEST_CURL_LOG="$dir/curl.log" \ + FM_TEST_CURL_COUNT="$dir/curl.count" FM_TEST_BETTER_STACK_TOKEN=synthetic-test-token \ + FM_TEST_API_BODY="$body" FM_TEST_API_CODE=200 \ + FM_CHECK_INTERVAL=0 FM_HEARTBEAT=999999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + "$WATCH"); rc=$? + expect_code 0 "$rc" "receipt recovery watcher exit" + assert_present "$state/better-stack-incidents.seen/92" \ + "receipt recovery must converge the private seen marker after queue drain" + [ ! -e "$state/.wake-queue" ] || [ -z "$(cat "$state/.wake-queue")" ] \ + || fail "receipt recovery after queue drain must not append a duplicate wake" + pass "incident receipt recovers marker publication after queue drain" +} + +test_bootstrap_arms_and_retires_home_check() { + local dir home out sum1 sum2 + dir=$(make_case bootstrap) + home="$dir/home" + : > "$home/config/better-stack-incidents" + + out=$(PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$home" "$ROOT/bin/fm-bootstrap.sh" 2>/dev/null) + assert_contains "$out" 'BETTER_STACK: incident monitoring on' \ + "bootstrap must announce Better Stack incident monitoring" + assert_present "$home/state/better-stack-incidents.check.sh" \ + "bootstrap must materialize the home-scoped custom check" + assert_present "$home/state/better-stack-incidents.check-trust" \ + "bootstrap must register the custom check bytes" + [ -x "$home/state/better-stack-incidents.check.sh" ] \ + || fail "Better Stack custom check must be executable" + assert_grep 'fm-better-stack-incidents-poll.sh' "$home/state/better-stack-incidents.check.sh" \ + "custom check shim must invoke the tracked poll" + + sum1=$(cat "$home/state/better-stack-incidents.check.sh" \ + "$home/state/better-stack-incidents.check-trust" | shasum) + PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$home" "$ROOT/bin/fm-bootstrap.sh" >/dev/null 2>&1 + sum2=$(cat "$home/state/better-stack-incidents.check.sh" \ + "$home/state/better-stack-incidents.check-trust" | shasum) + [ "$sum1" = "$sum2" ] || fail "bootstrap incident activation must be idempotent" + + rm "$home/config/better-stack-incidents" + out=$(PATH="$dir/fakebin:$BASE_PATH" \ + FM_HOME="$home" "$ROOT/bin/fm-bootstrap.sh" 2>/dev/null) + assert_contains "$out" 'BETTER_STACK: incident monitoring off' \ + "bootstrap must announce removal of an armed incident poll" + assert_absent "$home/state/better-stack-incidents.check.sh" \ + "opt-out must remove the Better Stack custom check" + assert_absent "$home/state/better-stack-incidents.check-trust" \ + "opt-out must remove the custom-check trust binding" + pass "bootstrap idempotently arms and retires the home-scoped incident check" +} + +test_incident_check_keeps_home_supervised() { + local dir state + dir=$(make_case supervision-need) + state="$dir/home/state" + : > "$state/better-stack-incidents.check.sh" + + fm_supervision_needed "$state" 300 \ + || fail "Better Stack incident monitoring must keep an otherwise-idle home supervised" + [ "$FM_SUP_IN_FLIGHT" -eq 0 ] \ + || fail "incident monitoring must not count as a project task" + [ "$FM_SUP_NEEDED" = true ] \ + || fail "incident monitoring must set the home supervision need" + pass "Better Stack incident polling remains supervised with no project work in flight" +} + +test_new_incident_and_duplicate_suppression +test_pagination_and_unsafe_target_rejection +test_api_error_reports_once_and_recovers +test_missing_token_reports_once +test_registered_check_delivers_check_wake +test_incident_receipt_recovers_after_marker_failure_and_queue_drain +test_bootstrap_arms_and_retires_home_check +test_incident_check_keeps_home_supervised