diff --git a/.agents/skills/fleet-dashboard/SKILL.md b/.agents/skills/fleet-dashboard/SKILL.md index 6aac1d2e79a..7be49b0330b 100644 --- a/.agents/skills/fleet-dashboard/SKILL.md +++ b/.agents/skills/fleet-dashboard/SKILL.md @@ -35,10 +35,13 @@ Always pass `--prompt` (or `--prompt-file`) with his own words, unedited: that b `--captain` takes `firstmate`, `dj`, or `river` - whichever of the three is actually driving the work. New cards default to `not_started`; only set `--status` explicitly if work is already under way. -## The six statuses +## The seven statuses -Use exactly these, never a synonym, so the board's filters and the auditor's checks stay meaningful: +Use exactly these, never a synonym, so the board's filters and the auditor's checks stay meaningful. +`needs-attention` and `testing` are easy to conflate and are opposite in the one way that matters: **`testing` is optional to him; `needs-attention` is blocking on him.** If he never looks at a `testing` card, nothing is lost - the work is otherwise complete. If he never looks at a `needs-attention` card, the work is stuck. Put a card in `needs-attention` only when it genuinely cannot proceed without him - a decision, an answer, a physical action, something only he can do or judge. A card that is simply done and could use his eyes stays in `testing`. + +- `needs-attention` - the work cannot proceed without him: a decision, an answer, a physical action, anything only he can supply. Use `--reason` to say what, briefly - that text renders directly on the card so he can act without opening it. This is the loudest status on the board and sorts first; do not use it for routine progress or for something an agent could resolve on its own. - `not-started` - received, nothing begun yet. - `working` - an agent is actively on it right now, not "queued" or "about to start." - `paused` - the Admiral paused it. Only he pauses a task; do not set this because a crew went idle for another reason - that is a stalled task, not a paused one, and belongs in a status update or the auditor's discrepancy log instead. @@ -73,11 +76,12 @@ If you are the fleet auditor (or standing in for it), your job every cycle is to 1. `bin/fm-dashboard.sh list --status working --json` - for each, confirm a real agent is actually on it (live crew/session state, not the card's own say-so). Anything claiming `working` with no corroborating live activity is a genuine discrepancy. 2. `bin/fm-dashboard.sh list --status waiting --json` - confirm the blocker is still real. If a `waiting_on_id` card is already `complete`, or the external condition it names has cleared, that is a discrepancy: the card is stale, not honestly waiting. 3. `bin/fm-dashboard.sh list --status paused --json` - confirm each is still genuinely paused by the Admiral's own word, not just quiet. If one has been unpaused, confirm it is *actually being worked* - an unpaused-but-idle task is exactly the failure this check exists to catch. -4. For anything you log as a discrepancy, use `bin/fm-dashboard.sh audit-log ""`. Be concrete: state what the card claims and what you actually observed, so firstmate can act without re-deriving it. -5. If you cannot verify a card at all (no linked reference, no way to check from where you sit), that is **not** a discrepancy - do not log it as one. Silence on an unverifiable card is correct; a false "wrong" trains the Admiral to ignore the log exactly the way an always-red marker already has once. -6. When the sweep finishes, always call `bin/fm-dashboard.sh audit-run --duration-seconds --checked --discrepancies `, even when nothing was wrong. This is what lets the page show "clean" as a real, timed answer instead of an absence. -7. If the sweep itself fails partway (a source you needed was unreadable, a check errored out), log that with `--kind error` via `audit-log --fleet ""` rather than silently posting a shorter, quieter run. A failed check must never look identical to a clean one. -8. Read the current cadence with `bin/fm-dashboard.sh audit-interval get` at the start of each cycle rather than assuming the last-known value; the Admiral can change it from the page at any time. +4. `bin/fm-dashboard.sh list --status needs-attention --json` - check each `needs_attention` card's own `show ` for how long it has actually sat in that status (its status history has the timestamp it last changed to `needs_attention`). **Age itself is the finding here, asymmetrically with every other status**: a `needs-attention` card that has sat for hours with no reply from him means he was not asked clearly, or the ask never reached him - log it even when the card's claim is otherwise accurate. A `testing` card sitting for the same length of time is not a discrepancy at all - `testing` is optional to him by design, so its age proves nothing. Do not apply this age check to any status but `needs-attention`. +5. For anything you log as a discrepancy, use `bin/fm-dashboard.sh audit-log ""`. Be concrete: state what the card claims and what you actually observed, so firstmate can act without re-deriving it. +6. If you cannot verify a card at all (no linked reference, no way to check from where you sit), that is **not** a discrepancy - do not log it as one. Silence on an unverifiable card is correct; a false "wrong" trains the Admiral to ignore the log exactly the way an always-red marker already has once. +7. When the sweep finishes, always call `bin/fm-dashboard.sh audit-run --duration-seconds --checked --discrepancies `, even when nothing was wrong. This is what lets the page show "clean" as a real, timed answer instead of an absence. +8. If the sweep itself fails partway (a source you needed was unreadable, a check errored out), log that with `--kind error` via `audit-log --fleet ""` rather than silently posting a shorter, quieter run. A failed check must never look identical to a clean one. +9. Read the current cadence with `bin/fm-dashboard.sh audit-interval get` at the start of each cycle rather than assuming the last-known value; the Admiral can change it from the page at any time. ## What not to do diff --git a/bin/fleet-dashboard/server/store.py b/bin/fleet-dashboard/server/store.py index 1b130e8ec3c..4434fc9b4c6 100644 --- a/bin/fleet-dashboard/server/store.py +++ b/bin/fleet-dashboard/server/store.py @@ -17,7 +17,7 @@ import time from contextlib import contextmanager -STATUSES = ("not_started", "working", "paused", "waiting", "testing", "complete") +STATUSES = ("needs_attention", "not_started", "working", "paused", "waiting", "testing", "complete") CAPTAINS = ("firstmate", "captain_dj", "captain_river") NOTE_TABS = ("interpretation", "communication", "needs") NOTE_AUTHORS = ("agent", "firstmate", "admiral") @@ -31,6 +31,7 @@ status TEXT NOT NULL, waiting_on_id TEXT, waiting_reason TEXT, + needs_attention_reason TEXT, starred INTEGER NOT NULL DEFAULT 0, backlog_ref TEXT, initial_prompt TEXT NOT NULL, @@ -85,6 +86,14 @@ DEFAULT_AUDIT_INTERVAL_MINUTES = 15 +# Columns added after the initial schema. CREATE TABLE IF NOT EXISTS leaves an +# already-created table untouched, so a database created before a column +# existed needs it added explicitly - this keeps existing cards and their +# history intact instead of requiring a fresh database. +_TASK_COLUMN_MIGRATIONS = ( + ("needs_attention_reason", "TEXT"), +) + _write_lock = threading.Lock() @@ -100,6 +109,10 @@ def __init__(self, db_path: str): os.makedirs(directory, exist_ok=True) with self._connect() as conn: conn.executescript(SCHEMA) + existing_columns = {row[1] for row in conn.execute("PRAGMA table_info(tasks)")} + for name, coltype in _TASK_COLUMN_MIGRATIONS: + if name not in existing_columns: + conn.execute(f"ALTER TABLE tasks ADD COLUMN {name} {coltype}") conn.execute( "INSERT OR IGNORE INTO settings(key, value) VALUES ('audit_interval_minutes', ?)", (str(DEFAULT_AUDIT_INTERVAL_MINUTES),), @@ -248,15 +261,19 @@ def set_status(self, task_id: str, status: str, waiting_on_id: str | None = None raise KeyError(task_id) if waiting_on_id and not self.task_exists(waiting_on_id): raise ValueError(f"waiting_on_id does not exist: {waiting_on_id!r}") + # `reason` is repurposed per status: what a card is waiting on for + # `waiting`, what is being asked of him for `needs_attention`. The two + # are mutually exclusive, so only the active status's column is kept. + waiting_reason = reason if status == "waiting" else None + needs_attention_reason = reason if status == "needs_attention" else None if status != "waiting": waiting_on_id = None - reason = reason if reason else None ts = now_iso() with self._cursor(write=True) as cur: cur.execute( """UPDATE tasks SET status = ?, waiting_on_id = ?, waiting_reason = ?, - updated_at = ? WHERE id = ?""", - (status, waiting_on_id, reason, ts, task_id), + needs_attention_reason = ?, updated_at = ? WHERE id = ?""", + (status, waiting_on_id, waiting_reason, needs_attention_reason, ts, task_id), ) cur.execute( """INSERT INTO status_history (task_id, from_status, to_status, changed_at, note) diff --git a/bin/fleet-dashboard/web/app.js b/bin/fleet-dashboard/web/app.js index 57817915d73..f0c8790af38 100644 --- a/bin/fleet-dashboard/web/app.js +++ b/bin/fleet-dashboard/web/app.js @@ -11,8 +11,9 @@ import htm from "./vendor/htm.module.js"; const { useState, useEffect, useCallback, useMemo, useRef } = React; const html = htm.bind(React.createElement); -const STATUSES = ["not_started", "working", "paused", "waiting", "testing", "complete"]; +const STATUSES = ["needs_attention", "not_started", "working", "paused", "waiting", "testing", "complete"]; const STATUS_META = { + needs_attention: { label: "Needs Attention" }, not_started: { label: "Not Started" }, working: { label: "Working" }, paused: { label: "Paused" }, @@ -121,6 +122,11 @@ function Card({ task, allTasks, onOpen, onToggleStar, onQuickStatus }) { <${StatusPill} status=${task.status} /> <${CaptainPill} captain=${task.captain} /> + ${task.status === "needs_attention" ? html` +
+ ⚑ ${task.needs_attention_reason || "Needs his decision or action to continue."} +
+ ` : null} ${task.status === "waiting" ? html`
${waitingOn @@ -574,8 +580,12 @@ function App() { return sorted; }, [tasks, filters]); - const favorites = useMemo(() => filtered.filter((t) => t.starred), [filtered]); - const rest = useMemo(() => filtered.filter((t) => !t.starred), [filtered]); + // Needs Attention is pulled out of every other section - blocking-on-him + // work must be the first thing he sees, never buried among starred or + // routine cards regardless of the active sort or filter. + const needsAttention = useMemo(() => filtered.filter((t) => t.status === "needs_attention"), [filtered]); + const favorites = useMemo(() => filtered.filter((t) => t.starred && t.status !== "needs_attention"), [filtered]); + const rest = useMemo(() => filtered.filter((t) => !t.starred && t.status !== "needs_attention"), [filtered]); const counts = useMemo(() => { const status = {}, captain = {}; @@ -598,6 +608,19 @@ function App() { <${ConnBanner} error=${connError} lastOkAt=${lastOkAt} /> + ${needsAttention.length > 0 ? html` +

Needs Attention (${needsAttention.length})

+
+ ${needsAttention.map((t) => html` + <${Card} + key=${t.id} task=${t} allTasks=${tasks} + onOpen=${setSelectedId} onToggleStar=${toggleStar} + onQuickStatus=${(task, status) => setStatus(task.id, status)} + /> + `)} +
+ ` : null} + ${favorites.length > 0 ? html`

Favorites (${favorites.length})

diff --git a/bin/fleet-dashboard/web/styles.css b/bin/fleet-dashboard/web/styles.css index 6c44133528c..4036bfc2992 100644 --- a/bin/fleet-dashboard/web/styles.css +++ b/bin/fleet-dashboard/web/styles.css @@ -12,12 +12,14 @@ --font: -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, Inter, system-ui, sans-serif; /* status colors */ + --st-needs_attention: #ff3b30; --st-not_started: #7a8699; --st-working: #1db954; --st-paused: #e0a940; --st-waiting: #a970ff; --st-testing: #ffd700; --st-complete: #5b8cff; + --st-needs_attention-bg: rgba(255, 59, 48, .12); /* captain tag colors */ --cap-firstmate: #9a9aa2; @@ -139,6 +141,10 @@ h2.section .count { color: var(--text); font-weight: 700; } display: flex; flex-direction: column; gap: 8px; min-width: 0; } .card:hover { border-color: #4a4a4a; } +.card.st-needs_attention { + border-left-width: 6px; border-left-color: var(--st-needs_attention); + border-color: var(--st-needs_attention); background: var(--st-needs_attention-bg); +} .card.st-not_started { border-left-color: var(--st-not_started); } .card.st-working { border-left-color: var(--st-working); } .card.st-paused { border-left-color: var(--st-paused); } @@ -160,6 +166,7 @@ h2.section .count { color: var(--text); font-weight: 700; } padding: 2.5px 8px; border-radius: 99px; white-space: nowrap; } .pill.status { color: #06110b; } +.pill.status.st-needs_attention { background: var(--st-needs_attention); color: #2b0503; } .pill.status.st-not_started { background: var(--st-not_started); color: #0c1116; } .pill.status.st-working { background: var(--st-working); } .pill.status.st-paused { background: var(--st-paused); color: #221a04; } @@ -175,6 +182,16 @@ h2.section .count { color: var(--text); font-weight: 700; } .card-meta { font-size: 12px; color: var(--text-muted); display: flex; justify-content: space-between; gap: 8px; } .card-agent { color: var(--text); font-weight: 500; } +.needs-attention-banner { + font-size: 13px; font-weight: 700; color: #ffb4ac; + background: var(--danger-bg); border: 1px solid var(--st-needs_attention); + border-radius: 6px; padding: 8px 10px; +} + +h2.section.needs-attention-section { color: var(--st-needs_attention); } +h2.section.needs-attention-section::after { background: var(--st-needs_attention); opacity: .4; } +.needs-attention-grid { margin-top: 10px; margin-bottom: 6px; } + .waiting-link-btn, .quick-action { font-size: 12px; font-weight: 600; border-radius: 6px; padding: 6px 10px; border: 1px solid var(--border); background: var(--surface2); color: var(--text); cursor: pointer; diff --git a/bin/fm-dashboard.sh b/bin/fm-dashboard.sh index 0c146a0988a..3ea957d0399 100755 --- a/bin/fm-dashboard.sh +++ b/bin/fm-dashboard.sh @@ -24,6 +24,8 @@ # fm-dashboard.sh captain # fm-dashboard.sh ref # fm-dashboard.sh status [--waiting-on ] [--reason ] +# --reason is what the card is waiting on for `waiting`, or what is +# being asked of him for `needs-attention`; ignored for other statuses. # fm-dashboard.sh star # fm-dashboard.sh unstar # fm-dashboard.sh note --tab \ @@ -36,7 +38,7 @@ # fm-dashboard.sh start|stop|restart|server-status (server process lifecycle) # fm-dashboard.sh --help # -# statuses: not-started working paused waiting testing complete +# statuses: needs-attention not-started working paused waiting testing complete # tabs: interpretation communication needs # captain shorthands: dj -> captain_dj, river -> captain_river, firstmate -> firstmate # @@ -110,8 +112,9 @@ json_escape() { need_tool jq; jq -Rs . <<<"$1"; } canon_status() { case "$1" in not-started|not_started) printf 'not_started' ;; + needs-attention|needs_attention) printf 'needs_attention' ;; working|paused|waiting|testing|complete) printf '%s' "$1" ;; - *) die "unknown status '$1' - valid: not-started working paused waiting testing complete" ;; + *) die "unknown status '$1' - valid: needs-attention not-started working paused waiting testing complete" ;; esac } @@ -203,6 +206,7 @@ cmd_show() { "agent: \(.agent)", "starred: \(.starred == 1)", (if .status == "waiting" then "waiting on: \(.waiting_on_id // "(no card)") - \(.waiting_reason // "")" else empty end), + (if .status == "needs_attention" then "needs attention: \(.needs_attention_reason // "(no reason recorded)")" else empty end), (if .backlog_ref then "ref: \(.backlog_ref)" else empty end), "", "--- prompt ---", @@ -432,7 +436,7 @@ main() { stop) cmd_server_stop ;; restart) cmd_server_stop 2>/dev/null; cmd_server_start ;; server-status) cmd_server_status ;; - ""|--help|-h|help) sed -n '2,45p' "${BASH_SOURCE[0]}" | sed 's/^# \{0,1\}//' ;; + ""|--help|-h|help) sed -n '2,48p' "${BASH_SOURCE[0]}" | sed 's/^# \{0,1\}//' ;; *) die "unknown command '$cmd' - run: fm-dashboard.sh --help" ;; esac } diff --git a/docs/dashboard.md b/docs/dashboard.md index 8cb3380c35a..deb7caed83b 100644 --- a/docs/dashboard.md +++ b/docs/dashboard.md @@ -1,7 +1,7 @@ # The Admiral's Fleet Dashboard A purpose-built task board, styled to match Spectra, that replaced the generated Lavish status page (`bin/fm-status-board.sh`) as the Admiral's primary fleet surface. -One card per task the fleet has been given; six statuses; a captain tag for who is driving it; four tabs per card; a bottom-of-page discrepancy log kept by the fleet auditor. +One card per task the fleet has been given; seven statuses; a captain tag for who is driving it; four tabs per card; a bottom-of-page discrepancy log kept by the fleet auditor. For exact current command syntax, run `bin/fm-dashboard.sh --help`. For when and why an agent calls which command, see [`fleet-dashboard`](../.agents/skills/fleet-dashboard/SKILL.md) - that skill is the single owner of agent-facing usage guidance; this page stays architecture, setup, and the decisions behind the shape. @@ -56,6 +56,14 @@ This was a deliberate choice among three options - read live from the backlog, o The fleet auditor exists specifically to bound that risk: every sweep re-derives ground truth from live crew/session state and the real backlog, compares it to what each card claims, and logs a discrepancy the moment the two disagree, with a timed record of how long the check took so a silently-skipped sweep is itself visible (see "Auditor integration" below). An optional `backlog_ref` field on a card (`bin/fm-dashboard.sh ref `) lets the auditor cross-check a specific card against a specific backlog entry when one exists; a card with no ref is not treated as wrong for that - see the next section. +## Why `needs-attention` is a separate status from `testing` + +Both statuses put a finished-enough card in front of the Admiral, which is why they used to get conflated - and why doing so once buried several of his genuinely open decisions in a place he had no reason to check closely. +The two are opposite on the one axis that matters: whether the work still needs him to move forward. +`testing` is done and optional - if he never opens the card, nothing is lost, it is otherwise complete. +`needs-attention` is stuck without him - a decision, an answer, or a physical action only he can supply. +That asymmetry is why `needs-attention` sorts first and renders loudest on the page, and why the fleet auditor treats its age as a finding in a way it never does for `testing` (see the [`fleet-dashboard`](../.agents/skills/fleet-dashboard/SKILL.md) skill for the exact status definitions and the auditor's per-status procedure). + ## Link policy (standing order 17) `bin/fm-dashboard.sh link` and the underlying `POST /api/tasks/{id}/notes` endpoint reject, structurally, any link whose host contains `github`, any link that is not a full `http(s)://` URL, and any link whose host is local-only and will not resolve from the Admiral's phone (`bin/fleet-dashboard/server/validation.py`). diff --git a/tests/fm-dashboard.test.sh b/tests/fm-dashboard.test.sh index e1bfc3c9db4..19fbfd87029 100755 --- a/tests/fm-dashboard.test.sh +++ b/tests/fm-dashboard.test.sh @@ -94,6 +94,27 @@ test_waiting_status_carries_target_and_reason() { pass "waiting status carries its target card and reason" } +test_needs_attention_status_carries_reason_and_sorts_first() { + local id working_id out + id=$("$DASH" add --title "Needs a decision" --captain firstmate --prompt "checking needs-attention" | awk '{print $1}') + working_id=$("$DASH" add --title "Being actively worked" --captain firstmate --prompt "sort-order control" --status working | awk '{print $1}') + + "$DASH" status "$id" needs-attention --reason "pick red or blue for the trim" >/dev/null \ + || fail "status transition to needs-attention failed" + out=$("$DASH" show "$id") + assert_contains "$out" "status: needs_attention" "needs-attention status did not persist" + assert_contains "$out" "needs attention: pick red or blue for the trim" "needs-attention reason did not persist" + + local first_id + first_id=$("$DASH" list --sort status | head -n1 | awk '{print $1}') + [ "$first_id" = "$id" ] || fail "needs-attention ($id) did not sort above a working card ($working_id) under --sort status, got: $first_id" + + "$DASH" status "$id" working >/dev/null || fail "leaving needs-attention failed" + assert_not_contains "$("$DASH" show "$id")" "needs attention:" "needs-attention reason was not cleared on status change" + + pass "needs-attention status carries a reason and sorts above every other status" +} + test_star_and_delete() { local id id=$(cat "$FM_HOME/task-id") @@ -181,6 +202,7 @@ test_status_and_captain_and_title_updates test_waiting_status_carries_target_and_reason test_notes_tabs_and_empty_tab_semantics test_link_policy_rejects_github_and_localhost +test_needs_attention_status_carries_reason_and_sorts_first test_audit_log_run_and_interval test_bad_input_fails_with_nonzero_exit test_star_and_delete