From daa3edadbf5925a50289269410c2b4fca0ef4779 Mon Sep 17 00:00:00 2001
From: Leon Cheng
Date: Sun, 30 Aug 2026 00:15:17 -0400
Subject: [PATCH 1/4] Retire the command catalogue for generic-argument
workflows
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Adds a generic argument kind to composer workflows, ports 16 of the 23
repository-owned OpenCode commands onto it verbatim, and deletes the
command catalogue, its parser, its simulations and its Playbooks UI.
A `WorkflowPreset` may now declare one `argument` spec whose typed value
IS the visible prompt, or a fixed `prompt` when it collects nothing. The
dialog previously branched on workflow id in four places, one of which
fell through to `objective.trim()` — so an unrecognized id was silently
treated as a Managed Child launch. It now keeps a list of what is
special (the five workflows with real bespoke fields or their own submit
path) rather than a list of what is supported, so an unknown id degrades
to "generic form, sent into this session".
`maxLength` is clamped server-side to the prompt route's own 100,000
character ceiling, and an optional field left blank is refused rather
than sent: the message would be the trusted injector with nothing to
apply it to.
The preview now states "Sent in this session's current mode" for every
workflow that sends into this session. The ported procedures could pin
their own agent in frontmatter (`agent: plan` for the read-only ones)
and a workflow cannot, so the UI says the guarantee is gone instead of
dropping it silently.
The seven commands already covered by same-subject reminders —
background, build-waves, handoff, duck-mode, grill-me, cite-file-lines,
diagram — are deleted rather than converted. `standup` could not be
ported cleanly: its three shell interpolations were its entire input
dataset and an injector is never expanded, so it now instructs the agent
to run them itself and says bash may be denied in a Plan session.
Playbooks becomes workflows-only: the per-project load-state badge, the
install-command blocks, the simulation player and `/playbooks/commands*`
are gone, and the page no longer calls `/api/catalog` at all. The
reminder picker's documentation link follows to `/playbooks/workflows/`
for the six reminders whose command survived as a workflow, and renders
nothing for the six whose command was deleted.
Known breakage, owned by a sibling PR: deleting `agent-skills/commands/`
and its parser leaves `scripts/agent-skills-site.ts` and
`tests/agent-skills-site.test.ts` unable to compile or run. Those files
are deliberately untouched here.
---
AGENTS.md | 103 +-
README.md | 24 +-
agent-skills/CREDITS.md | 23 +-
agent-skills/README.md | 127 +-
agent-skills/command-docs/diagram-examples.md | 62 -
.../command-simulations/background.md | 40 -
.../command-simulations/build-waves.md | 56 -
.../command-simulations/cite-file-lines.md | 30 -
agent-skills/command-simulations/dca.md | 61 -
.../command-simulations/deep-research.md | 55 -
agent-skills/command-simulations/diagram.md | 44 -
.../command-simulations/docs-preview.md | 50 -
agent-skills/command-simulations/duck-mode.md | 47 -
agent-skills/command-simulations/goal.md | 72 -
agent-skills/command-simulations/grill-me.md | 51 -
agent-skills/command-simulations/handoff.md | 65 -
.../leaving-now-wrap-up.md | 73 -
.../command-simulations/manager-children.md | 42 -
.../command-simulations/mini-design-doc.md | 185 ---
.../native-worktree-subagents.md | 37 -
agent-skills/command-simulations/red-team.md | 66 -
.../command-simulations/research-handoff.md | 48 -
.../command-simulations/review-learning.md | 69 -
.../command-simulations/session-handoff.md | 52 -
agent-skills/command-simulations/standup.md | 54 -
.../system-design-artifacts.md | 57 -
agent-skills/command-simulations/verify.md | 63 -
.../command-simulations/worktree-up.md | 49 -
agent-skills/commands/background.md | 44 -
agent-skills/commands/build-waves.md | 57 -
agent-skills/commands/cite-file-lines.md | 25 -
agent-skills/commands/dca.md | 133 --
agent-skills/commands/deep-research.md | 46 -
agent-skills/commands/diagram.md | 46 -
agent-skills/commands/docs-preview.md | 51 -
agent-skills/commands/duck-mode.md | 33 -
agent-skills/commands/goal.md | 51 -
agent-skills/commands/grill-me.md | 41 -
agent-skills/commands/handoff.md | 56 -
agent-skills/commands/leaving-now-wrap-up.md | 51 -
agent-skills/commands/manager-children.md | 66 -
agent-skills/commands/mini-design-doc.md | 45 -
.../commands/native-worktree-subagents.md | 38 -
agent-skills/commands/red-team.md | 40 -
agent-skills/commands/research-handoff.md | 46 -
agent-skills/commands/review-learning.md | 83 --
agent-skills/commands/session-handoff.md | 46 -
agent-skills/commands/standup.md | 38 -
.../commands/system-design-artifacts.md | 164 ---
agent-skills/commands/verify.md | 57 -
agent-skills/commands/worktree-up.md | 42 -
agent-skills/src/lib/commandInstall.test.ts | 32 -
agent-skills/src/lib/commandInstall.ts | 92 --
.../src/lib/commandSimulation.test.ts | 105 --
agent-skills/src/lib/commands.test.ts | 174 ---
agent-skills/src/lib/commands.ts | 151 --
agent-skills/src/lib/commandsSource.ts | 52 -
agent-skills/src/lib/designCommands.test.ts | 44 -
agent-skills/src/lib/frontmatter.test.ts | 312 -----
agent-skills/src/lib/frontmatter.ts | 387 ------
agent-skills/src/lib/reminderCommands.ts | 27 -
agent-skills/src/lib/repo.ts | 23 -
agent-skills/src/lib/simulation.test.ts | 194 ---
agent-skills/src/lib/simulation.ts | 183 ---
.../src/lib/simulationPlayback.test.ts | 36 -
agent-skills/src/lib/simulationPlayback.ts | 17 -
client/components/app-shell.tsx | 4 +-
client/components/playbook-simulation.tsx | 43 -
client/components/reminder-picker.tsx | 8 +-
client/components/workflow-dialog.tsx | 117 +-
client/lib/api.ts | 17 +
client/lib/playbooks.ts | 49 -
client/lib/reminderWorkflows.ts | 30 +
client/lib/usePlaybookInstallState.ts | 95 --
client/lib/workflows.ts | 59 +-
client/main.tsx | 3 -
client/pages/PlaybookDetail.tsx | 57 +-
client/pages/Playbooks.tsx | 61 +-
client/simulator/publicSimulator.ts | 18 +-
reminders/cite-file-lines/SKILL.md | 3 -
reminders/native-worktree-subagents/SKILL.md | 3 -
scripts/pr-screenshots.ts | 2 +-
server/routes/workflows.ts | 10 +-
server/workflows/workflows.ts | 1212 ++++++++++++++++-
tests/e2e/playbooks.ui.spec.ts | 195 +--
tests/e2e/smoke.ui.spec.ts | 17 +-
tests/e2e/workflows.ui.spec.ts | 103 +-
tests/playbook-install-state.test.ts | 56 -
tests/pr-screenshots.test.ts | 9 +-
tests/public-simulator.test.ts | 26 +-
tests/reminders.test.ts | 55 +-
tests/workflows.test.ts | 110 +-
vite.config.ts | 1 -
vitest.config.ts | 2 +-
94 files changed, 1915 insertions(+), 5283 deletions(-)
delete mode 100644 agent-skills/command-docs/diagram-examples.md
delete mode 100644 agent-skills/command-simulations/background.md
delete mode 100644 agent-skills/command-simulations/build-waves.md
delete mode 100644 agent-skills/command-simulations/cite-file-lines.md
delete mode 100644 agent-skills/command-simulations/dca.md
delete mode 100644 agent-skills/command-simulations/deep-research.md
delete mode 100644 agent-skills/command-simulations/diagram.md
delete mode 100644 agent-skills/command-simulations/docs-preview.md
delete mode 100644 agent-skills/command-simulations/duck-mode.md
delete mode 100644 agent-skills/command-simulations/goal.md
delete mode 100644 agent-skills/command-simulations/grill-me.md
delete mode 100644 agent-skills/command-simulations/handoff.md
delete mode 100644 agent-skills/command-simulations/leaving-now-wrap-up.md
delete mode 100644 agent-skills/command-simulations/manager-children.md
delete mode 100644 agent-skills/command-simulations/mini-design-doc.md
delete mode 100644 agent-skills/command-simulations/native-worktree-subagents.md
delete mode 100644 agent-skills/command-simulations/red-team.md
delete mode 100644 agent-skills/command-simulations/research-handoff.md
delete mode 100644 agent-skills/command-simulations/review-learning.md
delete mode 100644 agent-skills/command-simulations/session-handoff.md
delete mode 100644 agent-skills/command-simulations/standup.md
delete mode 100644 agent-skills/command-simulations/system-design-artifacts.md
delete mode 100644 agent-skills/command-simulations/verify.md
delete mode 100644 agent-skills/command-simulations/worktree-up.md
delete mode 100644 agent-skills/commands/background.md
delete mode 100644 agent-skills/commands/build-waves.md
delete mode 100644 agent-skills/commands/cite-file-lines.md
delete mode 100644 agent-skills/commands/dca.md
delete mode 100644 agent-skills/commands/deep-research.md
delete mode 100644 agent-skills/commands/diagram.md
delete mode 100644 agent-skills/commands/docs-preview.md
delete mode 100644 agent-skills/commands/duck-mode.md
delete mode 100644 agent-skills/commands/goal.md
delete mode 100644 agent-skills/commands/grill-me.md
delete mode 100644 agent-skills/commands/handoff.md
delete mode 100644 agent-skills/commands/leaving-now-wrap-up.md
delete mode 100644 agent-skills/commands/manager-children.md
delete mode 100644 agent-skills/commands/mini-design-doc.md
delete mode 100644 agent-skills/commands/native-worktree-subagents.md
delete mode 100644 agent-skills/commands/red-team.md
delete mode 100644 agent-skills/commands/research-handoff.md
delete mode 100644 agent-skills/commands/review-learning.md
delete mode 100644 agent-skills/commands/session-handoff.md
delete mode 100644 agent-skills/commands/standup.md
delete mode 100644 agent-skills/commands/system-design-artifacts.md
delete mode 100644 agent-skills/commands/verify.md
delete mode 100644 agent-skills/commands/worktree-up.md
delete mode 100644 agent-skills/src/lib/commandInstall.test.ts
delete mode 100644 agent-skills/src/lib/commandInstall.ts
delete mode 100644 agent-skills/src/lib/commandSimulation.test.ts
delete mode 100644 agent-skills/src/lib/commands.test.ts
delete mode 100644 agent-skills/src/lib/commands.ts
delete mode 100644 agent-skills/src/lib/commandsSource.ts
delete mode 100644 agent-skills/src/lib/designCommands.test.ts
delete mode 100644 agent-skills/src/lib/frontmatter.test.ts
delete mode 100644 agent-skills/src/lib/frontmatter.ts
delete mode 100644 agent-skills/src/lib/reminderCommands.ts
delete mode 100644 agent-skills/src/lib/repo.ts
delete mode 100644 agent-skills/src/lib/simulation.test.ts
delete mode 100644 agent-skills/src/lib/simulation.ts
delete mode 100644 agent-skills/src/lib/simulationPlayback.test.ts
delete mode 100644 agent-skills/src/lib/simulationPlayback.ts
delete mode 100644 client/components/playbook-simulation.tsx
delete mode 100644 client/lib/playbooks.ts
create mode 100644 client/lib/reminderWorkflows.ts
delete mode 100644 client/lib/usePlaybookInstallState.ts
delete mode 100644 tests/playbook-install-state.test.ts
diff --git a/AGENTS.md b/AGENTS.md
index a1ccb588..182a9c9a 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -53,10 +53,12 @@ several decisions below.
- `reminders//SKILL.md` is read at runtime, not emitted by `tsc`. Keep the root
catalogue beside `dist/` in deployments. Per-message injection accepts an ID only;
the BFF resolves the body and appends the `` sentinel.
-- `agent-skills/` holds repository-owned OpenCode commands, simulations, and supporting
- docs; it is content, not a second app. The Runner renders it natively under
- `/playbooks`. This remains separate from runtime reminders and `/api/catalog`, which
- reports external skills and commands loaded by the connected OpenCode process.
+- `agent-skills/` is **retired** (decision 21c): its commands, simulations, parser and
+ supporting docs are deleted, and only `README.md`, `CREDITS.md` and `LICENSE` remain
+ so the third-party attribution outlives the file that carried it. `/playbooks` now
+ renders the live workflow catalogue only. This remains separate from runtime
+ reminders and `/api/catalog`, which reports external skills and commands loaded by
+ the connected OpenCode process.
## Agent working conventions
@@ -451,7 +453,7 @@ several decisions below.
it gates. A client-side match grants nothing; an unverified candidate simply renders
as the ordinary text it renders as today.
21. **Composer workflows are forms first, and their injectors are visible-but-trusted.**
- The Workflows picker beside Reminder (#167) offers four guided actions —
+ The Workflows picker beside Reminder (#167) started with four guided actions —
Playwright UI review, snippet-by-snippet PR review, send an update to another
session, launch a Managed Child.
Choosing one only opens a form; the sole exits are Cancel, "Apply to composer"
@@ -468,6 +470,44 @@ several decisions below.
dialog states that prompt_async 204/202 means accepted, not completed. The
managed-child form reuses decision #19's route and creates no task card and no
automatic hand-back.
+21b. **Most workflows are one generic argument, not a bespoke branch.** Adding a
+ workflow used to mean adding a fifth, sixth and seventh id to four separate
+ ternaries in `workflow-dialog.tsx`, one of which ended `: objective.trim()` — so an
+ unrecognized id was silently treated as a Managed Child launch. A `WorkflowPreset`
+ may now declare one `argument` spec (label, placeholder, hint, required,
+ maxLength) whose typed value **is** the visible prompt, or a fixed `prompt` when it
+ collects nothing; the dialog renders both from the server's description. The
+ dialog keeps a list of what is *special* — the five workflows with real bespoke
+ fields or their own submit path — rather than a list of what is supported, so an
+ id this build has never seen degrades to "generic form, sent into this session"
+ instead of to "launch a child". `maxLength` is clamped server-side to the prompt
+ route's own 100,000-character ceiling, so a preset can never advertise a field
+ whose full contents the send would reject. An optional field left blank is refused
+ rather than sent: the message would be the trusted injector with nothing to apply
+ it to.
+ The preview states **"Sent in this session's current mode"** for every workflow
+ that sends into this session, and this is not decoration. The 16 procedures ported
+ out of the retired command catalogue could pin their own agent in frontmatter
+ (`agent: plan` for the read-only ones). A workflow carries no declarative mode, and
+ adding one was deliberately rejected as out of scope — so the guarantee is gone,
+ and the UI has to say so rather than let a reader assume it survived.
+21c. **The repository command catalogue is retired; its procedures are workflows.**
+ 23 commands under `agent-skills/commands/` became 16 workflows and 7 deletions.
+ The 7 — `background`, `build-waves`, `handoff`, `duck-mode`, `grill-me`,
+ `cite-file-lines`, `diagram` — were already covered by same-subject reminders, and
+ a second copy of the same instructions is worse than none. The 16 were ported
+ **verbatim**, with only their `$ARGUMENTS` sentences rewritten, because the typed
+ argument now precedes the injector as the prompt.
+ `standup` is the one that could not be ported cleanly: its three `` !`…` ``
+ shell interpolations were its entire input dataset and a workflow injector is
+ never expanded, so it now instructs the agent to run those commands itself and
+ says plainly that bash may be denied in a Plan session. Pretending the data was
+ pre-fetched would have produced a confident standup written from the transcript.
+ The reminder picker's per-reminder documentation link follows to
+ `/playbooks/workflows/` through `client/lib/reminderWorkflows.ts`. That map
+ covers 6 reminders, not the 12 the old command map covered, and the gap is the
+ honest number: the other 6 commands were deleted precisely because the reminder
+ already said what they said, so there is nothing to link to.
21a. **The PR review workflow accepts a number, never a repository.** Its only input is a
pull request, and `parsePullRequestNumber` reduces `253`, `#253` and a pasted PR URL
to the integer alone — a URL's owner, repository and host are **discarded rather than
@@ -893,41 +933,24 @@ several decisions below.
worth reaching. Its wording is duplicated in the outside-window notice, which tells
the reader to use it by name — change the two together or the notice points at a
control that is not on screen.
-31. **Repository Playbooks are commands-first, while connected-process skills remain
- a separate inventory.** The catalogue is a build-time, repository-owned command
- inventory compiled from `agent-skills/` into the bundle. Commands require explicit
- human invocation and contribute zero at-rest retrieval context, so each command owns
- its complete procedure, safety boundaries and failure handling. This repository ships
- no skills. Runtime reminders under root `reminders/` are also separate: they remain
- application-owned per-message prompt bodies and are never sourced from commands.
- The runtime `/skill` Catalog panel still reports whatever external skills and
- commands the connected OpenCode process loaded; do not remove or narrow
- `server/opencode/catalog.ts`. Whether a repository command is
- actually *loaded* is a per-directory runtime fact only `/api/catalog` knows.
- Those are different questions and the page now answers both without conflating
- them. Installation itself stays external: the UI only ever copies a shell
- command the human runs, and it says so rather than leaving "Install grill-me"
- to imply the app did something. The load-state badge is **always labelled with
- the project**, because `/playbooks` is a global route while installation is
- per-directory — a bare "Loaded" would be false in any other project, which is
- worse than saying nothing. Since the route carries no `?directory=`, the last
- selected project is resolved through the same `resolvePaletteDirectory` seam
- the palette and notification centre use. Every failure — no directory, an
- unreachable BFF, a rejected directory — renders **no claim at all** rather than
- defaulting to "not installed", so the badge's absence means "unknown" and never
- "absent". Source links follow the default branch and now say `main` out loud:
- what GitHub shows can be newer than the bundle being read, and looking pinned
- while tracking a moving target is the dishonest option. Related but distinct,
- and stated on the page because the composer's reminder picker deep-links here:
- attaching a reminder is a per-message action that needs no installation, and
- links through an explicit validated reminder-id-to-command mapping; name transforms
- are never guessed.
- Workflows are the live, first-class Playbooks category: they are read only from
- `GET /api/workflows`, including their exact trusted injector, and are never copied
- into a client catalogue. Their picker grouping is presentation-only; unknown ids
- fall under `Other`. Workflow-only routes never query `/api/catalog` or make command
- installation claims, and deep-link absence is reported only after a successful
- workflow catalogue load.
+31. **Playbooks is the live workflow catalogue, and connected-process skills remain
+ a separate inventory.** The repository-owned command catalogue this decision used
+ to describe is retired (decision 21c), and with it the per-project "Loaded in
+ " badge, the install-command copy blocks, the simulation player and the
+ `/playbooks/commands*` routes. Nothing replaced them, because a workflow needs no
+ installation: there is no per-directory question left to ask, so the page no longer
+ calls `/api/catalog` at all.
+ Workflows are read only from `GET /api/workflows`, including their exact trusted
+ injector, and are never copied into a client catalogue. Their grouping is
+ presentation-only and every shipped workflow must be placed in one of the five named
+ groups; the `Other` bucket exists for an id a newer server ships, and a shipped
+ workflow landing there is a placement bug that reads identically in the UI. A
+ detail route reports absence only after a successful catalogue load, never during
+ loading or after a failure.
+ Runtime reminders under root `reminders/` stay separate: application-owned
+ per-message prompt bodies, never sourced from a workflow. The runtime `/skill`
+ Catalog panel still reports whatever external skills and commands the connected
+ OpenCode process loaded; do not remove or narrow `server/opencode/catalog.ts`.
32. **RETIRED — the public command catalogue is gone, and `gh-pages:agent-skills/`
was deleted rather than left serving a stale index.** This decision used to
specify a dependency-free generator that reused the command parser and published
diff --git a/README.md b/README.md
index 7abca1b3..423fb156 100644
--- a/README.md
+++ b/README.md
@@ -27,19 +27,17 @@ the same server and can be attached at the same time, watching the same sessions
## Playbooks
-The Runner's native **Playbooks** section at `/playbooks` catalogs repository-owned,
-human-invoked OpenCode commands and their worked simulations. Their installable
-Markdown source lives in [`agent-skills/`](agent-skills/). Commands add zero retrieval
-context until explicitly invoked and contain their complete workflow and failure
-handling; this repository intentionally ships no skills.
-
-Install a command directly from this repository:
-
-```bash
-mkdir -p ~/.config/opencode/commands
-curl -sL https://raw.githubusercontent.com/leoncheng57/custom-dca-opencode/main/agent-skills/commands/verify.md \
- -o ~/.config/opencode/commands/verify.md
-```
+The Runner's native **Playbooks** section at `/playbooks` catalogs the live **composer
+workflows** the server supplies over `GET /api/workflows`, including the exact trusted
+injector each one appends. A workflow is a guided action: you fill in one form, read
+the exact prompt and the exact trusted instructions, and only then send. Nothing is
+installed, nothing is copied to your machine, and no restart is involved.
+
+The repository-owned command catalogue that used to live here is retired. Sixteen of
+its procedures were ported verbatim into workflows; the other seven were already
+covered by runtime reminders and were simply deleted. One thing genuinely changed
+rather than moved: a command could pin its own agent (`agent: plan`), and a workflow
+cannot — it is sent in the sending session's current mode, and the preview says so.
Runtime reminders under root `reminders/` remain separate per-message prompt content,
and the live `/skill` Catalog panel remains the connected OpenCode process's inventory
diff --git a/agent-skills/CREDITS.md b/agent-skills/CREDITS.md
index 54dc4ccf..b414e04c 100644
--- a/agent-skills/CREDITS.md
+++ b/agent-skills/CREDITS.md
@@ -1,7 +1,10 @@
# Credits
-One command in this repository is adapted from another open-source project.
-The original author retains copyright; their licence terms are reproduced below.
+One command in this repository was adapted from another open-source project.
+That command has since been retired along with the rest of this directory, but
+the attribution stays: the derived work remains in this repository's history,
+and a credit is not something a deletion cancels. The original author retains
+copyright; their licence terms are reproduced below.
---
@@ -14,12 +17,14 @@ The original author retains copyright; their licence terms are reproduced below.
### Adapted, not vendored
-No file in this repository is a copy of upstream. The command that derives
-from it is an original rewrite:
+No file in this repository was ever a copy of upstream. The command that
+derived from it was an original rewrite:
-[`commands/grill-me.md`](commands/grill-me.md) adopts the
-rounds-and-frontier model from upstream `skills/productivity/grilling/SKILL.md`
-@ `0ab1b63`, including its round output format. It differs deliberately:
+`agent-skills/commands/grill-me.md`, retired in the change that emptied this
+directory and readable in history at commit
+`fe9e5ede5f3dc749b0515372ee2e2bc2fc3b3fba`, adopted the rounds-and-frontier
+model from upstream `skills/productivity/grilling/SKILL.md` @ `0ab1b63`,
+including its round output format. It differed deliberately:
- Upstream ships a `grill-me` → `grilling` skill pair. This repository adapts
the behavior into one explicitly invoked `/grill-me` command with no
@@ -27,6 +32,10 @@ rounds-and-frontier model from upstream `skills/productivity/grilling/SKILL.md`
- A closing step offering to emit the outcome as a handoff document or an ADR is
added; upstream has no equivalent.
+The `grill-me` runtime reminder under `../reminders/` is a separate lineage with
+its own recorded provenance, not a relocation of this adapted command. This
+credit does not transfer to it.
+
### MIT Licence
```
diff --git a/agent-skills/README.md b/agent-skills/README.md
index 29962e73..cf10790b 100644
--- a/agent-skills/README.md
+++ b/agent-skills/README.md
@@ -1,8 +1,8 @@
-# OpenCode commands
+# Retired: repository-owned OpenCode commands
-Repository-owned Playbooks are explicit, human-invoked OpenCode slash commands.
-Their source, worked simulations, parser, and documentation live here and are
-built into the Runner's `/playbooks` catalogue.
+This directory used to hold 23 human-invoked OpenCode slash commands, their
+worked simulations, and the build-time parser that fed the Runner's
+`/playbooks` catalogue. All of it has been removed.
There is no public catalogue. The former commands-only static site under
`/custom-dca-opencode/agent-skills/` and the workflow that published it are retired,
@@ -11,91 +11,34 @@ the Runner's own `/playbooks`, which reads this content out of the bundle it was
built from. The archived separate repository still owns
; this repository's token cannot modify it.
-This repository intentionally ships no skills. Model-retrieved skills place
-descriptions in agent context before they are used, while a command costs zero
-context until a human types `/name`. Each command therefore carries its complete
-workflow, safety boundaries, edge cases, and failure handling rather than
-deferring to another content type.
-
-Runtime reminders under root [`reminders/`](../reminders/) are independent
-application-owned, per-message instructions. Their prompt bodies remain in that
-directory; command files are never used as reminder prompt sources. The live
-OpenCode `/skill` Catalog panel is different again: it reports external skills
-and commands loaded by the connected process for one project.
-
-## Layout
-
-```
-commands/.md invocable command
-command-simulations/.md worked transcript rendered by Playbooks
-command-docs/.md optional longer supporting examples
-src/lib/ build-time parser and command install helpers
-```
-
-Every `commands/*.md` file is auto-discovered. A command does not need a paired
-reminder or any naming relationship, so adding a standalone command requires
-only its command file and, normally, a same-named simulation.
-
-## Format
-
-```markdown
----
-description: Shown in OpenCode autocomplete
-agent: build # optional: build | plan
-model: provider/model # optional
-subtask: true # optional: run in a subagent
----
-
-The complete procedure. $ARGUMENTS is substituted, !`cmd` output is injected,
-and @path/to/file.ts is inlined before the model sees it.
-```
-
-Keep the body imperative and self-contained. Include preconditions, explicit
-stopping rules, safety constraints, boundary cases, and responses to likely
-failures. A command may be long; explicit invocation and zero at-rest retrieval
-context are the reason it can own the full procedure without duplication.
-
-## Install
-
-Install globally:
-
-```bash
-mkdir -p ~/.config/opencode/commands
-curl -sL https://raw.githubusercontent.com/leoncheng57/custom-dca-opencode/main/agent-skills/commands/verify.md \
- -o ~/.config/opencode/commands/verify.md
-```
-
-Or install for one project by writing the file under
-`.opencode/commands/.md`. Restart OpenCode after installation because the
-connected process reads commands at startup. Playbooks only shows and copies
-install commands; it never installs content itself.
-
-## Simulations
-
-All simulation frontmatter fields are required:
-
-```markdown
----
-title: Verifying a mixed UI change
-trigger: /verify
-caveat: The transcript omits the later human execution phase.
----
-
-## user
-
-/verify playbooks
-
-## assistant
-
-...
-```
-
-The trigger must be the exact slash command. Turns use `## user`,
-`## assistant`, `## tool`, or `## note`, optionally followed by ` —
>}
- {workflow.id === DESIGN_DOC_PROTOTYPE_WORKFLOW_ID && (
+ {isGeneric && (workflow.argument
+ ?
+ {workflow.argument.label} ({workflow.argument.required ? "required" : "optional"} — this text becomes the prompt)
+
// Deliberately no fields, and so no firstFieldRef: focus falls
// back to the dialog itself, which is what the effect already does
// when the ref is null.
-
- No input needed. Confirm to preview the exact prompt and trusted procedure below.
-
- )}
+ : workflow.prompt
+ ?
+ No input needed. Confirm to preview the exact prompt and trusted procedure below.
+
+ // A workflow that declares neither a field nor a fixed prompt
+ // has nothing to send. Say that instead of offering a dead
+ // Preview button with no explanation.
+ : This workflow supplies no prompt and collects no input, so there is nothing to send. It may need a newer version of this app.)}
{workflow.id === SESSION_UPDATE_WORKFLOW_ID && <>
@@ -635,12 +683,23 @@ export function WorkflowDialog({
{workflow.injector}
Appended by the server exactly as shown. The browser only names the workflow id.
"Apply to composer" only fills the message box for further editing — nothing is sent until you press Send.
- )}
+ {/*
+ * Stated rather than assumed. The procedures ported out of the
+ * retired command catalogue used to pin their own agent in
+ * frontmatter (`agent: plan` for the read-only ones); a workflow
+ * carries no declarative mode, so the session's current mode is
+ * what governs. Dropping that silently would be the expensive
+ * direction to be wrong in.
+ */}
+
+ Sent in this session's current mode. Right now that is {mode}, so a Plan session stops at any write this asks for rather than gaining write access here.
+
- This posts one comment on the pull request. Sent in this session's current mode, so a Plan session will stop at the write rather than post.
+ This posts one comment on the pull request, so a Plan session will stop at the write rather than post.
- {(workflow.id === PLAYWRIGHT_REVIEW_WORKFLOW_ID || workflow.id === PR_SNIPPET_REVIEW_WORKFLOW_ID || workflow.id === DESIGN_DOC_PROTOTYPE_WORKFLOW_ID) && <>
+ {sendsIntoThisSession && <>
>}
diff --git a/client/lib/api.ts b/client/lib/api.ts
index f3bc0a4f..94be1ee1 100644
--- a/client/lib/api.ts
+++ b/client/lib/api.ts
@@ -449,6 +449,19 @@ export interface ReminderSummary {
tags: string[];
}
+/**
+ * The one free-text field a workflow collects. Its typed value becomes the
+ * visible prompt, so a workflow can be added server-side without another
+ * branch in the dialog. `maxLength` is already bounded by the server.
+ */
+export interface WorkflowArgumentSpec {
+ label: string;
+ placeholder?: string;
+ hint?: string;
+ required: boolean;
+ maxLength: number;
+}
+
export interface WorkflowSummary {
id: string;
title: string;
@@ -459,6 +472,10 @@ export interface WorkflowSummary {
* server resolves this text again at submit time.
*/
injector: string;
+ /** Present when the workflow collects free text that becomes the prompt. */
+ argument?: WorkflowArgumentSpec;
+ /** Fixed visible prompt for a workflow that collects nothing. */
+ prompt?: string;
}
export interface MessagePage {
diff --git a/client/lib/playbooks.ts b/client/lib/playbooks.ts
deleted file mode 100644
index 3ba96857..00000000
--- a/client/lib/playbooks.ts
+++ /dev/null
@@ -1,49 +0,0 @@
-import { loadCommandsFromFiles, type Command } from "../../agent-skills/src/lib/commands.js";
-import { loadSimulationsFromFiles } from "../../agent-skills/src/lib/simulation.js";
-
-const commandFiles = import.meta.glob("../../agent-skills/commands/*.md", {
- eager: true,
- query: "?raw",
- import: "default",
-}) as Record;
-
-const commandSimulationFiles = import.meta.glob("../../agent-skills/command-simulations/*.md", {
- eager: true,
- query: "?raw",
- import: "default",
-}) as Record;
-
-function commandSimulationPaths(files: Record): Record {
- return Object.fromEntries(Object.entries(files).map(([path, raw]) => {
- const name = path.split("/").pop()?.replace(/\.md$/u, "") ?? "";
- return [`commands/${name}/SIMULATION.md`, raw];
- }));
-}
-
-export const commands: Command[] = loadCommandsFromFiles(commandFiles, {
- simulations: loadSimulationsFromFiles(commandSimulationPaths(commandSimulationFiles)),
-});
-
-export function findCommand(name: string): Command | undefined {
- return commands.find((command) => command.name === name);
-}
-
-const REPO_ROOT = "https://github.com/leoncheng57/custom-dca-opencode";
-
-/**
- * The revision these source links resolve against.
- *
- * Named rather than implied: the links follow the default branch, so what a
- * reader sees on GitHub can be newer than the bundle they are reading it from.
- * Stating `main` is honest; silently linking to a moving target while looking
- * like a pinned reference is not.
- */
-export const PLAYBOOK_SOURCE_REVISION = "main";
-const CONTENT_ROOT = `${REPO_ROOT}/blob/${PLAYBOOK_SOURCE_REVISION}/agent-skills`;
-
-export const playbookSource = {
- command: (name: string) => `${CONTENT_ROOT}/commands/${name}.md`,
- commandSimulation: (name: string) => `${CONTENT_ROOT}/command-simulations/${name}.md`,
-};
-
-export type { Command };
diff --git a/client/lib/reminderWorkflows.ts b/client/lib/reminderWorkflows.ts
new file mode 100644
index 00000000..135ef07d
--- /dev/null
+++ b/client/lib/reminderWorkflows.ts
@@ -0,0 +1,30 @@
+/**
+ * Reminders and workflows are independent mechanisms: a reminder attaches
+ * trusted instructions to the next message, a workflow is a guided action the
+ * human fills in and sends. This map is a documentation join and nothing more —
+ * it decides where the picker's "details" link points, and grants no authority
+ * to either side.
+ *
+ * It replaces the reminder-to-command join that lived in
+ * `agent-skills/src/lib/reminderCommands.ts`. That map covered twelve reminders
+ * because every reminder had a same-subject command beside it. Only six survive
+ * here, and that is the honest number: the other six commands were deleted
+ * rather than converted, precisely because the reminder already said everything
+ * they said. Linking those to an unrelated workflow would invent a relationship
+ * to keep a symmetry that no longer exists.
+ *
+ * A reminder with no entry simply renders no details link, which is already how
+ * the picker treats an unmapped reminder.
+ */
+export const REMINDER_WORKFLOWS: Readonly> = Object.freeze({
+ "deep-research-subagents": "deep-research",
+ "docs-and-diagram-tooling": "docs-preview",
+ "human-verification-steps": "verify",
+ "native-worktree-subagents": "native-worktree-subagents",
+ "parallel-research-handoff": "research-handoff",
+ "session-handoff": "session-handoff",
+});
+
+export function workflowForReminder(reminderID: string): string | undefined {
+ return REMINDER_WORKFLOWS[reminderID];
+}
diff --git a/client/lib/usePlaybookInstallState.ts b/client/lib/usePlaybookInstallState.ts
deleted file mode 100644
index 8cf5c3e4..00000000
--- a/client/lib/usePlaybookInstallState.ts
+++ /dev/null
@@ -1,95 +0,0 @@
-import { useEffect, useState } from "react";
-import { useLocation } from "react-router-dom";
-
-import { api, type CatalogResponse } from "./api.js";
-import { DIRECTORY_STORAGE_KEY, resolvePaletteDirectory } from "./palette.js";
-
-/**
- * Which repository-owned playbooks are actually loaded by the OpenCode server
- * for the project the user last selected (issue #232).
- *
- * The Playbooks catalogue is a build-time, repository-owned inventory; whether
- * a command is *installed* is a per-directory runtime fact from `/api/catalog`.
- * Playbooks routes carry no `?directory=`, so the last selected project is
- * resolved the same way the palette and notification centre resolve it.
- *
- * Because the answer is per-project but the page is global, `directoryLabel` is
- * not optional decoration: a bare "Installed" badge would be a lie in any other
- * project. Callers must render the label alongside the state.
- *
- * Every failure — no directory, an unreachable BFF, a directory the server
- * rejects — resolves to `status: "unknown"`, which renders no claim at all.
- */
-export type PlaybookInstallStatus = "unknown" | "ready";
-
-export interface PlaybookInstallState {
- status: PlaybookInstallStatus;
- /** Basename of the resolved project, for labelling the claim. */
- directoryLabel: string;
- installedCommands: ReadonlySet;
-}
-
-/** The "state nothing" value. Every failure path resolves to this. */
-export const UNKNOWN_INSTALL_STATE: PlaybookInstallState = {
- status: "unknown",
- directoryLabel: "",
- installedCommands: new Set(),
-};
-
-const EMPTY = UNKNOWN_INSTALL_STATE;
-
-/** Project label for the badge. Exported so the labelling rule is testable. */
-export function projectLabel(directory: string): string {
- const trimmed = directory.replace(/\/+$/u, "");
- return trimmed.slice(trimmed.lastIndexOf("/") + 1) || trimmed;
-}
-
-/**
- * Reduce a catalogue response into install state.
- *
- * Split out of the hook so the fail-closed rules can be unit tested without a
- * DOM: a blank directory, or a catalogue that could not be read, must produce
- * `unknown` rather than an empty "not installed" claim.
- */
-export function installStateFrom(directory: string, catalogue: CatalogResponse | null): PlaybookInstallState {
- if (!directory || !catalogue) return UNKNOWN_INSTALL_STATE;
- return {
- status: "ready",
- directoryLabel: projectLabel(directory),
- installedCommands: new Set(catalogue.commands.map((command) => command.name)),
- };
-}
-
-export function usePlaybookInstallState(): PlaybookInstallState {
- const location = useLocation();
- const [state, setState] = useState(EMPTY);
-
- useEffect(() => {
- const directory = resolvePaletteDirectory(
- location.search,
- typeof localStorage === "undefined" ? null : localStorage.getItem(DIRECTORY_STORAGE_KEY),
- );
- if (!directory) {
- setState(EMPTY);
- return;
- }
- const controller = new AbortController();
- let cancelled = false;
- void api
- .catalog(directory, controller.signal)
- .then((catalogue: CatalogResponse) => {
- if (!cancelled) setState(installStateFrom(directory, catalogue));
- })
- .catch(() => {
- // Fail closed: an unreachable or rejected catalogue must state nothing,
- // not imply "not installed".
- if (!cancelled) setState(EMPTY);
- });
- return () => {
- cancelled = true;
- controller.abort();
- };
- }, [location.search]);
-
- return state;
-}
diff --git a/client/lib/workflows.ts b/client/lib/workflows.ts
index 1ece4dcf..1d973c2c 100644
--- a/client/lib/workflows.ts
+++ b/client/lib/workflows.ts
@@ -36,10 +36,31 @@ export const START_DCA_SESSION_WORKFLOW_ID = "start-dca-session";
export const PR_SNIPPET_REVIEW_WORKFLOW_ID = "pr-snippet-review";
export const DESIGN_DOC_PROTOTYPE_WORKFLOW_ID = "design-doc-prototype";
+// Six semantic groups covering the whole catalogue. The sixteen ported
+// procedures are named as literals rather than as exported constants: unlike
+// the five above they are not referenced anywhere else, because they need no
+// bespoke branch — the generic argument field renders all of them.
+//
+// Nothing should land in the "Other" bucket `groupWorkflows` appends. That
+// bucket exists for a workflow a newer server ships that this build has never
+// heard of, not as a home for one this build forgot to place.
export const WORKFLOW_GROUPS = [
- { label: "Review", ids: [PLAYWRIGHT_REVIEW_WORKFLOW_ID, PR_SNIPPET_REVIEW_WORKFLOW_ID] },
- { label: "Coordinate", ids: [SESSION_UPDATE_WORKFLOW_ID, MANAGED_CHILD_WORKFLOW_ID, START_DCA_SESSION_WORKFLOW_ID] },
- { label: "Document", ids: [DESIGN_DOC_PROTOTYPE_WORKFLOW_ID] },
+ { label: "Review", ids: [PLAYWRIGHT_REVIEW_WORKFLOW_ID, PR_SNIPPET_REVIEW_WORKFLOW_ID, "red-team", "review-learning"] },
+ {
+ label: "Coordinate",
+ ids: [
+ SESSION_UPDATE_WORKFLOW_ID,
+ MANAGED_CHILD_WORKFLOW_ID,
+ START_DCA_SESSION_WORKFLOW_ID,
+ "manager-children",
+ "native-worktree-subagents",
+ "session-handoff",
+ ],
+ },
+ { label: "Execute", ids: ["goal", "dca", "worktree-up"] },
+ { label: "Investigate", ids: ["deep-research", "research-handoff"] },
+ { label: "Document", ids: [DESIGN_DOC_PROTOTYPE_WORKFLOW_ID, "docs-preview", "mini-design-doc", "system-design-artifacts"] },
+ { label: "Ship", ids: ["verify", "leaving-now-wrap-up", "standup"] },
] as const;
export interface WorkflowGroup {
@@ -85,7 +106,7 @@ export function isKnownAppRoute(value: string): boolean {
/^\/planning$/,
/^\/observability$/,
/^\/docs(?:\/[A-Za-z0-9_-]+)?$/,
- /^\/playbooks(?:\/(?:commands|workflows)(?:\/[A-Za-z0-9_-]+)?)?$/,
+ /^\/playbooks(?:\/workflows(?:\/[A-Za-z0-9_-]+)?)?$/,
/^\/sessions\/[A-Za-z0-9_-]+$/,
/^\/dsh(?:\/sessions\/[A-Za-z0-9_-]+)?$/,
].some((pattern) => pattern.test(url.pathname));
@@ -94,14 +115,32 @@ export function isKnownAppRoute(value: string): boolean {
}
}
-// ── Design prototype prompt ─────────────────────────────────────────────────
+// ── Generic argument workflows ──────────────────────────────────────────────
//
-// This workflow collects nothing, so its visible prompt is a fixed constant
-// rather than a builder: every instruction that varies lives in the trusted
-// server-resolved injector, which the preview shows before anything is sent.
+// A workflow that declares an `argument` collects one free-text field whose
+// typed value IS the visible prompt, exactly as "Send an update to another
+// session" and "Launch a Managed Child" already behave. A workflow that
+// declares neither an argument nor a fixed `prompt` has nothing to send, which
+// is why `genericWorkflowPrompt` can return an empty string and
+// `genericWorkflowValid` refuses it.
+
+/** The visible prompt for a workflow with no bespoke builder. */
+export function genericWorkflowPrompt(workflow: Pick, typed: string): string {
+ return workflow.argument ? typed.trim() : (workflow.prompt ?? "");
+}
-export const DESIGN_DOC_PROTOTYPE_PROMPT =
- "Capture a durable design prototype for this proposal and publish it for review.";
+/**
+ * Whether the generic form may advance. An optional argument left blank is
+ * allowed by the spec, but a send whose whole prompt would be empty is not:
+ * the resulting message would be the trusted injector with nothing to apply it
+ * to, and the prompt route rejects empty text anyway.
+ */
+export function genericWorkflowValid(workflow: Pick, typed: string): boolean {
+ const { argument } = workflow;
+ if (argument && argument.required && (typed.trim() === "" || typed.length > argument.maxLength)) return false;
+ if (argument && typed.length > argument.maxLength) return false;
+ return genericWorkflowPrompt(workflow, typed).trim() !== "";
+}
// ── Pull request review prompt generation ───────────────────────────────────
diff --git a/client/main.tsx b/client/main.tsx
index 58bf04d3..8035a9a7 100644
--- a/client/main.tsx
+++ b/client/main.tsx
@@ -22,7 +22,6 @@ import { PUBLIC_SIMULATOR } from "./lib/runtime.js";
import "./styles.css";
const PlaybooksPage = lazy(() => import("./pages/Playbooks.js").then((module) => ({ default: module.PlaybooksPage })));
-const CommandPlaybookPage = lazy(() => import("./pages/PlaybookDetail.js").then((module) => ({ default: module.CommandPlaybookPage })));
const WorkflowPlaybookPage = lazy(() => import("./pages/PlaybookDetail.js").then((module) => ({ default: module.WorkflowPlaybookPage })));
function playbookPage(page: ReactNode): ReactNode {
@@ -57,8 +56,6 @@ async function start(): Promise {
} />
} />
)} />
- )} />
- )} />
)} />
)} />
diff --git a/client/pages/PlaybookDetail.tsx b/client/pages/PlaybookDetail.tsx
index 95903bce..2a662320 100644
--- a/client/pages/PlaybookDetail.tsx
+++ b/client/pages/PlaybookDetail.tsx
@@ -1,25 +1,13 @@
-import { ExternalLink, X } from "lucide-react";
+import { X } from "lucide-react";
import { useEffect, useRef, type ReactNode } from "react";
import { useNavigate, useParams } from "react-router-dom";
-import { commandInstallMethods } from "../../agent-skills/src/lib/commandInstall.js";
-import { invocation } from "../../agent-skills/src/lib/commands.js";
import { PlaybookCopyButton } from "../components/playbook-copy-button.js";
-import { PlaybookSimulation } from "../components/playbook-simulation.js";
import { Alert } from "../ds/alert.js";
-import { Markdown } from "../ds/markdown.js";
-import { findCommand, playbookSource, PLAYBOOK_SOURCE_REVISION } from "../lib/playbooks.js";
-import { usePlaybookInstallState } from "../lib/usePlaybookInstallState.js";
import { groupWorkflows } from "../lib/workflows.js";
-import { InstallState, PlaybooksPage, useWorkflowCatalogue } from "./Playbooks.js";
+import { PlaybooksPage, useWorkflowCatalogue } from "./Playbooks.js";
import styles from "./playbooks.module.css";
-type Method = { id: string; label: string; scope: string; note: string; command: string };
-
-function InstallMethods({ methods }: { methods: Method[] }) {
- return {methods.map((method) =>
{method.label}
{method.scope}
{method.note}
{method.command}
)}
Copying a command here does not install anything. Run it yourself, then restart OpenCode: commands are read at startup.
This page describes repository-owned command content. Viewing or copying here changes nothing: it does not install anything and does not attach anything to a conversation. Runtime reminders are separate, application-owned, per-message instructions.
This guided action is supplied by the live server catalogue. Viewing or copying its injector does not run, attach, or install anything.
;
+ detail =
+
Workflow - {group}
{workflow.title}
{workflow.description}
/playbooks/workflows/{workflow.id}
+ {/*
+ * What the workflow collects is part of reading it: a workflow whose
+ * typed text becomes the prompt behaves very differently from one that
+ * ships a fixed prompt and asks for nothing.
+ */}
+
+
what it asks for
+
{workflow.argument
+ ? <>Collects one field, {workflow.argument.label} ({workflow.argument.required ? "required" : "optional"}, up to {workflow.argument.maxLength.toLocaleString()} characters). What you type becomes the prompt.{workflow.argument.hint ? ` ${workflow.argument.hint}` : ""}>
+ : workflow.prompt
+ ? <>Collects nothing. It sends this fixed prompt: "{workflow.prompt}">
+ : <>Collects nothing here — this workflow supplies its own form in the composer.>}
+
+
Exact trusted injector
{workflow.injector}
+
This guided action is supplied by the live server catalogue. Viewing or copying its injector does not run, attach, or install anything. It is sent in the sending session's current mode: a workflow carries no declarative Plan or Build setting of its own.
+ ;
}
return ;
}
diff --git a/client/pages/Playbooks.tsx b/client/pages/Playbooks.tsx
index 4b63f4c4..b142722a 100644
--- a/client/pages/Playbooks.tsx
+++ b/client/pages/Playbooks.tsx
@@ -1,14 +1,10 @@
-import { Search, Sparkles, TerminalSquare } from "lucide-react";
+import { Search, Sparkles } from "lucide-react";
import type { ReactNode } from "react";
-import { useEffect, useMemo, useRef, useState } from "react";
+import { useEffect, useRef, useState } from "react";
import { Link, useLocation } from "react-router-dom";
-import { filterCommands, invocation } from "../../agent-skills/src/lib/commands.js";
-import { COMMAND_SCOPES } from "../../agent-skills/src/lib/commandInstall.js";
import { Alert } from "../ds/alert.js";
import { api, type WorkflowSummary } from "../lib/api.js";
-import { commands, type Command } from "../lib/playbooks.js";
-import { usePlaybookInstallState, type PlaybookInstallState } from "../lib/usePlaybookInstallState.js";
import { groupWorkflows } from "../lib/workflows.js";
import styles from "./playbooks.module.css";
@@ -32,53 +28,20 @@ export function useWorkflowCatalogue(enabled = true): WorkflowCatalogueState {
return state;
}
-export function InstallState({ install, installed }: { install: PlaybookInstallState; installed: boolean }) {
- if (install.status !== "ready") return null;
- return {installed ? "Loaded" : "Not loaded"} in {install.directoryLabel};
-}
-
-function CommandCard({ command, install }: { command: Command; install: PlaybookInstallState }) {
- return
-
;
- if (state.status === "error") return Workflows could not be loaded. {commandsRendered ? "Commands remain available." : "Try again after the catalogue is available."};
+ if (state.status === "error") return Workflows could not be loaded. Try again after the catalogue is available.;
const needle = query.trim().toLowerCase();
const groups = groupWorkflows(state.workflows).map(({ label, workflows }) => ({ label, workflows: workflows.filter((workflow) => !needle || `${workflow.title} ${workflow.description} ${workflow.id} ${workflow.injector}`.toLowerCase().includes(needle)) })).filter(({ workflows }) => workflows.length);
const visibleCount = groups.reduce((total, group) => total + group.workflows.length, 0);
@@ -94,22 +57,16 @@ export function PlaybooksPage({ detail, workflowState: suppliedWorkflowState }:
const mainRef = useRef(null);
const location = useLocation();
const focusCatalog = (location.state as { focusCatalog?: boolean } | null)?.focusCatalog === true;
- const commandsOnly = location.pathname.startsWith("/playbooks/commands");
- const workflowsOnly = location.pathname.startsWith("/playbooks/workflows");
- const includeCommands = !workflowsOnly;
- const includeWorkflows = !commandsOnly;
- const fetchedWorkflowState = useWorkflowCatalogue(includeWorkflows && suppliedWorkflowState === undefined);
+ const fetchedWorkflowState = useWorkflowCatalogue(suppliedWorkflowState === undefined);
const workflowState = suppliedWorkflowState ?? fetchedWorkflowState;
useEffect(() => { if (focusCatalog) mainRef.current?.focus(); }, [focusCatalog]);
return
Playbooks is still work in progress and its UI/UX may contain bugs.
-
Commands and live workflows
Repeatable work, invoked on purpose.
Commands are repository-owned procedures. Workflows are guided actions loaded live from the trusted server catalogue; runtime reminders remain a separate per-message mechanism.
-
-
Catalogue
{commandsOnly ? "Commands" : workflowsOnly ? "Workflows" : "All Playbooks"}
Workflows are guided actions loaded live from the trusted server catalogue. Nothing here is installed, and nothing runs until you send it from the composer; runtime reminders remain a separate per-message mechanism.
{detail}
diff --git a/client/simulator/publicSimulator.ts b/client/simulator/publicSimulator.ts
index 4a60a963..6666352e 100644
--- a/client/simulator/publicSimulator.ts
+++ b/client/simulator/publicSimulator.ts
@@ -480,10 +480,26 @@ export function createPublicSimulator(): typeof fetch {
if (path === "/api/workflows") return response({ workflows: [
{ id: "playwright-ui-review", title: "Review a UI change with Playwright", description: "Drive a focused Playwright pass over one route or component and bring back targeted evidence, without a full deployment or a complete screenshot regeneration.", injector: "Drive the named route with Playwright and exercise exactly the requested state or interaction. Report what was verified, what failed, and where any evidence was written." },
{ id: "pr-snippet-review", title: "Post a snippet-by-snippet PR review", description: "Walk one pull request as an ordered sequence of explained snippets and post it as a single GitHub comment. Takes only the pull request number; the repository comes from this project directory.", injector: "Produce one ordered snippet-by-snippet GitHub review comment. Resolve the repository from this session's project directory and pin every link to the pull request head SHA." },
+ { id: "red-team", title: "Red-team the work just produced", description: "Argue against the plan or diff that was just produced, grounding every objection in evidence and closing with exactly one verdict.", injector: "Argue against the work above. Ground every objection in file:line or command output and close with exactly one verdict.", argument: { label: "Target", placeholder: "your most recent output, docs/rfc.md, or the plan above", hint: "Name the file or plan to argue against. Say \"your most recent output\" to red-team the work already in this transcript.", required: true, maxLength: 4_000 } },
+ { id: "review-learning", title: "Review code with learning excerpts", description: "Review a diff or pull request findings-first as a senior engineer, then teach from two to five exact, cited source excerpts.", injector: "Return findings first, ordered by severity, then teach from two to five exact cited excerpts.", argument: { label: "Review target", placeholder: "the current diff, pull request #253, or server/routes/sessions.ts", hint: "Name the diff, pull request, file, or subsystem to review. Nothing is changed, posted, or persisted.", required: true, maxLength: 4_000 } },
{ id: "session-update", title: "Send an update to another session", description: "Deliver a hand-off message to another session in this project after an explicit preview of the target and the exact prompt.", injector: "Treat this as new input from the user's other workstream. Delivery is asynchronous: accepted does not mean completed." },
{ id: "managed-child", title: "Launch a Managed Child", description: "Start an independent child session with its own transcript and a Plan or Build policy fixed at creation time. No native task card is created and no automatic hand-back occurs.", injector: "Complete the objective in this independent managed-child transcript and report outcomes, risks, and next steps." },
{ id: "start-dca-session", title: "Start a DCA session", description: "Start an independent root session in this project or an isolated worktree, after reviewing its Plan/Build mode, model, assignment, and trusted instructions.", injector: "Work only on the assignment under the selected mode. This is an independent root session with no parent or automatic hand-back." },
- { id: "design-doc-prototype", title: "Capture a Durable Design Prototype", description: "Mock up an unbuilt UI change as fast static HTML, screenshot it, and publish it into a dated engineering-design document - no fields to fill in, just confirm and send.", injector: "Build a self-contained static HTML mockup, capture desktop and mobile screenshots, and publish the durable design writeup for review." },
+ { id: "manager-children", title: "Run a wave of manager children", description: "Dispatch isolated OpenCode workers into sibling git worktrees and manage their status files, branches, and pull request wave.", injector: "Dispatch each disjoint assignment to its own worktree, branch, and status file, then integrate one reviewed branch at a time.", argument: { label: "Wave to manage", placeholder: "the four independent notification fixes in issues #288, #290, #291, #294", hint: "Describe the disjoint assignments. Each child gets its own worktree, branch, and status file.", required: true, maxLength: 20_000 } },
+ { id: "native-worktree-subagents", title: "Delegate to native worktree subagents", description: "Delegate disjoint edits to native Task children confined to a sibling worktree, with fail-closed containment guards before every mutation.", injector: "Confine every child to its assigned worktree with absolute paths and a pwd guard before any mutation.", argument: { label: "Work to delegate", placeholder: "the three independent client fixes on this branch", hint: "Describe the disjoint edits. Children stay scoped to this session's OpenCode directory regardless of their assigned worktree.", required: true, maxLength: 20_000 } },
+ { id: "session-handoff", title: "Hand off to a standalone session", description: "Carry the current task into one explicitly configured standalone OpenCode session, with a self-contained packet and explicit CLI flags.", injector: "Nothing is inherited automatically. Carry settings through explicit CLI flags and a self-contained handoff packet.", argument: { label: "Task to hand off", placeholder: "finish the epic hierarchy work on feat/planning-epic-hierarchy", hint: "Nothing is inherited automatically, so describe the task the way the receiving session will need to read it.", required: true, maxLength: 20_000 } },
+ { id: "goal", title: "Complete an objective autonomously", description: "Work an objective through research, implementation, fixes, and final verification as one sustained run with durable checkpoints.", injector: "Work the objective through to completion as one sustained run, with a durable checkpoint updated at every boundary.", argument: { label: "Objective", placeholder: "make the notification badge match the server's global unresolved count", hint: "The run continues without asking whether to proceed, so state the acceptance criteria you actually want.", required: true, maxLength: 100_000 } },
+ { id: "dca", title: "Operate DCA sessions over the BFF API", description: "Create, prompt, and observe DCA sessions through the agent-facing BFF HTTP API, distinct from the human-only composer workflows.", injector: "Use the BFF HTTP API to create, prompt, and observe sessions. Accepted is not completed; poll for the finished turn.", argument: { label: "Session work", placeholder: "launch a read-only child that audits export error handling and report its findings", hint: "Describe the session lifecycle work. The procedure below calls already-authorized BFF endpoints directly.", required: true, maxLength: 20_000 } },
+ { id: "worktree-up", title: "Create an isolated worktree", description: "Cut a sibling git worktree from the remote default branch, install its own dependencies, and prove a green baseline before any edits.", injector: "Fetch first, cut a sibling worktree from the remote default branch, install its dependencies, and prove a green baseline.", argument: { label: "Topic", placeholder: "planning-epic-hierarchy (issue #241)", hint: "Becomes the kebab-case worktree directory and branch name. Include the issue number when one exists.", required: true, maxLength: 200 } },
+ { id: "deep-research", title: "Fan out deep research", description: "Split one broad research question across concurrent read-only agents, then reconcile their cited evidence into a single answer.", injector: "Split the question into non-overlapping axes, launch read-only agents concurrently, then synthesize rather than concatenate.", argument: { label: "Research question", placeholder: "how does notification suppression interact with grouping, badges, and Web Push?", hint: "Worth delegating only with three or more independent unknowns. A needle lookup should stay inline.", required: true, maxLength: 20_000 } },
+ { id: "research-handoff", title: "Research, then compile handoff prompts", description: "Research several independent tasks read-only, compile decision-closed handoff prompts, and stop for review before launching anything.", injector: "Research read-only, compile decision-closed prompts, and stop for review before launching anything.", argument: { label: "Tasks to research", placeholder: "one task per line: the epic hierarchy feed, the badge count, the push rotation fix", hint: "List the independent tasks. Nothing is launched until you answer the two questions at the end.", required: true, maxLength: 100_000 } },
+ { id: "design-doc-prototype", title: "Capture a Durable Design Prototype", description: "Mock up an unbuilt UI change as fast static HTML, screenshot it, and publish it into a dated engineering-design document — no fields to fill in, just confirm and send.", injector: "Build a self-contained static HTML mockup, capture desktop and mobile screenshots, and publish the durable design writeup for review.", prompt: "Capture a durable design prototype for this proposal and publish it for review." },
+ { id: "docs-preview", title: "Choose, render, and preview documentation", description: "Pick the documentation medium from where the reader opens it, render it with a real renderer, and preview the result before claiming completion.", injector: "Choose the medium from where the reader opens it, render it with a real renderer, and preview the result before claiming completion.", argument: { label: "What to document", placeholder: "the notification suppression pipeline, as a diagram for the README", hint: "Name the subject and, if you already know it, where the reader will open it.", required: true, maxLength: 20_000 } },
+ { id: "mini-design-doc", title: "Write a mini design doc", description: "Produce a transcript-first technical design narrative that takes under five minutes to read and ends with one clear recommendation.", injector: "Write a transcript-first design narrative under five minutes long, ending with one clear recommendation.", argument: { label: "Subject", placeholder: "whether session status should be joined into the notification centre", hint: "One medium-sized decision. The result stays in this transcript and creates no files.", required: true, maxLength: 20_000 } },
+ { id: "system-design-artifacts", title: "Build a system-design review package", description: "Assemble an evidence-led senior system-design review package, with every load-bearing claim tagged by its evidence class.", injector: "Audit evidence first, tag every load-bearing claim with an evidence class, and select artifacts that each answer a distinct review question.", argument: { label: "Subject", placeholder: "the notification delivery and suppression system, current-state", hint: "Name the system or proposal, and say current-state, target-state, or mixed if you already know which you want.", required: true, maxLength: 20_000 } },
+ { id: "verify", title: "Run the checks, then write verification steps", description: "Discover and run this repository's real verification contract, then write numbered human verification steps with expected results and failure signals.", injector: "Discover this repository's real verification contract, run it, then write numbered human verification steps with expected results and failure signals.", argument: { label: "Surface to verify", placeholder: "the notification popover on mobile", hint: "Scopes the human checklist. The automated checks are discovered from the repository either way.", required: true, maxLength: 4_000 } },
+ { id: "leaving-now-wrap-up", title: "Wrap up and leave an accurate status", description: "Stop this run safely, push authorized progress, refresh every status artifact, and end with one accurate merge verdict.", injector: "Stop this run's work, preserve authorized progress, refresh every status artifact, and end with one accurate merge verdict.", argument: { label: "Wrap-up note", placeholder: "branch feat/planning-epic-hierarchy, PR #312 — post the update there", hint: "May name the branch, pull request, issue, or human update destination. No destination is invented for you.", required: true, maxLength: 4_000 } },
+ { id: "standup", title: "Write a standup update", description: "Gather the last day's commits and pull requests, then turn them into a three-section standup update written to be read aloud.", injector: "Gather the commit and pull request data yourself, then write a three-section standup update from what the commands actually returned.", argument: { label: "Scope", placeholder: "everything, or just the notification work", hint: "Narrows the update to one project or topic. The commit and pull request data is gathered by the agent, not pre-fetched.", required: true, maxLength: 2_000 } },
] });
// -----------------------------------------------------------------------
diff --git a/reminders/cite-file-lines/SKILL.md b/reminders/cite-file-lines/SKILL.md
index ffa91120..7f1ae5ce 100644
--- a/reminders/cite-file-lines/SKILL.md
+++ b/reminders/cite-file-lines/SKILL.md
@@ -3,9 +3,6 @@ name: cite-file-lines
title: Cite File Lines
description: Reference code as file_path:line_number so the user can jump straight to it.
tags: verification
-source_repo: https://github.com/leoncheng57/custom-dca-opencode
-source_path: agent-skills/commands/cite-file-lines.md
-source_commit: fe9e5ede5f3dc749b0515372ee2e2bc2fc3b3fba
---
When you reference a function, class, config key, or any specific piece of code, cite it as
diff --git a/reminders/native-worktree-subagents/SKILL.md b/reminders/native-worktree-subagents/SKILL.md
index cb6f4ac9..d0023d91 100644
--- a/reminders/native-worktree-subagents/SKILL.md
+++ b/reminders/native-worktree-subagents/SKILL.md
@@ -3,9 +3,6 @@ name: native-worktree-subagents
title: Native Worktree Subagents
description: Run mutating OpenCode Task children in isolated sibling worktrees while preserving parent/child session behavior.
tags: worktrees, subagents
-source_repo: https://github.com/leoncheng57/custom-dca-opencode
-source_path: agent-skills/commands/native-worktree-subagents.md
-source_commit: fe9e5ede5f3dc749b0515372ee2e2bc2fc3b3fba
---
Use native Task subagents so delegated work retains `parentID`, sidebar visibility, foreground/background behavior, and result hand-back. Give each mutating child a separate sibling Git worktree and branch created from fresh `origin/main`.
diff --git a/scripts/pr-screenshots.ts b/scripts/pr-screenshots.ts
index 2aeede45..1f1a8cce 100644
--- a/scripts/pr-screenshots.ts
+++ b/scripts/pr-screenshots.ts
@@ -100,7 +100,7 @@ function validateRoute(route: string): void {
const url = new URL(route, "http://screenshot.invalid");
if (url.origin !== "http://screenshot.invalid") throw new Error(`route ${JSON.stringify(route)} may not specify a scheme or host`);
- if (![/^\/$/, /^\/settings$/, /^\/settings\/notifications$/, /^\/tools$/, /^\/planning$/, /^\/observability$/, /^\/docs(?:\/[A-Za-z0-9_-]+)?$/, /^\/playbooks(?:\/(?:commands|workflows)(?:\/[A-Za-z0-9_-]+)?)?$/, /^\/sessions\/[A-Za-z0-9_-]+$/, /^\/dsh(?:\/sessions\/[A-Za-z0-9_-]+)?$/].some((pattern) => pattern.test(url.pathname))) {
+ if (![/^\/$/, /^\/settings$/, /^\/settings\/notifications$/, /^\/tools$/, /^\/planning$/, /^\/observability$/, /^\/docs(?:\/[A-Za-z0-9_-]+)?$/, /^\/playbooks(?:\/workflows(?:\/[A-Za-z0-9_-]+)?)?$/, /^\/sessions\/[A-Za-z0-9_-]+$/, /^\/dsh(?:\/sessions\/[A-Za-z0-9_-]+)?$/].some((pattern) => pattern.test(url.pathname))) {
throw new Error(`route ${JSON.stringify(route)} is not a known UI route`);
}
}
diff --git a/server/routes/workflows.ts b/server/routes/workflows.ts
index 1da1372b..a6de09f1 100644
--- a/server/routes/workflows.ts
+++ b/server/routes/workflows.ts
@@ -9,13 +9,21 @@ export function workflowRoutes(): Router {
// content to be visible before submission. It stays trusted because the
// prompt routes resolve it again by id — a browser-supplied body is never
// accepted anywhere.
+ //
+ // The projection is explicit rather than a spread so a field added to a
+ // preset is a deliberate contract change here. `argument` and `prompt` are
+ // presentation for the field the browser renders and the fixed prompt it
+ // shows; neither grants authority, because the injector is still resolved by
+ // id at send time.
router.get("/workflows", (_req, res) => {
res.json({
- workflows: workflowCatalogue().map(({ id, title, description, injector }) => ({
+ workflows: workflowCatalogue().map(({ id, title, description, injector, argument, prompt }) => ({
id,
title,
description,
injector,
+ ...(argument ? { argument } : {}),
+ ...(prompt ? { prompt } : {}),
})),
});
});
diff --git a/server/workflows/workflows.ts b/server/workflows/workflows.ts
index e1ebd3b1..9f9cb9fe 100644
--- a/server/workflows/workflows.ts
+++ b/server/workflows/workflows.ts
@@ -12,16 +12,47 @@
export const WORKFLOW_ID_RE = /^[a-z0-9]+(-[a-z0-9]+)*$/;
export const WORKFLOW_ID_MAX = 255;
+/**
+ * The hard ceiling on any workflow argument, whatever a preset declares. The
+ * prompt route refuses text longer than 100,000 characters, so a preset must
+ * never advertise a field whose full contents the send would then reject.
+ */
+export const WORKFLOW_ARGUMENT_MAX_LENGTH = 100_000;
+
+/**
+ * The single free-text field a workflow collects. Its typed value IS the
+ * visible prompt — the same shape "Send an update to another session" and
+ * "Launch a Managed Child" already have, generalized so a workflow can be
+ * added by describing its field rather than by growing another branch in the
+ * dialog.
+ *
+ * The spec is presentation for a field the browser renders; it grants no
+ * authority. Trust still comes from resolving `injector` by id at send time.
+ */
+export interface WorkflowArgumentSpec {
+ label: string;
+ placeholder?: string;
+ hint?: string;
+ required: boolean;
+ /** Bounded by WORKFLOW_ARGUMENT_MAX_LENGTH before the catalogue is served. */
+ maxLength: number;
+}
+
export interface WorkflowPreset {
id: string;
title: string;
description: string;
/** Trusted, server-resolved instructions appended to the submitted prompt. */
injector: string;
+ /** Collected free text. When present, what the human types is the prompt. */
+ argument?: WorkflowArgumentSpec;
+ /** Fixed visible prompt for a workflow that collects nothing at all. */
+ prompt?: string;
}
-// Catalogue order is picker order; the two review workflows sit together.
-const CATALOGUE: WorkflowPreset[] = [
+// Catalogue order follows the picker's semantic groups: Review, Coordinate,
+// Execute, Investigate, Document, Ship.
+const RAW_CATALOGUE: WorkflowPreset[] = [
{
id: "playwright-ui-review",
title: "Review a UI change with Playwright",
@@ -62,6 +93,149 @@ const CATALOGUE: WorkflowPreset[] = [
"Post exactly one comment. Report the resulting comment URL when you are done.",
].join("\n"),
},
+ {
+ id: "red-team",
+ title: "Red-team the work just produced",
+ description:
+ "Argue against the plan or diff that was just produced, grounding every objection in evidence and closing with exactly one verdict.",
+ argument: {
+ label: "Target",
+ placeholder: "your most recent output, docs/rfc.md, or the plan above",
+ hint: "Name the file or plan to argue against. Say \"your most recent output\" to red-team the work already in this transcript.",
+ required: true,
+ maxLength: 4_000,
+ },
+ injector: [
+ "Red-team what you just produced. If the note above names a file or a plan, target",
+ "that; otherwise target your own most recent output.",
+ "",
+ "Open with the side-switch, verbatim, as the first line:",
+ "",
+ "> Red-teaming the work above. I am arguing against it.",
+ "",
+ "Then:",
+ "",
+ "1. Work all six objection classes — wrong problem, cheaper alternative, hidden",
+ " coupling, operational cost, reversibility, the unchecked assumption. Say so",
+ " explicitly when a class yields nothing.",
+ "2. Ground every objection in `file:line`, pasted command output, or a doc URL.",
+ " Grep for the other callers rather than reasoning about them.",
+ "3. Keep objections you cannot ground in a separate, clearly labelled",
+ " speculative bucket. Never mix them with grounded ones.",
+ "4. Rank the grounded objections by likelihood x cost-if-true x cheapness-to-check",
+ " and present them as a table.",
+ "5. Close with the single cheapest experiment that would kill the work, and a",
+ " verdict of exactly one of `proceed`, `proceed-with-change`, or `stop`.",
+ "",
+ "Do not re-litigate the work's merits. The case for it has already been made.",
+ "Prefer a fresh subtask context for anything larger than a small diff so the",
+ "reviewer does not inherit the author's sunk cost.",
+ "",
+ "Score likelihood, cost if true, and inverted cost-to-check from 1-5; sort by",
+ "their product. Cheap checks on plausible expensive failures should rise first.",
+ "",
+ "| Failure | Response |",
+ "|---|---|",
+ "| Review hedges or praises the work | Re-state the side switch and delete the defense |",
+ "| Concerns are plausible but ungrounded | Run the grep/curl/read or move them to speculative |",
+ "| Highest concern cannot be acted on | Include cost-to-check and identify one experiment |",
+ "| Same blind spot survives | Move the artifact into a fresh subtask context |",
+ "| Review ends with concerns but no decision | Emit one of the three exact verdicts |",
+ ].join("\n"),
+ },
+ {
+ id: "review-learning",
+ title: "Review code with learning excerpts",
+ description:
+ "Review a diff or pull request findings-first as a senior engineer, then teach from two to five exact, cited source excerpts.",
+ argument: {
+ label: "Review target",
+ placeholder: "the current diff, pull request #253, or server/routes/sessions.ts",
+ hint: "Name the diff, pull request, file, or subsystem to review. Nothing is changed, posted, or persisted.",
+ required: true,
+ maxLength: 4_000,
+ },
+ injector: [
+ "Review the target above as a senior engineer and turn the result into a concise",
+ "learning walkthrough. If no target is supplied, review the current diff. Honor",
+ "any requested subsystem or boundary first, such as authentication or an",
+ "external-runtime integration.",
+ "",
+ "This procedure is self-contained. Do not load or defer to a skill. Do not change",
+ "code, post comments, open issues, or persist the walkthrough unless the user",
+ "explicitly asks after the review.",
+ "",
+ "## Review before teaching",
+ "",
+ "Inspect the complete diff and enough surrounding code, callers, tests,",
+ "configuration, and contracts to understand behavior across layers. For a PR,",
+ "review the whole change at its pinned head revision rather than only the latest",
+ "commit. Reproduce or run focused checks when safe and feasible.",
+ "",
+ "Return findings first, ordered by severity. Each finding must contain:",
+ "",
+ "- severity and a precise title;",
+ "- repository-relative `path:line-line` at the defect or risky behavior;",
+ "- concrete failure mode and affected user/system;",
+ "- evidence establishing the claim;",
+ "- the smallest useful remediation direction, without implementing it.",
+ "",
+ "Separate `Verified findings` from `Unverified risks`. A verified finding is",
+ "supported by implementation, a reproducer, test output, or an authoritative",
+ "contract. An unverified risk names exactly what evidence is missing and the",
+ "cheapest check. Do not inflate educational observations into findings.",
+ "",
+ "If there are no findings, say that explicitly before teaching and name residual",
+ "risks and test gaps. \"No findings\" never means \"proved correct\".",
+ "",
+ "## Then teach from a small evidence set",
+ "",
+ "After findings, select two to five high-value excerpts. Choose code that explains",
+ "an invariant, boundary, state transition, failure strategy, or non-obvious",
+ "tradeoff. Do not quote routine plumbing, the whole diff, or several snippets",
+ "that teach the same lesson.",
+ "",
+ "For every excerpt:",
+ "",
+ "1. give an exact repository-relative `path:line-line` verified against the",
+ " reviewed revision;",
+ "2. quote the smallest contiguous source range that is independently readable;",
+ "3. state whether it is `Finding evidence` or `Educational`;",
+ "4. explain the engineering lesson and why the implementation shape matters;",
+ "5. connect it to the preceding and following layer;",
+ "6. state what the excerpt does not prove.",
+ "",
+ "Use this output shape:",
+ "",
+ "```text",
+ "Verified findings",
+ "1. [severity] Title — path:line-line",
+ " Failure, evidence, remediation direction.",
+ "",
+ "Unverified risks",
+ "- Risk — missing evidence; cheapest verification.",
+ "",
+ "Layer map",
+ "HTTP route -> service/pool -> sidecar protocol -> external SDK",
+ "",
+ "Learning excerpts",
+ "1. path:line-line — Finding evidence | Educational",
+ " ",
+ " Lesson: ...",
+ " Connection: previous layer -> this code -> next layer.",
+ " Does not prove: ...",
+ "",
+ "Residual risks and test gaps",
+ "- ...",
+ "```",
+ "",
+ "Adapt the layer map to the repository. Distinguish control flow from authority:",
+ "a route calling a service does not prove the service may trust browser input,",
+ "and a mock sidecar does not prove the external SDK behaves the same way. Keep",
+ "quoted lines exact, explanations concise, and educational value subordinate to",
+ "the correctness review.",
+ ].join("\n"),
+ },
{
id: "session-update",
title: "Send an update to another session",
@@ -97,11 +271,562 @@ const CATALOGUE: WorkflowPreset[] = [
"- End with a clear summary of outcomes, verification performed, remaining risks, and suggested next steps.",
].join("\n"),
},
+ {
+ id: "manager-children",
+ title: "Run a wave of manager children",
+ description:
+ "Dispatch isolated OpenCode workers into sibling git worktrees and manage their status files, branches, and pull request wave.",
+ argument: {
+ label: "Wave to manage",
+ placeholder: "the four independent notification fixes in issues #288, #290, #291, #294",
+ hint: "Describe the disjoint assignments. Each child gets its own worktree, branch, and status file.",
+ required: true,
+ maxLength: 20_000,
+ },
+ injector: [
+ "Manage the work described above through separate OpenCode workers in sibling git",
+ "worktrees and unfocused CMUX workspaces.",
+ "",
+ "Before dispatch:",
+ "",
+ "1. Write a durable wave plan with ownership, dependencies, integration order,",
+ " and final verification.",
+ "2. State explicitly that standalone children continue after this turn but do",
+ " not automatically resume the manager.",
+ "3. Create one worktree and branch per disjoint assignment from the current",
+ " remote default branch.",
+ "4. Require status files, heartbeats, verified commits, pushed branches, and PRs.",
+ "",
+ "Standalone children cannot resume this manager by writing status, pushing a PR,",
+ "changing a badge, or running `cmux notify`. Those are durable evidence or human",
+ "alerts, not an OpenCode callback. The manager resumes only on a user message, an",
+ "in-process task result, or a separately tested supervisor prompt. Never promise",
+ "unattended progression without that wake channel.",
+ "",
+ "Cut waves on complete artifacts and disjoint files. Parallel workers may read",
+ "shared files but must not edit the same integration file, lockfile, migration,",
+ "generated output, port, or database. Cap a wave around five children. Create",
+ "each sibling worktree after fetching the remote default branch, verify its path,",
+ "branch, clean status, dependencies, baseline, and fixed-port ownership.",
+ "",
+ "Every cold-start assignment must name the absolute worktree and branch,",
+ "objective, owned and forbidden files, sibling workers, settled contracts,",
+ "non-goals, permission posture, model, exact tests, definition of done, and the",
+ "rule that children never push the default branch. Require a gitignored",
+ "`.agent-status.json` with phases `assigned`, `working`, `verifying`, `pushed`,",
+ "`pr-open`, `blocked`, and `done`; UTC timestamps come from `date -u`, update on",
+ "every transition, and heartbeat at least every ten minutes.",
+ "",
+ "Before launching, ask the user which model to use when neither the repository",
+ "nor the current session has an explicit model contract that settles the choice.",
+ "Do not guess from availability, cost, or a previous unrelated child. Never add",
+ "`--auto` or any equivalent broad permission-approval mode unless the user",
+ "explicitly authorizes it for these workers. A copied launch command is not",
+ "authorization. Record both decisions in each assignment and launch command.",
+ "",
+ "Monitor in this order: status file, Git/remote/PR/CI evidence, then the child",
+ "screen only when evidence is stale or contradictory. A stale heartbeat or a",
+ "`done` badge without a pushed branch is not delivery proof.",
+ "",
+ "On every resumed turn, read the durable plan and status files, fetch remotes,",
+ "inspect exact-head checks, reconcile ownership, then continue the queued action.",
+ "If nothing is ready, report that and stop rather than busy-waiting. Integrate one",
+ "reviewed branch at a time, run the full suite after each merge, and clean a",
+ "worktree only after merge and after confirming no follow-up needs it.",
+ "",
+ "| Failure | Response |",
+ "|---|---|",
+ "| Manager stops after dispatch | Wait for a real inbound turn; restore durable state |",
+ "| Notification produces no manager action | Correct the claim: it notified a human only |",
+ "| Two children edit one seam | Stop one writer and sequence ownership |",
+ "| Tracker says done but no PR exists | Inspect Git and require push/PR evidence |",
+ "| Child works in the wrong checkout | Stop and relaunch with absolute containment |",
+ "| Model was not specified or contractually settled | Ask before launch |",
+ "| Launch template contains `--auto` without explicit authorization | Remove it and ask; do not broaden permissions by convenience |",
+ "| Automated wake duplicates turns | Disable it until idle checks, dedupe, and serialization are proven |",
+ "| Manager loses the next action | Restore the synchronized plan and task queue |",
+ ].join("\n"),
+ },
+ {
+ id: "native-worktree-subagents",
+ title: "Delegate to native worktree subagents",
+ description:
+ "Delegate disjoint edits to native Task children confined to a sibling worktree, with fail-closed containment guards before every mutation.",
+ argument: {
+ label: "Work to delegate",
+ placeholder: "the three independent client fixes on this branch",
+ hint: "Describe the disjoint edits. Children stay scoped to this session's OpenCode directory regardless of their assigned worktree.",
+ required: true,
+ maxLength: 20_000,
+ },
+ injector: [
+ "Delegate the work described above only after confirming a fresh Build-only parent and a",
+ "dedicated sibling worktree from fresh `origin/main`. A parent that previously",
+ "activated Plan may pass historical denies to children even after its own Build",
+ "tools return. If the child cannot pass preflight, stop; never weaken policy or",
+ "substitute an unrelated root session.",
+ "",
+ "The child remains scoped to the parent's OpenCode directory. External-directory",
+ "permission does not change relative path resolution, shell CWD, LSP/VCS scope,",
+ "or event directory. Its cold-start prompt must require:",
+ "",
+ "1. The absolute worktree path and branch, with edits allowed only there.",
+ "2. Every Bash call setting `workdir` there or using `git -C `.",
+ "3. Every read, edit, and patch using an absolute path inside the worktree.",
+ "4. Exclusive file ownership, non-goals, exact verification, commit/push rules,",
+ " and the final report.",
+ "5. This guard before edits, tests, commit, and push:",
+ "",
+ " pwd",
+ " git rev-parse --show-toplevel",
+ " git status --short --branch",
+ "",
+ "The child must stop without mutation unless both `pwd` and Git top-level equal",
+ "the assignment. Never fall back to the parent checkout, force-push, or push the",
+ "default branch. Parallel children must not share a lockfile, migration, port,",
+ "database, generated artifact, or integration file. Review the diff and checks",
+ "at hand-back before presenting or merging it, and never duplicate its work.",
+ "",
+ "| Failure | Response |",
+ "|---|---|",
+ "| Child resolves paths in the parent checkout | Stop and relaunch with absolute containment rules |",
+ "| Tool remains denied in Build | Use a fresh Build-only parent; do not weaken rules |",
+ "| Separate branches still conflict | Sequence shared ownership or give it to one owner |",
+ "| Hand-back is unclear | Fix the prompt's deliverable and verification contract before launch |",
+ ].join("\n"),
+ },
+ {
+ id: "session-handoff",
+ title: "Hand off to a standalone session",
+ description:
+ "Carry the current task into one explicitly configured standalone OpenCode session, with a self-contained packet and explicit CLI flags.",
+ argument: {
+ label: "Task to hand off",
+ placeholder: "finish the epic hierarchy work on feat/planning-epic-hierarchy",
+ hint: "Nothing is inherited automatically, so describe the task the way the receiving session will need to read it.",
+ required: true,
+ maxLength: 20_000,
+ },
+ injector: [
+ "Prepare one standalone OpenCode session for the task described above.",
+ "",
+ "Nothing is inherited automatically. Carry settings through explicit CLI flags",
+ "and a self-contained handoff packet.",
+ "",
+ "1. Choose the mechanism: fresh interactive TUI for steerable work,",
+ " `opencode run` for scripted work, `--session --fork` only when history",
+ " must be copied, or the task tool when this is actually a subagent job.",
+ "2. Inspect `opencode agent list`, `opencode models`, the repository root, branch,",
+ " worktree state, and baseline. Mark anything you cannot verify as UNVERIFIED.",
+ "3. Write a prompt file outside the worktree containing the absolute path,",
+ " branch, objective, progress, settled decisions with rationale, owned and",
+ " forbidden files, requested agent/model/variant, permission posture,",
+ " verification commands, stop condition, and unverified assumptions.",
+ "4. Show the packet and exact launch command before executing it. Use",
+ " `--agent plan` or `--agent build`; do not express mode as prompt prose.",
+ " Use `--variant` for provider-specific reasoning effort. `--thinking` controls",
+ " display only. Never add `--auto` unless the user explicitly requested it.",
+ "5. Keep secrets out of the packet and process arguments. Verify the target",
+ " branch before allowing edits.",
+ "6. After launch, verify the working directory, branch, selected agent and model,",
+ " and that the first reply understood the packet. A successful process start",
+ " proves what was requested, not what the provider accepted.",
+ "",
+ "Use cmux only as an optional presentation wrapper and never steal focus. Store",
+ "the packet outside every worktree and keep secrets out of it: prompt text can",
+ "surface in shell history or process arguments. Require the child's first reply",
+ "to restate its path, branch, objective, ownership, agent, model, variant,",
+ "permission posture, and stop condition before work begins.",
+ "",
+ "| Failure | Response |",
+ "|---|---|",
+ "| Child edits instead of planning | Relaunch with `--agent plan`; prose is not mode control |",
+ "| History was copied unexpectedly | Use a fresh TUI or run without continuation/fork flags |",
+ "| Child opens the wrong repository | Pass an absolute project path or `--dir` |",
+ "| Parent and child overwrite one another | Assign disjoint ownership and stop one writer |",
+ "| Secret appears in process arguments | Stop, remove it, and rotate the exposed credential |",
+ "| Provider ignores the reasoning variant | Mark acceptance UNVERIFIED until reported |",
+ "",
+ "After launch, report what started and end the parent turn. Do not begin the",
+ "child's assigned work.",
+ ].join("\n"),
+ },
+ {
+ id: "goal",
+ title: "Complete an objective autonomously",
+ description:
+ "Work an objective through research, implementation, fixes, and final verification as one sustained run with durable checkpoints.",
+ argument: {
+ label: "Objective",
+ placeholder: "make the notification badge match the server's global unresolved count",
+ hint: "The run continues without asking whether to proceed, so state the acceptance criteria you actually want.",
+ required: true,
+ maxLength: 100_000,
+ },
+ injector: [
+ "Complete the objective above as one sustained run.",
+ "",
+ "Do not ask whether to continue. Work through research, implementation, fixes,",
+ "and final verification until the objective is complete or a real user decision",
+ "blocks it.",
+ "",
+ "Before changing code:",
+ "",
+ "1. Read repository instructions and inspect the relevant code, tests, current",
+ " branch, and worktree state. Do not overwrite unrelated changes.",
+ "2. Translate the objective into acceptance criteria and a task list that",
+ " includes final verification. Record assumptions separately from facts.",
+ "3. Create a durable checkpoint using the project's existing status/plan",
+ " convention. If none exists, write a clearly named progress file outside the",
+ " git worktree in a persistent user-state directory, not a temporary directory,",
+ " and report its absolute path. Include the objective, criteria, assumptions,",
+ " current task, completed tasks, touched files, verification, and restart",
+ " instructions. Mirror the active task in the session todo list.",
+ "",
+ "During the run:",
+ "",
+ "- Make the best reasonable guess for ambiguous, reversible, non-safety choices.",
+ " Record the guess and rationale in the checkpoint, implement it, and continue.",
+ "- Prefer the smallest correct change and established repository conventions.",
+ "- Update the checkpoint and todo list after each meaningful boundary, before a",
+ " long operation, and after any verification failure. Keep timestamps in UTC.",
+ "- Treat failed checks as work to diagnose and fix, not as a reason to ask whether",
+ " to proceed. Rerun affected checks after each bounded fix.",
+ "- Preserve Plan/Build and tool permissions. Never switch modes, alter policy, or",
+ " bypass a permission denial to gain authority the session does not have.",
+ "- Obey every confirmation gate. Do not invent credentials, tokens, approvals,",
+ " facts, test results, or access. Never expose secrets in checkpoints or output.",
+ "- Ask one focused question only when progress requires a genuinely irreversible",
+ " or destructive action, security/privacy authorization, privilege escalation,",
+ " spending decision, unavailable credential/access grant, or product choice",
+ " whose plausible answers have materially different consequences. State what",
+ " was tried, the exact decision needed, safe options, and what remains preserved.",
+ "- Do not turn inconvenience, ordinary ambiguity, or a recoverable test failure",
+ " into a user question. Continue all independent work before declaring blocked.",
+ "",
+ "At completion, run the repository's relevant focused and broad verification.",
+ "Inspect the final diff and status for accidental or unrelated changes. Update",
+ "the durable checkpoint to completed or blocked, with exact commands and results.",
+ "Report acceptance criteria, changed files, assumptions made, verification,",
+ "remaining risks, and any work that exists only locally. Never claim completion",
+ "from an accepted asynchronous operation; poll or otherwise verify its outcome.",
+ ].join("\n"),
+ },
+ {
+ id: "dca",
+ title: "Operate DCA sessions over the BFF API",
+ description:
+ "Create, prompt, and observe DCA sessions through the agent-facing BFF HTTP API, distinct from the human-only composer workflows.",
+ argument: {
+ label: "Session work",
+ placeholder: "launch a read-only child that audits export error handling and report its findings",
+ hint: "Describe the session lifecycle work. The procedure below calls already-authorized BFF endpoints directly.",
+ required: true,
+ maxLength: 20_000,
+ },
+ injector: [
+ "Use DCA's BFF API for the request above. This is the agent-facing HTTP workflow for",
+ "creating, prompting, and observing sessions. It is distinct from DCA's",
+ "human-only \"Start a DCA session\" composer workflow: that UI asks a human to",
+ "review and launch a session, while this command calls already-authorized BFF",
+ "endpoints directly.",
+ "",
+ "First establish the boundary:",
+ "",
+ "1. Set `DCA_BASE_URL` to the deployment's canonical BFF origin, without a",
+ " trailing slash. Do not assume `127.0.0.1`, a port, or even that loopback is",
+ " reachable from this shell. A remote, containerized, sandboxed, or phone-side",
+ " shell may need a private DNS/Tailscale URL or may have no route at all.",
+ "2. Use the deployment's existing authentication mechanism. The examples below",
+ " omit auth headers; that is valid only for a private deployment that actually",
+ " has no app-level auth. Never invent a token or put a secret in a tracked file.",
+ "3. Confirm network access and upstream health before creating anything:",
+ "",
+ "```bash",
+ ": \"${DCA_BASE_URL:?Set DCA_BASE_URL to the reachable DCA BFF origin}\"",
+ "curl -sS --fail-with-body \"$DCA_BASE_URL/api/health\" | jq .",
+ "```",
+ "",
+ "`healthy: true` proves the BFF answered. Also inspect `upstream.reachable`,",
+ "`upstream.versionMatches`, and `events.connected`; do not flatten those distinct",
+ "signals into one claim.",
+ "",
+ "Use an absolute project path and URL-encode it. List root sessions:",
+ "",
+ "```bash",
+ "PROJECT=/absolute/path/to/project",
+ "curl -sS --fail-with-body --get \"$DCA_BASE_URL/api/sessions\" \\",
+ " --data-urlencode \"directory=$PROJECT\" \\",
+ " --data-urlencode \"roots=true\" \\",
+ " --data-urlencode \"limit=50\" | jq .",
+ "```",
+ "",
+ "Create a root and submit its first prompt in one request. `mode` is `plan` or",
+ "`build`; the server validates any `model` and optional isolated worktree fields.",
+ "",
+ "```bash",
+ "BODY=$(jq -n \\",
+ " --arg directory \"$PROJECT\" \\",
+ " --arg title \"Investigate flaky export test\" \\",
+ " --arg prompt \"Diagnose the flaky export test. Do not edit files; report evidence.\" \\",
+ " '{directory:$directory,title:$title,prompt:$prompt,mode:\"plan\"}')",
+ "CREATED=$(curl -sS --fail-with-body -X POST \"$DCA_BASE_URL/api/sessions\" \\",
+ " -H 'Content-Type: application/json' -d \"$BODY\")",
+ "SESSION_ID=$(jq -er '.session.id' <<<\"$CREATED\")",
+ "printf '%s\\n' \"$SESSION_ID\"",
+ "```",
+ "",
+ "For an existing session, submit a prompt asynchronously. HTTP `202` and",
+ "`{\"accepted\":true}` mean accepted, not completed:",
+ "",
+ "```bash",
+ "KNOWN_ASSISTANTS=$(curl -sS --fail-with-body --get \\",
+ " \"$DCA_BASE_URL/api/sessions/$SESSION_ID/messages\" \\",
+ " --data-urlencode \"directory=$PROJECT\" --data-urlencode \"limit=100\" \\",
+ " | jq -c '[.messages[]? | select(.info.role == \"assistant\") | .info.id]')",
+ "BODY=$(jq -n --arg directory \"$PROJECT\" --arg text \"Continue and verify the fix.\" \\",
+ " '{directory:$directory,text:$text,mode:\"build\"}')",
+ "curl -sS --fail-with-body -X POST \\",
+ " \"$DCA_BASE_URL/api/sessions/$SESSION_ID/prompt\" \\",
+ " -H 'Content-Type: application/json' -d \"$BODY\" | jq .",
+ "```",
+ "",
+ "Poll messages with the same directory until the submitted turn has a completed",
+ "assistant response and `running` is false. Responses contain raw OpenCode",
+ "`{info,parts}` messages, plus `nextCursor` and `running`; inspect the payload",
+ "rather than guessing a fixed array position. Use `before=` for older",
+ "pages and keep `limit` between 1 and 100.",
+ "",
+ "```bash",
+ "while :; do",
+ " PAGE=$(curl -sS --fail-with-body --get \\",
+ " \"$DCA_BASE_URL/api/sessions/$SESSION_ID/messages\" \\",
+ " --data-urlencode \"directory=$PROJECT\" --data-urlencode \"limit=100\") || exit",
+ " jq '{running, messages}' <<<\"$PAGE\"",
+ " # Stop only after identifying this turn's completed assistant message.",
+ " jq -e --argjson known \"$KNOWN_ASSISTANTS\" '.running == false and",
+ " any(.messages[]?; .info.role == \"assistant\" and",
+ " .info.time.completed != null and",
+ " (.info.id as $id | ($known | index($id)) == null))' \\",
+ " >/dev/null <<<\"$PAGE\" && break",
+ " sleep 3",
+ "done",
+ "```",
+ "",
+ "Use a Managed Child instead of a root when parent/child accounting, inherited",
+ "project scope, child status, or parent-scoped abort authority matters. Query the",
+ "server-owned agent catalogue first. Read-only agents reject `authorization`;",
+ "agents whose catalogue entry has `access: \"can-modify\"` require the explicit",
+ "`authorization: \"modify\"` field. The idempotency key must be stable for a retry",
+ "of the same launch and unique for different work.",
+ "",
+ "```bash",
+ "PARENT_ID=ses_parent",
+ "curl -sS --fail-with-body --get \"$DCA_BASE_URL/api/managed-child-agents\" \\",
+ " --data-urlencode \"directory=$PROJECT\" | jq .",
+ "",
+ "KEY=\"audit-export-$(date -u +%Y%m%dT%H%M%SZ)\"",
+ "BODY=$(jq -n --arg directory \"$PROJECT\" --arg key \"$KEY\" \\",
+ " --arg prompt \"Audit export error handling and report findings with file lines.\" \\",
+ " '{directory:$directory,prompt:$prompt,agent:\"explore\",idempotencyKey:$key}')",
+ "CHILD=$(curl -sS --fail-with-body -X POST \\",
+ " \"$DCA_BASE_URL/api/sessions/$PARENT_ID/managed-children\" \\",
+ " -H 'Content-Type: application/json' -d \"$BODY\")",
+ "CHILD_ID=$(jq -er '.session.id' <<<\"$CHILD\")",
+ "```",
+ "",
+ "The creation request submits the Managed Child's initial prompt. Poll the child",
+ "through `/api/sessions/$CHILD_ID/messages`; later prompts use the ordinary",
+ "`/api/sessions/$CHILD_ID/prompt` route, which re-verifies its managed",
+ "configuration. Inspect parent accounting with",
+ "`GET /api/sessions/$PARENT_ID/subagents?directory=...`. Abort a verified child",
+ "with `POST /api/sessions/$PARENT_ID/subagents/$CHILD_ID/abort` and the directory",
+ "in the JSON body; do not use abort as routine completion handling.",
+ "",
+ "Useful read surfaces are `GET /api/sessions/:id`, `todos`, `diff` with a",
+ "`userMessageID`, `/api/session-agents`, `/api/models`, and `/api/events` for SSE.",
+ "Sharing, deleting, auto-approval, aborting, isolated worktree creation, and",
+ "modify-capable Managed Children are mutations or privilege-bearing operations;",
+ "call them only when the user's request and current permissions authorize them.",
+ "",
+ "Prefer this BFF API over cmux for DCA session lifecycle. cmux is optional",
+ "presentation or orchestration only when its app, CLI/socket access, and current",
+ "environment are actually available; it is not a transport fallback and this",
+ "command never assumes it exists.",
+ ].join("\n"),
+ },
+ {
+ id: "worktree-up",
+ title: "Create an isolated worktree",
+ description:
+ "Cut a sibling git worktree from the remote default branch, install its own dependencies, and prove a green baseline before any edits.",
+ argument: {
+ label: "Topic",
+ placeholder: "planning-epic-hierarchy (issue #241)",
+ hint: "Becomes the kebab-case worktree directory and branch name. Include the issue number when one exists.",
+ required: true,
+ maxLength: 200,
+ },
+ injector: [
+ "Create a git worktree for the topic named above now.",
+ "",
+ "1. Find the repository root and origin's default branch. Run `git fetch origin`",
+ " before branching; start from `origin/`, never a stale local branch.",
+ "2. Create the worktree beside the repository at",
+ " `.worktrees/`, on branch `/`. Use kebab-case and",
+ " include the issue number when one exists.",
+ "3. Install dependencies in the new worktree because gitignored directories such",
+ " as `node_modules` and `.venv` are not shared.",
+ "4. Copy required gitignored local configuration such as `.env`; make any paths",
+ " inside it absolute when they refer back to state in the original checkout.",
+ "5. Establish a green baseline with the repository's typecheck, tests, and build",
+ " before writing code.",
+ "6. Before starting anything on fixed ports, run",
+ " `lsof -nP -iTCP: -sTCP:LISTEN` and confirm the owning PID. Only one",
+ " worktree may run a fixed-port stack at a time.",
+ "7. Report the absolute worktree path, branch, base revision, dependency status,",
+ " baseline result, and any port conflict.",
+ "",
+ "Do not work around Git's worktree safety checks. If `origin/HEAD` is unset, run",
+ "`git remote set-head origin --auto` and re-read it. If the branch exists, omit",
+ "`-b`; if it is checked out elsewhere, find and use that worktree instead.",
+ "",
+ "Worktrees must be siblings, never nested under the clone. On cleanup, inspect",
+ "for uncommitted work before `git worktree remove`; use `--force` only when that",
+ "work is explicitly disposable. Run `git worktree prune` for registrations whose",
+ "directories were deleted outside Git.",
+ "",
+ "| Failure | Response |",
+ "|---|---|",
+ "| New branch is already behind | Fetch and recreate it from the remote default branch |",
+ "| Branch is already checked out | Use `git worktree list`; do not bypass the refusal |",
+ "| New worktree commands fail immediately | Install its own dependencies |",
+ "| App has no local config | Copy required ignored config and fix relative paths |",
+ "| Server behaves like another branch | Verify the listening PID and its worktree |",
+ "| Two stacks share writable state | Stop one stack; use stack-free verification tiers |",
+ "| Deleted path remains registered | Prune stale worktrees, then confirm the list |",
+ ].join("\n"),
+ },
+ {
+ id: "deep-research",
+ title: "Fan out deep research",
+ description:
+ "Split one broad research question across concurrent read-only agents, then reconcile their cited evidence into a single answer.",
+ argument: {
+ label: "Research question",
+ placeholder: "how does notification suppression interact with grouping, badges, and Web Push?",
+ hint: "Worth delegating only with three or more independent unknowns. A needle lookup should stay inline.",
+ required: true,
+ maxLength: 20_000,
+ },
+ injector: [
+ "Research the question above properly.",
+ "",
+ "First decide whether delegation pays. Use this procedure only when the question",
+ "contains at least three independent unknowns, spans unrelated files or sources,",
+ "or would require roughly twenty tool calls. For a needle lookup, do it inline.",
+ "",
+ "If it qualifies:",
+ "",
+ "1. Split it into 3–5 non-overlapping axes. Name what each agent owns and what it",
+ " must not read so they do not converge on the first grep hit.",
+ "2. Launch all agents concurrently in one message. Use `explore` with",
+ " \"very thorough\" for codebase research; use `general` read-only only when bash",
+ " or another unavailable tool is genuinely required.",
+ "3. Every prompt starts and ends with READ-ONLY, supplies the absolute repository",
+ " path, asks numbered questions, requires `file:line` or verbatim URL evidence,",
+ " asks what does **not** exist, and ends with `UNVERIFIED:`.",
+ "4. Spot-check one load-bearing claim from each report yourself.",
+ "5. Synthesize rather than concatenate: answer the original question first,",
+ " reconcile disagreements by reading the cited evidence, merge all unverified",
+ " items, preserve citations, and say what surprised you.",
+ "",
+ "Do not dispatch sequential questions whose later shape depends on an earlier",
+ "answer, and do not let research agents mutate state.",
+ "",
+ "Use a flat fan-out: the usual subagent depth is one. Cap the batch at five;",
+ "above that overlap and synthesis cost usually erase the gain. A `general` agent",
+ "must be told READ-ONLY at both the start and end of its prompt; prefer the",
+ "enforced read-only `explore` agent whenever its tools are sufficient.",
+ "",
+ "Specify a bounded deliverable, such as one answer-first section per numbered",
+ "question under 800 words. If a live API is in scope, allow GET only and request",
+ "the verbatim schema rather than a paraphrase.",
+ "",
+ "| Failure | Response |",
+ "|---|---|",
+ "| Reports repeat each other | Split by artifact or directory and name exclusions |",
+ "| Reports are essays without evidence | Ask numbered questions and require citations |",
+ "| A report contains invented certainty | Merge `UNVERIFIED` lists and spot-check its load-bearing claim |",
+ "| A child mutates state | Stop it; use enforced read-only delegation |",
+ "| Calls ran sequentially | Relaunch independent axes concurrently or keep the work inline |",
+ "| Synthesis is longer than the reports | Answer first, reconcile conflicts, preserve only decisive evidence |",
+ ].join("\n"),
+ },
+ {
+ id: "research-handoff",
+ title: "Research, then compile handoff prompts",
+ description:
+ "Research several independent tasks read-only, compile decision-closed handoff prompts, and stop for review before launching anything.",
+ argument: {
+ label: "Tasks to research",
+ placeholder: "one task per line: the epic hierarchy feed, the badge count, the push rotation fix",
+ hint: "List the independent tasks. Nothing is launched until you answer the two questions at the end.",
+ required: true,
+ maxLength: 100_000,
+ },
+ injector: [
+ "Research and prepare handoffs for the tasks listed above.",
+ "",
+ "Three phases, in order:",
+ "",
+ "1. **Read-only research.** Split the supplied tasks into independent axes and",
+ " launch one read-only agent per task concurrently. Require structure, prior",
+ " art, the nearest analogue, concrete integration points, live API truth when",
+ " reachable by GET, testing conventions, explicit gaps, `file:line` evidence,",
+ " and a final `UNVERIFIED:` list.",
+ "2. **Compile prompts.** Turn each report into a decision-closed prompt containing",
+ " the absolute worktree/branch state, docs to read first,",
+ " `PRE-RESEARCHED - DO NOT RE-DERIVE`, settled decisions with rationale,",
+ " `GOTCHA:` lines, numbered build steps, reasoned exclusions, constraints,",
+ " exact verification, `SHARED-RESOURCE RULE`, and the report-back contract.",
+ " Store prompt files outside every git worktree.",
+ "3. **Stop for review.** Show every prompt before launching. Ask whether the",
+ " receiving sessions should plan first or edit immediately, and whether they",
+ " should open PRs or leave local commits. Do not fire anything until those two",
+ " choices are answered.",
+ "",
+ "Use plain ASCII in prompt files, and do not recreate a baseline from a stale",
+ "local default branch. Research axes must be independent, read-only, and large",
+ "enough to justify delegation; a needle lookup stays inline. State READ-ONLY at",
+ "both ends of each prompt. Restrict live API probes to GET and preserve verbatim",
+ "response shapes.",
+ "",
+ "For later launch, fetch first, create sibling worktrees from the remote default",
+ "branch, install dependencies in each, and prove a baseline before allowing",
+ "edits. Enumerate fixed ports, writable state, databases, generated output, and",
+ "lockfiles in every receiving prompt. Only one worker may own a shared resource.",
+ "Never steal focus when creating a session.",
+ "",
+ "| Failure | Response |",
+ "|---|---|",
+ "| Research agent starts implementing | Stop it and strengthen the read-only boundary |",
+ "| Receiving agent re-greps everything | Add `file:line` evidence and explicit negative findings |",
+ "| Scope is relitigated | Preserve the decision's rationale and evidence |",
+ "| Both workers bind one port or state directory | Keep stack-free checks parallel; serialize the shared stack |",
+ "| First test is red | Establish whether baseline or worker caused it before proceeding |",
+ "| Prompt is mangled | Store plain-ASCII text in a file; do not inline multiline shell arguments |",
+ "| Live feature silently no-ops | Probe its actual gate or API before writing the handoff |",
+ ].join("\n"),
+ },
{
id: "design-doc-prototype",
title: "Capture a Durable Design Prototype",
description:
"Mock up an unbuilt UI change as fast static HTML, screenshot it, and publish it into a dated engineering-design document — no fields to fill in, just confirm and send.",
+ // Collects nothing, so its visible prompt is fixed here rather than in the
+ // browser: every instruction that varies lives in the injector below.
+ prompt: "Capture a durable design prototype for this proposal and publish it for review.",
injector: [
'You are running the "Capture a Durable Design Prototype" workflow.',
"- Use this only for a proposal that is NOT yet built. For reviewing an already-shipped",
@@ -125,8 +850,491 @@ const CATALOGUE: WorkflowPreset[] = [
"- Report the branch, the raw URLs, and the created or updated Notion page URL.",
].join("\n"),
},
+ {
+ id: "docs-preview",
+ title: "Choose, render, and preview documentation",
+ description:
+ "Pick the documentation medium from where the reader opens it, render it with a real renderer, and preview the result before claiming completion.",
+ argument: {
+ label: "What to document",
+ placeholder: "the notification suppression pipeline, as a diagram for the README",
+ hint: "Name the subject and, if you already know it, where the reader will open it.",
+ required: true,
+ maxLength: 20_000,
+ },
+ injector: [
+ "Document or render the subject above.",
+ "",
+ "Choose the medium from where the reader opens it:",
+ "",
+ "- terminal, PR diff, commit message, AGENTS.md, code comment -> ASCII",
+ "- GitHub/GitLab README or wiki -> Mermaid",
+ "- slide, issue attachment, external document -> rendered SVG",
+ "- a static docs site -> Mermaid only after proving that site renders it;",
+ " otherwise ASCII or a checked-in SVG",
+ "",
+ "Then:",
+ "",
+ "1. Read one existing document from the same repository or docs site and",
+ " match its house style before writing.",
+ "2. Use tables for comparisons, lists for linear sequences, diagrams for",
+ " topology, and prose only for the reasoning that connects them.",
+ "3. Cite the code as `file:line`, and call out the limitation or trap a reader",
+ " cannot infer from the source.",
+ "4. Use the available renderer or `diagram` subagent rather than hand-writing a",
+ " Mermaid block and hoping it parses. Pass an explicit output directory for",
+ " SVG artifacts.",
+ "5. Preview the finished document: `cmux markdown open` for Markdown, or build",
+ " or serve the docs site locally with its own generator. Inspect the rendered",
+ " output and build logs before claiming completion.",
+ "",
+ "If a named renderer or MCP server is unavailable in this session, say so and",
+ "offer the medium that can actually be verified. \"It should render\" is not",
+ "verification.",
+ "",
+ "When unsure, choose ASCII because it remains readable without a renderer. Never",
+ "maintain ASCII and Mermaid copies of the same diagram. For rendered SVG, pass an",
+ "explicit output directory and use a light theme for documents normally read on",
+ "a light page. A Markdown preview validates structure but may show Mermaid as a",
+ "code block; build the actual docs site to prove its plugin chain renders it.",
+ "",
+ "Check the current agent and MCP roster instead of assuming machine-local tools",
+ "are connected. A table beats a diagram for two-axis comparison, and a numbered",
+ "list beats one for a branchless sequence.",
+ "",
+ "| Failure | Response |",
+ "|---|---|",
+ "| Mermaid is visible as source | Use ASCII or a rendered SVG |",
+ "| SVG lands outside the documentation tree | Render again with an explicit output directory |",
+ "| New page does not appear in site navigation | Follow the site's navigation configuration and nearest page |",
+ "| Preview looks right but the build is red | Report failure; rendered appearance is not build verification |",
+ "| Tool named in the procedure is unavailable | State the limitation and use a verifiable fallback |",
+ ].join("\n"),
+ },
+ {
+ id: "mini-design-doc",
+ title: "Write a mini design doc",
+ description:
+ "Produce a transcript-first technical design narrative that takes under five minutes to read and ends with one clear recommendation.",
+ argument: {
+ label: "Subject",
+ placeholder: "whether session status should be joined into the notification centre",
+ hint: "One medium-sized decision. The result stays in this transcript and creates no files.",
+ required: true,
+ maxLength: 20_000,
+ },
+ injector: [
+ "Write a mini design doc for the subject above. If no subject is supplied, use",
+ "the medium-sized technical or product decision currently under discussion.",
+ "",
+ "This procedure is self-contained. Do not load or defer to a skill. The result",
+ "belongs in the current transcript, must take less than five minutes to read, and",
+ "must not create files or expand into a multi-artifact RFC unless explicitly",
+ "requested.",
+ "",
+ "Inspect the relevant code, callers, tests, contracts, and existing decisions",
+ "before recommending anything. Identify the single decision the reader needs to",
+ "make. Label load-bearing statements as `Verified`, `Inferred`, or `Proposed`",
+ "when prose alone could blur their status, and cite repository-relative",
+ "`path:line-line` or primary sources for claims that determine the decision.",
+ "",
+ "Use the smallest useful subset of this sequence:",
+ "",
+ "1. **Today / Problem:** concrete current behavior and the friction or failure.",
+ "2. **Proposed Experience or Design:** make the target state tangible.",
+ "3. **Flow:** trace the main user action, request, state, or data path.",
+ "4. **Rules and Boundaries:** accepted/rejected inputs, trust and authority,",
+ " state ownership, constraints, and the expensive direction to be wrong.",
+ "5. **Alternatives:** compare only credible options on decision-relevant axes.",
+ "6. **Why Not:** reject the most tempting oversized or unsafe choice with facts.",
+ "7. **Scope Split:** `Now`, `Later`, and `Non-goals`.",
+ "8. **Recommendation:** one clear sentence naming the central choice and why.",
+ "",
+ "Omit a section that adds no decision value. Do not fill headings to imitate",
+ "rigor, pad alternatives with strawmen, or repeat a diagram in prose.",
+ "",
+ "Use compact ASCII only when spatial grouping, sequence, side-by-side comparison,",
+ "architecture, or a UI sketch is faster to understand than prose. Prefer at most",
+ "one current/target comparison and one execution flow. Use realistic names,",
+ "values, and actions; annotate risks at the step where they occur; keep lines at",
+ "or below 100 columns when practical. Use a compact table for alternatives and a",
+ "short list for linear scope or acceptance rules.",
+ "",
+ "Explicitly cover security, mobile, accessibility, operations, compatibility,",
+ "and dependencies only when they can change the choice. Keep follow-up work",
+ "separate from the immediate recommendation. End with the recommendation, not a",
+ "generic summary or an open-ended list of options.",
+ ].join("\n"),
+ },
+ {
+ id: "system-design-artifacts",
+ title: "Build a system-design review package",
+ description:
+ "Assemble an evidence-led senior system-design review package, with every load-bearing claim tagged by its evidence class.",
+ argument: {
+ label: "Subject",
+ placeholder: "the notification delivery and suppression system, current-state",
+ hint: "Name the system or proposal, and say current-state, target-state, or mixed if you already know which you want.",
+ required: true,
+ maxLength: 20_000,
+ },
+ injector: [
+ "Create a senior-SWE system-design review package for the subject above. If no",
+ "subject is named, use the system or proposal currently under discussion.",
+ "",
+ "This procedure is self-contained. Do not load or defer to a skill.",
+ "",
+ "## Establish the assignment",
+ "",
+ "Before writing, state one mode:",
+ "",
+ "- `current-state`: explain only behavior supported by present evidence;",
+ "- `target-state`: design proposed behavior without presenting it as shipped;",
+ "- `mixed`: keep current and target views visibly separate in every artifact.",
+ "",
+ "Identify the audience, decision they need to make, repository boundary, and",
+ "whether the user requested transcript output, repository files, or publication.",
+ "Do not create files or publish anything unless requested. Never publish directly",
+ "to a default branch. If publication is requested, use a draft PR.",
+ "",
+ "## Audit evidence first",
+ "",
+ "Inspect the relevant implementation, callers, tests, fixtures, schemas/live",
+ "contracts, decisions, incidents, and operational configuration. Prefer the live",
+ "contract over secondary documentation when the repository says it is",
+ "authoritative. Do not run destructive probes, production operations, migrations,",
+ "or writes merely to strengthen a document. Ask before a probe that has cost,",
+ "external effects, credentials, or production reach.",
+ "",
+ "Tag important claims in notes and artifacts with exactly one evidence class:",
+ "",
+ "| Class | Meaning |",
+ "| --- | --- |",
+ "| `observed` | Reproduced by a named command or live probe, with environment and date |",
+ "| `code-supported` | Directly established by cited implementation or tests |",
+ "| `mock-only` | Demonstrated only by a fixture, fake, simulator, or test double |",
+ "| `inferred` | Best explanation from evidence, but not directly established |",
+ "| `unknown` | Evidence is absent, contradictory, inaccessible, or intentionally unprobed |",
+ "",
+ "Mocks never prove live upstream behavior. A test proves only the contract it",
+ "actually exercises. Keep proposed behavior out of current-state claims. Cite",
+ "repository-relative `path:line-line`, a command plus bounded output, or a primary",
+ "source URL for every load-bearing claim.",
+ "",
+ "## Build the system model",
+ "",
+ "Before selecting artifacts, identify:",
+ "",
+ "1. system and trust boundaries;",
+ "2. state owner for every important datum;",
+ "3. durable, process-local, derived, cached, and presentation-only state;",
+ "4. authority and permission checks at each mutation boundary;",
+ "5. concurrency units, serialization points, dedupe keys, replay, and idempotency;",
+ "6. lifecycle state separately from transport events and UI presentation;",
+ "7. failure boundaries, restart reconciliation, partial success, and rollback;",
+ "8. the expensive direction to be wrong for each uncertain decision.",
+ "",
+ "Write down the load-bearing invariants. Examples of useful invariant shapes are",
+ "\"absence from a process-local status map is not proof of idle\" and \"a browser",
+ "candidate grants no authority until the server validates it\". Use repository",
+ "facts, not these examples, in the package.",
+ "",
+ "## Select artifacts; do not generate a checklist blindly",
+ "",
+ "Start with a manifest and include an artifact only when it gives a distinct",
+ "review perspective:",
+ "",
+ "```yaml",
+ "mode: current-state | target-state | mixed",
+ "subject: ",
+ "audience: senior-swe",
+ "decision: ",
+ "evidence:",
+ " live_probes: required | optional | prohibited",
+ " destructive_actions: prohibited-by-default",
+ "artifacts:",
+ " - id: ",
+ " question: ",
+ " evidence_classes: [observed, code-supported]",
+ "omissions:",
+ " - artifact: ",
+ " reason: ",
+ "```",
+ "",
+ "Use this selection logic:",
+ "",
+ "| Review question | Prefer |",
+ "| --- | --- |",
+ "| Why does the system behave this way? | executive system guide or implementation RFC |",
+ "| What changes between now and target? | paired architecture/data-flow diagrams |",
+ "| Which transitions are legal? | state machine plus normative transition table |",
+ "| What crosses a boundary? | API contract, sequence diagram, and data model |",
+ "| Who owns and persists state? | ownership/persistence matrix |",
+ "| What happens on crash, retry, or restart? | failure/reconciliation diagram and catalogue |",
+ "| What can an attacker or confused deputy do? | threat model and security-test matrix |",
+ "| Why this choice? | focused ADRs for consequential alternatives only |",
+ "| How can this ship safely? | dependency graph, milestones, migration, rollout, rollback |",
+ "| How will operators know? | signals, SLOs, alerts, dashboards, and runbook |",
+ "| How can a reviewer safely verify it? | non-destructive lab or commands with expected output |",
+ "| Does interaction or motion carry the idea? | responsive HTML, accessible SVG, or short demo |",
+ "",
+ "Do not create interactive HTML, animation, video, or a machine-readable failure",
+ "catalogue unless the medium itself answers a review question. Decorative copies",
+ "of prose are omissions, not deliverables.",
+ "",
+ "## Artifact contracts",
+ "",
+ "Every selected artifact begins with: purpose, mode, evidence classes used,",
+ "authoritative sources, uncertainties, and links to related artifacts. Then apply",
+ "the relevant contract:",
+ "",
+ "- **Guide/RFC:** problem and decision first; explain causality, invariants,",
+ " boundaries, ownership, failure behavior, and consequences.",
+ "- **Diagram:** include a legend; label authority/state boundaries; annotate races",
+ " and failure points where they occur; pair mixed-mode views instead of blending",
+ " them.",
+ "- **State table:** name state owner and persistence; include trigger, guard,",
+ " transition, side effect, retry/idempotency behavior, invalid transition, and",
+ " restart outcome.",
+ "- **API/data contract:** include caller, authority, validation, request/response,",
+ " errors, idempotency, compatibility, limits, and redaction. Mark illustrative",
+ " schemas as proposals.",
+ "- **Ownership matrix:** include datum, authority, writer, readers, persistence,",
+ " cache/derivation, reconciliation, and deletion/retention.",
+ "- **Threat model:** identify assets, actors, entry points, trust boundaries,",
+ " abuse cases, mitigations, residual risk, and executable security checks.",
+ "- **ADR:** one decision and status; context, credible alternatives, choice,",
+ " consequences, reversibility, and evidence that would reopen it.",
+ "- **Implementation plan:** order by dependencies; name ownership, acceptance",
+ " evidence, rollout gates, migration, rollback, and intentionally deferred work.",
+ "- **Operations:** connect each SLO and alert to a user-visible failure, then give",
+ " diagnosis, safe mitigation, escalation, and recovery verification.",
+ "- **Failure catalogue:** stable scenario ID, preconditions, injection, expected",
+ " state/events, invariant, observability, cleanup, and safety classification.",
+ "- **Interactive artifact:** keyboard and touch operation, reduced motion,",
+ " semantic structure, no secret/live data, responsive checks, and a static",
+ " fallback carrying the same information.",
+ "",
+ "## Link and verify the package",
+ "",
+ "Create one index with a recommended review order. For each artifact list its",
+ "question, mode, evidence status, prerequisite, and unresolved gaps. Cross-link",
+ "concepts by stable anchors or relative paths. Do not make the reviewer hunt for",
+ "the current/target boundary or the source behind a claim.",
+ "",
+ "Verify only what exists:",
+ "",
+ "1. check every cited path and line range against the reviewed revision;",
+ "2. lint/parse Markdown, Mermaid, YAML, JSON, OpenAPI, and HTML with repository",
+ " tooling where available;",
+ "3. execute safe examples and contract validation in an isolated environment;",
+ "4. render diagrams and inspect labels, clipping, and legibility;",
+ "5. test interactive output at desktop and mobile widths, keyboard-only, screen",
+ " reader semantics, contrast, and reduced motion;",
+ "6. run relevant repository typechecks/tests and record exact commands/results;",
+ "7. list anything not run and why; never turn \"not checked\" into \"passed\".",
+ "",
+ "Return or commit only the requested output. If a draft PR was requested, ensure",
+ "it clearly says `Draft`, separates current facts from proposals, links the",
+ "package index, reports verification and gaps, includes human verification steps,",
+ "and does not claim that opening the PR deploys or operates the system.",
+ ].join("\n"),
+ },
+ {
+ id: "verify",
+ title: "Run the checks, then write verification steps",
+ description:
+ "Discover and run this repository's real verification contract, then write numbered human verification steps with expected results and failure signals.",
+ argument: {
+ label: "Surface to verify",
+ placeholder: "the notification popover on mobile",
+ hint: "Scopes the human checklist. The automated checks are discovered from the repository either way.",
+ required: true,
+ maxLength: 4_000,
+ },
+ injector: [
+ "Discover this repository's verification contract before running anything:",
+ "",
+ "1. Read its agent instructions and contribution guide.",
+ "2. Inspect changed files, package or task manifests, lockfiles, CI workflows,",
+ " and adjacent tests to identify the language, package manager, and required",
+ " checks. Do not infer npm merely because this is an OpenCode project.",
+ "3. Prefer documented repository commands. When none exist, choose the narrowest",
+ " standard checks supported by the detected tooling and state why.",
+ "4. Run the relevant focused checks, then the repository's required aggregate",
+ " typecheck, test, lint, and build checks when practical. Record every exact",
+ " command, exit status, and useful result.",
+ "5. Inspect `git status --short`, the diff against the actual base branch, and the",
+ " diff stat before writing the checklist.",
+ "",
+ "Never claim a check ran based on old CI, prompt text, or a remembered convention.",
+ "If any required check is red, stop and report the failure. Do not send a human",
+ "to verify a build that is already broken.",
+ "",
+ "If everything is green, write the human verification checklist for the change",
+ "shown in the diff, scoped to the surface named above when one is named:",
+ "",
+ "- 5 to 12 numbered steps, each with the action, the expected result, and the",
+ " failure signal.",
+ "- Name the exact URL, command, viewport, theme, or test data each step needs.",
+ "- Check what automated tests cannot: visual layout, interaction, keyboard",
+ " access, responsive behaviour, deployed behaviour.",
+ "- Cover a boundary, not only the happy path — empty state, error state, narrow",
+ " viewport, or reduced motion, whichever the diff actually touches.",
+ "- Separate VERIFIED, FAILED, and UNVERIFIED. Never report an unreachable",
+ " surface as passing.",
+ "- End with a disposition: Ready to ship, Fixes required, Partially verified, or",
+ " Blocked on human access.",
+ "",
+ "Research the changed routes, help text, API contracts, tests, start commands,",
+ "ports, fixtures, roles, flags, and deployment target yourself. Ask the user only",
+ "for access or product intent that the repository cannot establish. Do not use",
+ "implementation-detail checks; verify what a user can see or accomplish.",
+ "",
+ "A screenshot proves one visual instant, not focus movement, persistence, time,",
+ "keyboard operation, or error handling. Exercise those behaviors directly.",
+ "",
+ "After execution, list `VERIFIED`, `FAILED`, and `UNVERIFIED` separately, keeping",
+ "empty categories visible. End with exactly one disposition: **Ready to ship**,",
+ "**Fixes required**, **Partially verified**, or **Blocked on human access**.",
+ "",
+ "| Failure | Response |",
+ "|---|---|",
+ "| Repository has no documented check commands | Inspect its manifests and CI, run only checks supported by detected tooling, and explain the choice |",
+ "| Automation is red | Stop with Fixes required |",
+ "| Infrastructure prevents a check | Mark it UNVERIFIED, never passed |",
+ "| A step lacks URL, data, viewport, role, or expected output | Add the missing setup before handing it to a human |",
+ "| Checklist exceeds 12 steps | Remove automated or low-information duplication |",
+ ].join("\n"),
+ },
+ {
+ id: "leaving-now-wrap-up",
+ title: "Wrap up and leave an accurate status",
+ description:
+ "Stop this run safely, push authorized progress, refresh every status artifact, and end with one accurate merge verdict.",
+ argument: {
+ label: "Wrap-up note",
+ placeholder: "branch feat/planning-epic-hierarchy, PR #312 — post the update there",
+ hint: "May name the branch, pull request, issue, or human update destination. No destination is invented for you.",
+ required: true,
+ maxLength: 4_000,
+ },
+ injector: [
+ "Wrap up the current run now. The note above may name the branch, PR, issue, or",
+ "human update destination. Complete these steps in order; do not stop after a",
+ "chat-only summary.",
+ "",
+ "1. Stop work owned by this run. Cancel this run's background agents, test",
+ " runners, preview servers, and other processes that could keep mutating state.",
+ " Verify they stopped. Never kill a process merely because its name or port",
+ " looks familiar: establish that this run started or owns it. Leave unrelated",
+ " user and sibling-worktree processes alone.",
+ "2. Inspect the repository root, branch, worktree, status, diff, staged diff, and",
+ " recent commits. Separate this run's intended changes from pre-existing or",
+ " unrelated user changes. Scan intended paths for credentials and generated",
+ " secret files. Never stage or commit secrets, and never absorb unrelated",
+ " changes just to make the tree clean.",
+ "3. Run the most relevant affordable verification for this run's changes. Record",
+ " each check as green, red, or not run, including the command and concise reason.",
+ " A red check does not justify hiding otherwise useful progress.",
+ "4. Preserve real progress under the repository's permissions and the user's",
+ " explicit commit/push instructions. If authorized, stage only owned files,",
+ " commit them, and push the current feature branch even when verification is",
+ " red, with the failure stated plainly. Never push `main` or another protected",
+ " default branch. Do not force-push. If commit or push is not authorized, is",
+ " rejected, or would include unsafe/unrelated content, leave it uncommitted or",
+ " local and report that fact as a blocker instead of bypassing the gate.",
+ "5. Refresh every project-defined status artifact with the current UTC timestamp,",
+ " exact branch/commit/push state, verification results, blockers, and next",
+ " action. Supersede stale claims. Reconcile the task list immediately: completed",
+ " work is completed, intentionally dropped work is cancelled, and only genuine",
+ " remaining work stays pending. If a status artifact is tracked, include its",
+ " final update in the authorized push, using a small follow-up commit when the",
+ " progress commit already exists. Do not let status sync create new local-only",
+ " state.",
+ "6. Post one human-readable update to each destination the project explicitly",
+ " defines and this run is authorized to use, such as its PR or issue. Do not",
+ " invent a destination or leak details to a public channel. Include what is",
+ " green and red, why anything is red, what is committed and pushed versus only",
+ " local, open decisions requiring a human, and the next command or owner.",
+ "7. If there is a PR, end the posted update and final response with exactly one",
+ " accurate verdict: `SAFE TO MERGE` only when the intended change is pushed,",
+ " required checks are green, and no blocking decision remains; otherwise `DO",
+ " NOT MERGE`, followed by the blocking facts. Re-read remote/PR state after the",
+ " push before choosing the verdict.",
+ "",
+ "Finish with the branch, pushed commit (or `not pushed`), stopped-work evidence,",
+ "verification matrix, status destinations updated, open decisions, and merge",
+ "verdict. Accuracy is more important than making the wrap-up look green.",
+ ].join("\n"),
+ },
+ {
+ id: "standup",
+ title: "Write a standup update",
+ description:
+ "Gather the last day's commits and pull requests, then turn them into a three-section standup update written to be read aloud.",
+ argument: {
+ label: "Scope",
+ placeholder: "everything, or just the notification work",
+ hint: "Narrows the update to one project or topic. The commit and pull request data is gathered by the agent, not pre-fetched.",
+ required: true,
+ maxLength: 2_000,
+ },
+ injector: [
+ "Turn the last day's work into a standup update.",
+ "",
+ "Nothing is pre-fetched for you. Gather the data yourself by running these three",
+ "commands, then use only what they actually returned:",
+ "",
+ " git log --all --author=\"$(git config user.email)\" --since=\"24 hours ago\" --pretty=format:'%h %s' --no-merges",
+ " gh pr list --author \"@me\" --state merged --limit 10 --json number,title,mergedAt -q '.[] | \"#\\(.number) \\(.title)\"'",
+ " gh pr list --author \"@me\" --state open --limit 10 --json number,title,isDraft -q '.[] | \"#\\(.number)\\(if .isDraft then \" DRAFT\" else \"\" end) \\(.title)\"'",
+ "",
+ "This workflow is sent in this session's current mode. In a Plan session bash is",
+ "likely denied, so these commands may not run at all. If they do not, say so",
+ "plainly and stop rather than reconstructing a standup from memory or from this",
+ "transcript and presenting it as today's commits. If `gh` is missing or",
+ "unauthenticated, report the pull request sections as unavailable instead of",
+ "inventing them.",
+ "",
+ "Write the standup update from that output, in three sections:",
+ "",
+ "**Yesterday** — what actually landed, grouped by theme rather than listed per",
+ "commit. Say what it does for a reader, not what the diff touched.",
+ "",
+ "**Today** — what the open work implies is next. Mark anything that is a guess.",
+ "",
+ "**Blocked** — only genuine blockers. If there are none, say \"nothing blocked\"",
+ "rather than inventing one.",
+ "",
+ "Rules:",
+ "",
+ "- Three to six bullets per section. This gets read aloud.",
+ "- No commit hashes and no branch names unless someone would need to go find it.",
+ "- If the log is empty, say so plainly. Do not pad it out of the PR list.",
+ "- The scope note above may narrow this to one project or topic; if it does,",
+ " drop everything else.",
+ "",
+ "This is deliberately self-contained and adds no retrieval context until a human",
+ "invokes it.",
+ ].join("\n"),
+ },
];
+/**
+ * Every declared argument bound before the catalogue is observable, so the
+ * clamp cannot be skipped by a caller reading the presets directly.
+ */
+const CATALOGUE: WorkflowPreset[] = RAW_CATALOGUE.map((preset) => preset.argument
+ ? {
+ ...preset,
+ argument: {
+ ...preset.argument,
+ maxLength: Math.max(1, Math.min(preset.argument.maxLength, WORKFLOW_ARGUMENT_MAX_LENGTH)),
+ },
+ }
+ : preset);
+
export function workflowCatalogue(): WorkflowPreset[] {
return CATALOGUE;
}
diff --git a/tests/e2e/playbooks.ui.spec.ts b/tests/e2e/playbooks.ui.spec.ts
index e2e2939e..49a3ada2 100644
--- a/tests/e2e/playbooks.ui.spec.ts
+++ b/tests/e2e/playbooks.ui.spec.ts
@@ -1,16 +1,11 @@
import { expect, test } from "@playwright/test";
-import { readdirSync } from "node:fs";
-// The mock canonicalizes its fixture directory, so macOS needs the /private
-// spelling or the catalogue resolves to a different project.
-const DIR = process.platform === "darwin" ? "/private/tmp/mock-project" : "/tmp/mock-project";
-// Derived from the catalogue on disk rather than a fixed list or total, so a
-// command added by any branch is covered without editing this assertion.
-const COMMAND_NAMES = readdirSync(new URL("../../agent-skills/commands/", import.meta.url), {
- withFileTypes: true,
-})
- .filter((entry) => entry.isFile() && entry.name.endsWith(".md"))
- .map((entry) => entry.name.replace(/\.md$/, ""));
+// A stable workflow to use as the modal fixture. These tests are about the
+// dialog — focus, scroll lock, dismissal, light mode, phone width — and used to
+// open a command detail page to get one. The command catalogue is retired, so
+// they open a workflow detail page instead; the behaviour under test is the
+// same modal component.
+const FIXTURE = "/playbooks/workflows/verify";
test.describe("Playbooks", () => {
test("is first-class navigation on the bar beside Planning", async ({ page }) => {
@@ -22,50 +17,50 @@ test.describe("Playbooks", () => {
await expect(page).toHaveTitle("Playbooks | DCA");
await expect(page.getByTestId("opencode-playbooks-wip-warning")).toHaveText("Playbooks is still work in progress and its UI/UX may contain bugs.");
await expect(page.getByRole("heading", { name: "Repeatable work, invoked on purpose." })).toBeVisible();
- expect(COMMAND_NAMES).toContain("goal");
- expect(COMMAND_NAMES).toContain("system-design-artifacts");
- for (const command of COMMAND_NAMES) {
- await expect(page.getByTestId(`opencode-playbook-command-${command}`)).toBeVisible();
- }
// Workflows are the live server category, so this count is the workflow
- // catalogue's own contract, never a repository command inventory.
- await expect(page.getByTestId("opencode-playbook-workflow-card")).toHaveCount(6);
+ // catalogue's own contract. It is deliberately exact: the retired command
+ // inventory used to be derived from disk, but this one has to match what
+ // the server actually served.
+ await expect(page.getByTestId("opencode-playbook-workflow-card")).toHaveCount(22);
+ await expect(page.getByTestId("opencode-playbook-workflow-verify")).toBeVisible();
+ await expect(page.getByTestId("opencode-playbook-workflow-system-design-artifacts")).toBeVisible();
});
- test("presents commands as human-invoked with zero at-rest context", async ({ page }) => {
+ test("presents workflows as guided actions with zero at-rest context", async ({ page }) => {
await page.goto("/playbooks");
- const command = page.getByTestId("opencode-playbook-command-card").first();
- await expect(command).toContainText("human-invoked");
- await expect(command).toHaveAttribute("data-playbook-kind", "command");
+ const workflow = page.getByTestId("opencode-playbook-workflow-card").first();
+ await expect(workflow).toContainText("guided action");
+ await expect(workflow).toHaveAttribute("data-playbook-kind", "workflow");
await expect(page.getByText("At-rest tokens")).toBeVisible();
});
- test("filters commands and opens one with simulation and installation", async ({ page }) => {
- await page.goto("/playbooks/commands");
- await page.getByTestId("opencode-playbook-filter").fill("design tree");
- await expect(page.getByTestId("opencode-playbook-command-card")).toHaveCount(1);
+ test("filters workflows and opens one with its argument and exact injector", async ({ page }) => {
+ await page.goto("/playbooks/workflows");
+ await page.getByTestId("opencode-playbook-filter").fill("standup update");
+ await expect(page.getByTestId("opencode-playbook-workflow-card")).toHaveCount(1);
- await page.getByTestId("opencode-playbook-command-grill-me").click();
- await expect(page).toHaveURL("/playbooks/commands/grill-me");
- await expect(page).toHaveTitle("Command | Playbooks | DCA");
+ await page.getByTestId("opencode-playbook-workflow-standup").click();
+ await expect(page).toHaveURL("/playbooks/workflows/standup");
+ await expect(page).toHaveTitle("Workflow | Playbooks | DCA");
await expect(page.getByTestId("opencode-playbook-dialog")).toBeVisible();
await expect(page.getByTestId("opencode-playbook-dialog").getByTestId("opencode-playbooks-wip-warning")).toHaveText("Playbooks is still work in progress and its UI/UX may contain bugs.");
await expect(page.getByTestId("opencode-playbooks")).toBeVisible();
- await expect(page.getByTestId("opencode-playbook-simulation")).toBeVisible();
- await page.getByText("Install /grill-me", { exact: true }).click();
- await expect(page.getByText("curl -sL", { exact: false }).first()).toBeVisible();
+ // What it collects is part of reading it.
+ await expect(page.getByTestId("opencode-playbook-workflow-input")).toContainText("Scope");
+ await expect(page.getByTestId("opencode-playbook-workflow-input")).toContainText("becomes the prompt");
+ // The three shell interpolations the retired command relied on have no
+ // workflow equivalent, so the injector must tell the agent to run them.
+ await expect(page.getByTestId("opencode-playbook-workflow-injector")).toContainText("Nothing is pre-fetched for you");
});
test("renders live workflows by shared semantic group and shows the exact injector", async ({ page }) => {
await page.goto("/playbooks/workflows");
await expect(page).toHaveTitle("Workflows | Playbooks | DCA");
- await expect(page.getByTestId("opencode-playbook-workflow-group")).toHaveCount(3);
- await expect(page.getByTestId("opencode-playbook-workflow-card")).toHaveCount(6);
+ await expect(page.getByTestId("opencode-playbook-workflow-group")).toHaveCount(6);
+ await expect(page.getByTestId("opencode-playbook-workflow-card")).toHaveCount(22);
await page.getByTestId("opencode-playbook-workflow-start-dca-session").click();
await expect(page).toHaveTitle("Workflow | Playbooks | DCA");
await expect(page.getByTestId("opencode-playbook-workflow-injector")).toContainText("independent root session");
- await expect(page.getByTestId("opencode-playbook-dialog").getByTestId("opencode-playbook-command-load-state")).toHaveCount(0);
- await expect(page.getByTestId("opencode-playbook-source-link")).toHaveCount(0);
});
test("keeps workflow loading, failure, empty, and not-found states honest", async ({ page }) => {
@@ -84,10 +79,9 @@ test.describe("Playbooks", () => {
await expect(page.getByTestId("opencode-playbook-workflow-error")).toBeVisible();
await expect(page.getByTestId("opencode-playbook-workflow-not-found")).toHaveCount(0);
await page.goto("/playbooks");
+ // With commands gone the catalogue has nothing to fall back to, so the
+ // failure must be stated rather than leaving an empty page.
await expect(page.getByTestId("opencode-playbook-workflows-error")).toBeVisible();
- // The static commands survive a workflow failure. Derived from disk like
- // the coverage assertion above, so adding a command never edits this.
- await expect(page.getByTestId("opencode-playbook-command-card")).toHaveCount(COMMAND_NAMES.length);
await page.unroute("**/api/workflows");
await page.route("**/api/workflows", (route) => route.fulfill({ json: { workflows: [] } }));
@@ -95,7 +89,7 @@ test.describe("Playbooks", () => {
await expect(page.getByTestId("opencode-playbook-workflows-empty")).toBeVisible();
});
- test("puts unknown live workflows in Other and never asks for command install state on workflow-only routes", async ({ page }) => {
+ test("puts unknown live workflows in Other and never asks for a project catalogue", async ({ page }) => {
const requests: string[] = [];
page.on("request", (request) => requests.push(new URL(request.url()).pathname));
await page.route("**/api/workflows", (route) => route.fulfill({ json: { workflows: [{ id: "future-action", title: "Future action", description: "Arrived from the server.", injector: "Use the future procedure exactly." }] } }));
@@ -103,12 +97,14 @@ test.describe("Playbooks", () => {
const group = page.getByTestId("opencode-playbook-workflow-group");
await expect(group).toHaveAccessibleName("Other");
await expect(group).toContainText("Future action");
+ // Playbooks reports a repository-owned, server-supplied catalogue. It has
+ // no per-project installation question left to ask.
expect(requests).not.toContain("/api/catalog");
});
- test("closes a direct detail URL back to the all-playbooks catalogue", async ({ page }) => {
+ test("closes a direct detail URL back to the workflow catalogue", async ({ page }) => {
await page.setViewportSize({ width: 1280, height: 800 });
- await page.goto("/playbooks/commands/diagram");
+ await page.goto(FIXTURE);
const dialog = page.getByTestId("opencode-playbook-dialog");
await expect(dialog).toBeVisible();
const box = await dialog.boundingBox();
@@ -116,30 +112,30 @@ test.describe("Playbooks", () => {
expect(Math.abs((box!.x + box!.width / 2) - 640)).toBeLessThanOrEqual(2);
expect(Math.abs((box!.y + box!.height / 2) - 400)).toBeLessThanOrEqual(2);
await dialog.getByRole("button", { name: "Close playbook" }).click();
- await expect(page).toHaveURL("/playbooks");
+ await expect(page).toHaveURL("/playbooks/workflows");
await expect(dialog).toHaveCount(0);
});
test("dismisses with Escape and with a backdrop click", async ({ page }) => {
await page.setViewportSize({ width: 1280, height: 800 });
- await page.goto("/playbooks/commands/diagram");
+ await page.goto(FIXTURE);
const dialog = page.getByTestId("opencode-playbook-dialog");
await expect(dialog).toBeVisible();
await page.keyboard.press("Escape");
- await expect(page).toHaveURL("/playbooks");
+ await expect(page).toHaveURL("/playbooks/workflows");
await expect(dialog).toHaveCount(0);
- await page.goto("/playbooks/commands/diagram");
+ await page.goto(FIXTURE);
await expect(dialog).toBeVisible();
// Click the ::backdrop region, which is the dialog element itself.
await page.mouse.click(6, 6);
- await expect(page).toHaveURL("/playbooks");
+ await expect(page).toHaveURL("/playbooks/workflows");
await expect(dialog).toHaveCount(0);
});
test("keeps focus in the document after the modal closes", async ({ page }) => {
await page.setViewportSize({ width: 1280, height: 800 });
- await page.goto("/playbooks/commands/diagram");
+ await page.goto(FIXTURE);
const dialog = page.getByTestId("opencode-playbook-dialog");
await expect(dialog).toBeVisible();
await expect(dialog.getByTestId("opencode-playbook-close")).toBeFocused();
@@ -150,77 +146,37 @@ test.describe("Playbooks", () => {
expect(await page.evaluate(() => document.activeElement?.tagName)).not.toBe("BODY");
});
- test("puts simulation controls above the transcript and locks background scroll", async ({ page }) => {
+ test("locks background scroll while the modal is open and restores it after", async ({ page }) => {
await page.setViewportSize({ width: 1280, height: 800 });
- await page.goto("/playbooks/commands/grill-me");
- const simulation = page.getByTestId("opencode-playbook-simulation");
- const context = page.getByTestId("opencode-playbook-context");
- await expect(simulation).toBeVisible();
+ await page.goto(FIXTURE);
+ await expect(page.getByTestId("opencode-playbook-dialog")).toBeVisible();
expect(await page.evaluate(() => getComputedStyle(document.body).overflow)).toBe("hidden");
-
- const controls = simulation.getByTestId("opencode-playbook-simulation-controls");
- const transcript = simulation.locator("ol").first();
- const [controlsBox, transcriptBox, contextBox, simulationBox] = await Promise.all([controls.boundingBox(), transcript.boundingBox(), context.boundingBox(), simulation.boundingBox()]);
- expect(controlsBox!.y, "controls sit above the transcript they drive").toBeLessThan(transcriptBox!.y);
- expect(simulationBox!.y, "simulation sits below Context").toBeGreaterThan(contextBox!.y + contextBox!.height);
- expect(Math.abs(simulationBox!.x - contextBox!.x), "simulation shares Context's full width").toBeLessThanOrEqual(1);
- expect(Math.abs(simulationBox!.width - contextBox!.width), "simulation shares Context's full width").toBeLessThanOrEqual(1);
-
- // The progress bar is always mounted, so play/pause cannot shift layout.
- await expect(simulation.getByTestId("opencode-playbook-simulation-progress")).toBeAttached();
- await simulation.getByTestId("opencode-playbook-simulation-reset").click();
- await expect(simulation.getByTestId("opencode-playbook-simulation-status")).toContainText("frame 1 of");
- await expect(simulation.getByTestId("opencode-playbook-simulation-reset")).toBeDisabled();
- await simulation.getByTestId("opencode-playbook-simulation-next").click();
- await expect(simulation.getByTestId("opencode-playbook-simulation-status")).toContainText("frame 2 of");
+ await page.getByTestId("opencode-playbook-close").click();
+ await expect(page.getByTestId("opencode-playbook-dialog")).toHaveCount(0);
+ expect(await page.evaluate(() => getComputedStyle(document.body).overflow)).not.toBe("hidden");
});
- test("names the project when reporting load state, and claims nothing without one", async ({ page }) => {
- // No project selected yet: the page must make no installation claim at all
- // rather than implying "not installed".
- await page.goto("/playbooks");
- await page.evaluate(() => localStorage.removeItem("opencode.directory.v1"));
- await page.reload();
- await expect(page.getByTestId("opencode-playbook-command-card").first()).toBeVisible();
- await expect(page.getByTestId("opencode-playbook-command-load-state")).toHaveCount(0);
-
- // With a project, every claim carries the project name — installation is
- // per-directory while this page is global.
- await page.goto(`/playbooks?directory=${encodeURIComponent(DIR)}`);
- const state = page.getByTestId("opencode-playbook-command-load-state").first();
- await expect(state).toBeVisible();
- await expect(state).toContainText("mock-project");
- await expect(state).toHaveAttribute("data-installed", /true|false/);
-
- const loaded = page.locator('[data-testid="opencode-playbook-command-load-state"][data-installed="true"]');
- const notLoaded = page.locator('[data-testid="opencode-playbook-command-load-state"][data-installed="false"]');
- expect(await loaded.count() + await notLoaded.count()).toBeGreaterThan(0);
- });
-
- test("says plainly that viewing or copying installs nothing", async ({ page }) => {
- await page.goto("/playbooks/commands/grill-me");
- await expect(page.getByTestId("opencode-playbook-scope-note")).toContainText("does not install anything");
- await expect(page.getByTestId("opencode-playbook-scope-note")).toContainText("does not attach anything to a conversation");
-
- // The install disclosure must not imply the app performs the install.
- await page.getByText("Install /grill-me", { exact: true }).click();
- await expect(page.getByTestId("opencode-playbook-install-note")).toContainText("does not install anything");
-
- // Source links name the revision they follow rather than implying a pin.
- await expect(page.getByTestId("opencode-playbook-source-link")).toContainText("main");
+ test("says plainly that viewing or copying runs nothing, and names the mode risk", async ({ page }) => {
+ await page.goto(FIXTURE);
+ const note = page.getByTestId("opencode-playbook-scope-note");
+ await expect(note).toContainText("does not run, attach, or install anything");
+ // A workflow has no `agent:` frontmatter to pin its mode, unlike the
+ // command it replaced. The page has to say so rather than let a reader
+ // assume the old guarantee survived.
+ await expect(note).toContainText("current mode");
});
- test("keeps terminal inline code legible in light mode", async ({ page }) => {
- // The terminal keeps its dark surface in both themes, but the shared inline
- // code rule follows the theme — in light mode that painted a near-white
- // block on the dark panel and hid the text entirely.
+ test("keeps the injector's code legible in light mode", async ({ page }) => {
+ // The injector panel keeps its dark surface in both themes, but the shared
+ // inline code rule follows the theme — in light mode that painted a
+ // near-white block on the dark panel and hid the text entirely.
await page.addInitScript(() => localStorage.setItem("theme", "light"));
await page.emulateMedia({ colorScheme: "light" });
- await page.goto("/playbooks/commands/grill-me");
- const code = page.getByTestId("opencode-playbook-simulation").locator("code").first();
+ await page.goto(FIXTURE);
+ const code = page.getByTestId("opencode-playbook-workflow-injector").locator("code").first();
await expect(code).toBeVisible();
const { background, colour } = await code.evaluate((node) => ({
- background: getComputedStyle(node).backgroundColor,
+ background: getComputedStyle(node.parentElement ?? node).backgroundColor,
colour: getComputedStyle(node).color,
}));
const luminance = (value: string) => {
@@ -232,20 +188,13 @@ test.describe("Playbooks", () => {
expect(Math.abs(luminance(background) - luminance(colour))).toBeGreaterThan(60);
});
- test("states the caveat exactly once", async ({ page }) => {
- await page.goto("/playbooks/commands/grill-me");
- await expect(page.getByTestId("opencode-playbook-dialog")).toBeVisible();
- await expect(page.getByText("Caveat:", { exact: false })).toHaveCount(1);
- await expect(page.getByText("Simulation disclosure")).toHaveCount(0);
- });
-
- test("shows the complete command procedure and remains usable on a phone", async ({ page }) => {
+ test("shows the complete workflow procedure and remains usable on a phone", async ({ page }) => {
await page.setViewportSize({ width: 390, height: 740 });
- await page.goto("/playbooks/commands/verify");
+ await page.goto(FIXTURE);
- await expect(page).toHaveTitle("Command | Playbooks | DCA");
- await expect(page.getByTestId("opencode-playbook-dialog").getByRole("heading", { name: "/verify " })).toBeVisible();
- await page.getByText("Command template", { exact: true }).click();
+ await expect(page).toHaveTitle("Workflow | Playbooks | DCA");
+ await expect(page.getByTestId("opencode-playbook-dialog").getByRole("heading", { name: "Run the checks, then write verification steps" })).toBeVisible();
+ // The full ported procedure is present, tables and all.
await expect(page.getByTestId("opencode-playbook-dialog")).toContainText("Failure");
expect(await page.evaluate(() => document.documentElement.scrollWidth)).toBeLessThanOrEqual(390);
@@ -253,9 +202,5 @@ test.describe("Playbooks", () => {
const box = await page.getByTestId("opencode-playbook-dialog").boundingBox();
expect(box!.width).toBeGreaterThanOrEqual(389);
expect(box!.height).toBeGreaterThanOrEqual(739);
- const contextBox = await page.getByTestId("opencode-playbook-context").boundingBox();
- const simulationBox = await page.getByTestId("opencode-playbook-simulation").boundingBox();
- expect(simulationBox!.y).toBeGreaterThan(contextBox!.y + contextBox!.height);
- expect(Math.abs(simulationBox!.width - contextBox!.width)).toBeLessThanOrEqual(1);
});
});
diff --git a/tests/e2e/smoke.ui.spec.ts b/tests/e2e/smoke.ui.spec.ts
index 530ffa47..f5b10a7a 100644
--- a/tests/e2e/smoke.ui.spec.ts
+++ b/tests/e2e/smoke.ui.spec.ts
@@ -1119,11 +1119,14 @@ test.describe("composer", () => {
await page.goto(`/sessions/ses_mock_done?directory=${encodeURIComponent(DIR)}`);
await page.getByTestId("composer-workflow-select").click();
await expect(page.getByTestId("composer-workflow-search")).toBeFocused();
- await expect(page.getByTestId("composer-workflow-option")).toHaveCount(6);
+ await expect(page.getByTestId("composer-workflow-option")).toHaveCount(22);
// Decision 21: this promise must survive the header gaining a search box.
await expect(page.getByTestId("composer-workflow-panel")).toContainText("Nothing is sent or launched until you confirm.");
- await page.getByTestId("composer-workflow-search").fill("pull request");
+ // Search matters more now that the catalogue is 22 entries rather than six,
+ // so the needle has to be specific: "pull request" alone also describes the
+ // review-learning workflow.
+ await page.getByTestId("composer-workflow-search").fill("snippet-by-snippet");
await expect(page.getByTestId("composer-workflow-option")).toHaveCount(1);
await expect(page.getByTestId("composer-workflow-option")).toContainText("Post a snippet-by-snippet PR review");
@@ -1159,7 +1162,10 @@ test.describe("composer", () => {
const humanVerification = page.locator('[data-testid="composer-reminder-option"][data-reminder-id="human-verification-steps"]');
await expect(humanVerification).toHaveAccessibleName("Attach Write Human Verification Steps");
const details = page.locator('[data-testid="composer-reminder-details"][data-reminder-id="human-verification-steps"]');
- await expect(details).toHaveAttribute("href", "/playbooks/commands/verify");
+ // The retired command catalogue used to host this documentation; the
+ // ported workflow does now, and the link has to follow it rather than keep
+ // resolving to a route that no longer exists.
+ await expect(details).toHaveAttribute("href", "/playbooks/workflows/verify");
await expect(details).toHaveAttribute("target", "_blank");
await expect(details).toHaveAccessibleName("Open Write Human Verification Steps details in a new tab");
// The details link is a real touch target in its own right (matching
@@ -1172,6 +1178,11 @@ test.describe("composer", () => {
expect(detailsBox?.width, "the details link stays a minority of the tile, not the whole row").toBeLessThan((tileBox?.width ?? 0) / 2);
const unknown = page.locator('[data-testid="composer-reminder-tile"][data-reminder-id="new-server-reminder"]');
await expect(unknown.getByTestId("composer-reminder-details")).toHaveCount(0);
+ // Half of the old reminder-to-command links have no workflow to point at,
+ // because those commands were deleted rather than converted: the reminder
+ // beside them already said everything they said. Those tiles render no link
+ // at all rather than one that goes somewhere merely adjacent.
+ await expect(page.locator('[data-testid="composer-reminder-tile"][data-reminder-id="cite-file-lines"]').getByTestId("composer-reminder-details")).toHaveCount(0);
await unknown.getByTestId("composer-reminder-option").click();
await expect(picker).toHaveAttribute("value", "new-server-reminder");
await picker.click();
diff --git a/tests/e2e/workflows.ui.spec.ts b/tests/e2e/workflows.ui.spec.ts
index 7347dd0f..6d0cff01 100644
--- a/tests/e2e/workflows.ui.spec.ts
+++ b/tests/e2e/workflows.ui.spec.ts
@@ -41,17 +41,50 @@ test.describe("workflow catalogue API", () => {
expect(payload.workflows.map((workflow) => workflow.id)).toEqual([
"playwright-ui-review",
"pr-snippet-review",
+ "red-team",
+ "review-learning",
"session-update",
"managed-child",
"start-dca-session",
+ "manager-children",
+ "native-worktree-subagents",
+ "session-handoff",
+ "goal",
+ "dca",
+ "worktree-up",
+ "deep-research",
+ "research-handoff",
"design-doc-prototype",
+ "docs-preview",
+ "mini-design-doc",
+ "system-design-artifacts",
+ "verify",
+ "leaving-now-wrap-up",
+ "standup",
]);
+ // The four core keys are required of every workflow; `argument` and
+ // `prompt` are optional additions. This is deliberately not a bare superset
+ // check: an unexpected key would mean the projection started leaking
+ // catalogue internals, which is the failure worth catching.
+ const ALLOWED = new Set(["description", "id", "injector", "title", "argument", "prompt"]);
for (const workflow of payload.workflows) {
- expect(Object.keys(workflow).sort()).toEqual(["description", "id", "injector", "title"]);
+ expect(Object.keys(workflow)).toEqual(expect.arrayContaining(["description", "id", "injector", "title"]));
+ for (const key of Object.keys(workflow)) expect(ALLOWED.has(key), `${String(workflow.id)} exposes ${key}`).toBe(true);
expect(String(workflow.injector).length).toBeGreaterThan(0);
}
const sessionUpdate = payload.workflows.find((workflow) => workflow.id === "session-update")!;
expect(String(sessionUpdate.injector)).toContain("204");
+
+ // A workflow whose typed text becomes the prompt has to describe its field,
+ // and the length it advertises has to be one the prompt route will accept.
+ const standup = payload.workflows.find((workflow) => workflow.id === "standup")!;
+ expect(standup.argument).toMatchObject({ label: "Scope", required: true });
+ expect((standup.argument as { maxLength: number }).maxLength).toBeLessThanOrEqual(100_000);
+ expect(String(standup.injector)).toContain("Nothing is pre-fetched for you");
+ // A workflow that collects nothing carries its fixed prompt instead.
+ const designDoc = payload.workflows.find((workflow) => workflow.id === "design-doc-prototype")!;
+ expect(designDoc.argument).toBeUndefined();
+ expect(designDoc.prompt).toBe("Capture a durable design prototype for this proposal and publish it for review.");
});
test("rejects unknown and malformed workflow ids at prompt time", async ({ request }) => {
@@ -120,18 +153,25 @@ test.describe("workflow picker UI", () => {
await expect(picker).toContainText("Workflows");
await picker.click();
- // The chooser offers exactly the initial catalogue, in order.
+ // The chooser offers exactly the catalogue, in group order, with every
+ // workflow placed — nothing may fall into the "Other" bucket, which exists
+ // for a workflow a newer server ships and not for one this build forgot.
const options = page.getByTestId("composer-workflow-option");
- await expect(options).toHaveCount(6);
- await expect(page.getByTestId("composer-workflow-group")).toHaveCount(3);
- await expect(page.getByTestId("composer-workflow-group").nth(0)).toHaveAccessibleName("Review");
- await expect(page.getByTestId("composer-workflow-icon")).toHaveCount(6);
+ await expect(options).toHaveCount(22);
+ const groups = page.getByTestId("composer-workflow-group");
+ await expect(groups).toHaveCount(6);
+ for (const [index, label] of ["Review", "Coordinate", "Execute", "Investigate", "Document", "Ship"].entries()) {
+ await expect(groups.nth(index)).toHaveAccessibleName(label);
+ }
+ await expect(page.getByTestId("composer-workflow-icon")).toHaveCount(22);
await expect(options.nth(0)).toContainText("Review a UI change with Playwright");
await expect(options.nth(1)).toContainText("Post a snippet-by-snippet PR review");
- await expect(options.nth(2)).toContainText("Send an update to another session");
- await expect(options.nth(3)).toContainText("Launch a Managed Child");
- await expect(options.nth(4)).toContainText("Start a DCA session");
- await expect(options.nth(5)).toContainText("Capture a Durable Design Prototype");
+ await expect(options.nth(2)).toContainText("Red-team the work just produced");
+ await expect(options.nth(4)).toContainText("Send an update to another session");
+ await expect(options.nth(5)).toContainText("Launch a Managed Child");
+ await expect(options.nth(6)).toContainText("Start a DCA session");
+ await expect(options.nth(15)).toContainText("Capture a Durable Design Prototype");
+ await expect(options.nth(21)).toContainText("Write a standup update");
// Choosing a workflow opens its form — it never sends.
await page.locator('[data-testid="composer-workflow-option"][data-workflow-id="playwright-ui-review"]').click();
@@ -185,6 +225,49 @@ test.describe("workflow picker UI", () => {
expect(text).toContain("Never regenerate the complete screenshot set.");
});
+ test("generic argument: typed text is the prompt, and the preview names the mode", async ({ page }) => {
+ const marker = `WF-ARG-${Date.now()}`;
+ await page.goto(mainSession);
+ await page.getByTestId("composer-workflow-select").click();
+ await page.locator('[data-testid="composer-workflow-option"][data-workflow-id="verify"]').click();
+ const dialog = page.getByTestId("composer-workflow-dialog");
+ await expect(dialog).toHaveAttribute("data-workflow", "verify");
+
+ // One generic field, focused, described by the server's own spec — no
+ // bespoke branch exists for this workflow in the dialog.
+ const field = dialog.getByTestId("composer-workflow-field-argument");
+ await expect(field).toBeFocused();
+ await expect(field).toHaveAttribute("placeholder", /notification popover/u);
+ await expect(dialog.getByTestId("composer-workflow-argument-hint")).toContainText("Scopes the human checklist");
+ // Required means required: an empty field cannot reach the preview.
+ await expect(dialog.getByTestId("composer-workflow-preview")).toBeDisabled();
+ await field.fill(` ${marker} `);
+ await expect(dialog.getByTestId("composer-workflow-preview")).toBeEnabled();
+ await dialog.getByTestId("composer-workflow-preview").click();
+
+ // The typed text IS the prompt, trimmed and otherwise untouched.
+ await expect(dialog.getByTestId("composer-workflow-prompt-preview")).toHaveText(marker);
+ await expect(dialog.getByTestId("composer-workflow-injector")).toContainText('server-resolved from id "verify"');
+ await expect(dialog.getByTestId("composer-workflow-injector")).toContainText("Ready to ship");
+ // The ported command pinned its own agent in frontmatter; a workflow cannot,
+ // so the preview has to say what governs instead of dropping it silently.
+ await expect(dialog.getByTestId("composer-workflow-mode-note")).toContainText("Sent in this session's current mode");
+ expect(await promptPayloadContaining(marker)).toBeUndefined();
+
+ await expect(dialog.getByTestId("composer-workflow-apply")).toBeVisible();
+ await dialog.getByTestId("composer-workflow-send").click();
+ await expect(dialog).toHaveCount(0);
+
+ const payload = await expectPromptPayloadContaining(marker);
+ expect(payload!.sessionID).toBe(MAIN);
+ const text = promptText(payload!);
+ expect(text).toContain(marker);
+ expect(text).toContain('');
+ expect(text).toContain("Ready to ship");
+ // The command's own substitution token must not survive the port.
+ expect(text).not.toContain("$ARGUMENTS");
+ });
+
test("design prototype: no fields, fixed prompt, sends into this session", async ({ page }) => {
// The prompt is a fixed constant rather than a marker-bearing draft, so
// this counts matching payloads instead of asserting absence — a retry of
diff --git a/tests/playbook-install-state.test.ts b/tests/playbook-install-state.test.ts
deleted file mode 100644
index 26e6c6e3..00000000
--- a/tests/playbook-install-state.test.ts
+++ /dev/null
@@ -1,56 +0,0 @@
-import { describe, expect, it } from "vitest";
-
-import type { CatalogResponse } from "../client/lib/api.js";
-import { installStateFrom, projectLabel, UNKNOWN_INSTALL_STATE } from "../client/lib/usePlaybookInstallState.js";
-
-const catalogue = (skills: string[], commands: string[]): CatalogResponse => ({
- servers: {},
- skills: skills.map((name) => ({ name, description: `${name} description` })),
- commands: commands.map((name) => ({ name })),
- tools: null,
- omitted: { skills: [], commands: [], servers: [] },
- refreshedAt: "2026-08-29T00:00:00Z",
-});
-
-describe("projectLabel", () => {
- it("names the project a claim belongs to", () => {
- expect(projectLabel("/Users/x/Documents/Projects/custom-dca-opencode")).toBe("custom-dca-opencode");
- expect(projectLabel("/tmp/mock-project/")).toBe("mock-project");
- // A worktree directory is named after its branch, which is exactly why the
- // label has to be shown rather than assumed to match the repository.
- expect(projectLabel("/Users/x/.local/share/opencode/worktree/abc/curious-nebula")).toBe("curious-nebula");
- });
-
- it("degrades to the input rather than an empty label", () => {
- expect(projectLabel("standalone")).toBe("standalone");
- expect(projectLabel("")).toBe("");
- });
-});
-
-describe("installStateFrom", () => {
- it("reports what the server says is loaded, labelled by project", () => {
- const state = installStateFrom("/tmp/mock-project", catalogue(["grill-me"], ["verify"]));
- expect(state.status).toBe("ready");
- expect(state.directoryLabel).toBe("mock-project");
- expect(state.installedCommands.has("verify")).toBe(true);
- });
-
- it("states nothing when there is no project", () => {
- // The alternative — an empty "ready" state — would render "Not loaded" for
- // every playbook, which is a claim the app cannot support.
- expect(installStateFrom("", catalogue(["grill-me"], []))).toEqual(UNKNOWN_INSTALL_STATE);
- });
-
- it("states nothing when the catalogue could not be read", () => {
- expect(installStateFrom("/tmp/mock-project", null)).toEqual(UNKNOWN_INSTALL_STATE);
- });
-
- it("reports an empty catalogue as ready with nothing loaded", () => {
- // Distinct from the failure case above: here the server answered, and
- // "nothing is installed in this project" is a supportable claim.
- const state = installStateFrom("/tmp/mock-project", catalogue([], []));
- expect(state.status).toBe("ready");
- expect(state.directoryLabel).toBe("mock-project");
- expect(state.installedCommands.size).toBe(0);
- });
-});
diff --git a/tests/pr-screenshots.test.ts b/tests/pr-screenshots.test.ts
index 69753ba0..dd50dd29 100644
--- a/tests/pr-screenshots.test.ts
+++ b/tests/pr-screenshots.test.ts
@@ -89,10 +89,15 @@ describe("PR screenshot requests", () => {
});
it("accepts Playbooks catalog and detail routes", () => {
- expect(parseScreenshotBlock("```screenshots\n/playbooks\n/playbooks/commands/grill-me\n/playbooks/workflows\n/playbooks/workflows/start-dca-session\n```").requests)
- .toHaveLength(4);
+ expect(parseScreenshotBlock("```screenshots\n/playbooks\n/playbooks/workflows\n/playbooks/workflows/start-dca-session\n```").requests)
+ .toHaveLength(3);
expect(() => parseScreenshotBlock("```screenshots\n/playbooks/skills/grill-me\n```"))
.toThrow(/not a known UI route/u);
+ // The command catalogue is retired, so its routes are no longer known.
+ expect(() => parseScreenshotBlock("```screenshots\n/playbooks/commands\n```"))
+ .toThrow(/not a known UI route/u);
+ expect(() => parseScreenshotBlock("```screenshots\n/playbooks/commands/verify\n```"))
+ .toThrow(/not a known UI route/u);
});
it("accepts the DSH lab and one DSH conversation, but not an arbitrary DSH path", () => {
diff --git a/tests/public-simulator.test.ts b/tests/public-simulator.test.ts
index f2fdaff3..675ff428 100644
--- a/tests/public-simulator.test.ts
+++ b/tests/public-simulator.test.ts
@@ -2,20 +2,26 @@ import { describe, expect, it } from "vitest";
import type { DshTrajectoryPage, PlanningSnapshot } from "../client/lib/api.js";
import { createPublicSimulator } from "../client/simulator/publicSimulator.js";
+import { workflowCatalogue } from "../server/workflows/workflows.js";
describe("public simulator workflow fixtures", () => {
- it("mirrors the six current workflow ids", async () => {
+ // The fixture is the whole catalogue in catalogue order, because the preview
+ // has no BFF to ask: a short fixture would show a Playbooks page that quietly
+ // disagrees with the real one.
+ it("mirrors the current workflow ids, in catalogue order", async () => {
const response = await createPublicSimulator()("https://preview.invalid/api/workflows");
- const payload = await response.json() as { workflows: Array<{ id: string; injector: string }> };
- expect(payload.workflows.map(({ id }) => id)).toEqual([
- "playwright-ui-review",
- "pr-snippet-review",
- "session-update",
- "managed-child",
- "start-dca-session",
- "design-doc-prototype",
- ]);
+ const payload = await response.json() as { workflows: Array<{ id: string; injector: string; argument?: { label: string; required: boolean; maxLength: number }; prompt?: string }> };
+ expect(payload.workflows.map(({ id }) => id)).toEqual(workflowCatalogue().map(({ id }) => id));
expect(payload.workflows.every(({ injector }) => injector.length > 0)).toBe(true);
+ // The fixture injectors are deliberately short summaries, but the field a
+ // workflow collects is part of its shape and must match the real server.
+ for (const real of workflowCatalogue()) {
+ const fixture = payload.workflows.find(({ id }) => id === real.id)!;
+ expect(fixture.title, real.id).toBe(real.title);
+ expect(fixture.description, real.id).toBe(real.description);
+ expect(fixture.argument, real.id).toEqual(real.argument);
+ expect(fixture.prompt, real.id).toEqual(real.prompt);
+ }
});
});
diff --git a/tests/reminders.test.ts b/tests/reminders.test.ts
index ca9a5025..e40204b9 100644
--- a/tests/reminders.test.ts
+++ b/tests/reminders.test.ts
@@ -3,9 +3,9 @@ import path from "node:path";
import { describe, expect, it } from "vitest";
import { splitReminderTags as clientSplitReminderTags } from "../client/lib/reminders.js";
-import { commands } from "../agent-skills/src/lib/commandsSource.js";
-import { commandForReminder, REMINDER_COMMANDS } from "../agent-skills/src/lib/reminderCommands.js";
+import { REMINDER_WORKFLOWS, workflowForReminder } from "../client/lib/reminderWorkflows.js";
import { reminderCatalogue } from "../server/reminders/loader.js";
+import { workflowCatalogue } from "../server/workflows/workflows.js";
import {
REMINDER_BODY_MAX,
REMINDER_DESCRIPTION_MAX,
@@ -133,7 +133,6 @@ describe("isValidReminderId", () => {
describe("shipped catalogue", () => {
const sourceCommit = "8b036a41f578dc6c6307ae0a8dd2857121afcabb";
- const localSourceCommit = "fe9e5ede5f3dc749b0515372ee2e2bc2fc3b3fba";
const importedIds = new Set([
"ascii-diagrams",
"background-subagent",
@@ -146,7 +145,6 @@ describe("shipped catalogue", () => {
"parallel-research-handoff",
"session-handoff",
]);
- const locallyDocumentedIds = new Set(["cite-file-lines", "native-worktree-subagents"]);
const dir = path.join(import.meta.dirname, "..", "reminders");
const ids = readdirSync(dir, { withFileTypes: true }).filter((entry) => entry.isDirectory()).map((entry) => entry.name);
@@ -161,20 +159,25 @@ describe("shipped catalogue", () => {
expect(parsedIds).toHaveLength(ids.length);
});
- it("maps every runtime reminder to one real command", () => {
- const commandNames = new Set(commands.map(({ name }) => name));
- const reminderIds = reminderCatalogue().map(({ id }) => id).sort();
- expect(Object.keys(REMINDER_COMMANDS).sort()).toEqual(reminderIds);
- expect(new Set(Object.values(REMINDER_COMMANDS)).size).toBe(reminderIds.length);
- for (const id of reminderIds) {
- const command = commandForReminder(id);
- expect(command, `${id} has no command mapping`).toBeDefined();
- expect(commandNames.has(command!), `${id} maps to a missing command`).toBe(true);
+ // The retired command catalogue used to give every reminder a documentation
+ // target, so the old invariant could demand total coverage. Workflows do not:
+ // six of those commands were deleted rather than converted because the
+ // reminder already said what they said. What must still hold is that every
+ // link the picker renders resolves — a dead link is the failure this guards.
+ it("links every mapped reminder to a real reminder and a real workflow", () => {
+ const reminderIds = new Set(reminderCatalogue().map(({ id }) => id));
+ const workflowIds = new Set(workflowCatalogue().map(({ id }) => id));
+ expect(Object.keys(REMINDER_WORKFLOWS).length).toBeGreaterThan(0);
+ for (const [reminderId, workflowId] of Object.entries(REMINDER_WORKFLOWS)) {
+ expect(reminderIds.has(reminderId), `${reminderId} is not a shipped reminder`).toBe(true);
+ expect(workflowIds.has(workflowId), `${reminderId} maps to missing workflow ${workflowId}`).toBe(true);
}
+ expect(new Set(Object.values(REMINDER_WORKFLOWS)).size).toBe(Object.keys(REMINDER_WORKFLOWS).length);
});
- it("does not guess a command for an unknown reminder", () => {
- expect(commandForReminder("new-server-reminder")).toBeUndefined();
+ it("does not guess a workflow for an unmapped reminder", () => {
+ expect(workflowForReminder("cite-file-lines")).toBeUndefined();
+ expect(workflowForReminder("new-server-reminder")).toBeUndefined();
});
it("keeps reminder tags application-owned and non-empty", () => {
@@ -198,7 +201,7 @@ describe("shipped catalogue", () => {
it("records complete, pinned provenance for every imported preset", () => {
const catalogue = reminderCatalogue();
- expect(catalogue.filter(({ source }) => source)).toHaveLength(importedIds.size + locallyDocumentedIds.size);
+ expect(catalogue.filter(({ source }) => source)).toHaveLength(importedIds.size);
for (const reminder of catalogue) {
if (importedIds.has(reminder.id)) {
expect(reminder.source).toEqual({
@@ -210,20 +213,12 @@ describe("shipped catalogue", () => {
expect(reminder.body).not.toContain(reminder.source!.repo);
continue;
}
- if (locallyDocumentedIds.has(reminder.id)) {
- expect(reminder.source).toEqual({
- repo: "https://github.com/leoncheng57/custom-dca-opencode",
- path: `agent-skills/commands/${reminder.id}.md`,
- commit: localSourceCommit,
- });
- expect(reminder.body).not.toContain(localSourceCommit);
- expect(reminder.body).not.toContain(reminder.source!.repo);
- continue;
- }
- {
- expect(reminder.source).toBeUndefined();
- continue;
- }
+ // cite-file-lines and native-worktree-subagents used to cite the
+ // repository command they were written beside. Those files are gone, so
+ // the citation is dropped rather than left pointing at a path that no
+ // longer exists on the default branch; they are now ordinary local
+ // presets, which is what reminders/README.md already calls them.
+ expect(reminder.source, `${reminder.id} declares unexpected provenance`).toBeUndefined();
}
});
diff --git a/tests/workflows.test.ts b/tests/workflows.test.ts
index 84d20153..dc49dd7d 100644
--- a/tests/workflows.test.ts
+++ b/tests/workflows.test.ts
@@ -5,6 +5,8 @@ import {
buildPrSnippetReviewPrompt,
captureScopeLabel,
DESIGN_DOC_PROTOTYPE_WORKFLOW_ID,
+ genericWorkflowPrompt,
+ genericWorkflowValid,
groupWorkflows,
MANAGED_CHILD_WORKFLOW_ID,
KNOWN_APP_ROUTES,
@@ -22,6 +24,7 @@ import {
isValidWorkflowId,
splitWorkflowTags as serverSplitWorkflowTags,
withWorkflowTag,
+ WORKFLOW_ARGUMENT_MAX_LENGTH,
workflowCatalogue,
workflowTag,
} from "../server/workflows/workflows.js";
@@ -29,20 +32,46 @@ import {
describe("workflow catalogue", () => {
it("shares semantic picker groups and sends unknown workflows to Other", () => {
const catalogue = [...workflowCatalogue(), { id: "future-workflow", title: "Future", description: "New server workflow", injector: "Do the future work." }];
- expect(groupWorkflows(catalogue).map(({ label }) => label)).toEqual(["Review", "Coordinate", "Document", "Other"]);
+ expect(groupWorkflows(catalogue).map(({ label }) => label)).toEqual(["Review", "Coordinate", "Execute", "Investigate", "Document", "Ship", "Other"]);
expect(groupWorkflows(catalogue).at(-1)?.workflows.map(({ id }) => id)).toEqual(["future-workflow"]);
});
it("contains the shipped workflows, in picker order", () => {
expect(workflowCatalogue().map((workflow) => workflow.id)).toEqual([
PLAYWRIGHT_REVIEW_WORKFLOW_ID,
PR_SNIPPET_REVIEW_WORKFLOW_ID,
+ "red-team",
+ "review-learning",
SESSION_UPDATE_WORKFLOW_ID,
MANAGED_CHILD_WORKFLOW_ID,
START_DCA_SESSION_WORKFLOW_ID,
+ "manager-children",
+ "native-worktree-subagents",
+ "session-handoff",
+ "goal",
+ "dca",
+ "worktree-up",
+ "deep-research",
+ "research-handoff",
DESIGN_DOC_PROTOTYPE_WORKFLOW_ID,
+ "docs-preview",
+ "mini-design-doc",
+ "system-design-artifacts",
+ "verify",
+ "leaving-now-wrap-up",
+ "standup",
]);
});
+ // The "Other" bucket exists for a workflow a newer server ships. A shipped
+ // workflow landing there means this build simply forgot to place it, which
+ // reads identically in the UI and is a different bug entirely.
+ it("places every shipped workflow in a named group, never Other", () => {
+ const grouped = groupWorkflows(workflowCatalogue());
+ expect(grouped.map(({ label }) => label)).not.toContain("Other");
+ expect(grouped.flatMap(({ workflows }) => workflows.map(({ id }) => id)).sort())
+ .toEqual(workflowCatalogue().map(({ id }) => id).sort());
+ });
+
it("binds the PR review injector to this project's repository and one comment", () => {
const injector = workflowCatalogue().find((workflow) => workflow.id === PR_SNIPPET_REVIEW_WORKFLOW_ID)?.injector ?? "";
// The only accepted input is a number; the repository must never come from
@@ -95,6 +124,85 @@ describe("workflow catalogue", () => {
});
});
+describe("generic argument workflows", () => {
+ const generic = workflowCatalogue().filter((workflow) => workflow.argument);
+
+ it("declares a usable, server-bounded field for every argument workflow", () => {
+ expect(generic.length).toBeGreaterThanOrEqual(16);
+ for (const { id, argument } of generic) {
+ expect(argument!.label.trim(), `${id} has a blank label`).not.toBe("");
+ expect(argument!.label.length).toBeLessThanOrEqual(60);
+ expect(argument!.maxLength, `${id} declares a non-positive maxLength`).toBeGreaterThan(0);
+ expect(argument!.maxLength, `${id} exceeds the server bound`).toBeLessThanOrEqual(WORKFLOW_ARGUMENT_MAX_LENGTH);
+ expect(argument!.placeholder?.trim()).not.toBe("");
+ expect(argument!.hint?.trim()).not.toBe("");
+ }
+ });
+
+ // A workflow the dialog renders generically must be able to produce a prompt
+ // from something. One that can produce neither would offer a Send button with
+ // an empty message behind it.
+ it("gives every non-bespoke workflow either a field or a fixed prompt", () => {
+ const bespoke = new Set([
+ PLAYWRIGHT_REVIEW_WORKFLOW_ID,
+ PR_SNIPPET_REVIEW_WORKFLOW_ID,
+ SESSION_UPDATE_WORKFLOW_ID,
+ MANAGED_CHILD_WORKFLOW_ID,
+ START_DCA_SESSION_WORKFLOW_ID,
+ ]);
+ for (const workflow of workflowCatalogue()) {
+ if (bespoke.has(workflow.id)) continue;
+ expect(Boolean(workflow.argument || workflow.prompt?.trim()), `${workflow.id} can produce no prompt`).toBe(true);
+ }
+ });
+
+ // The ported procedures came from command files whose text OpenCode expanded
+ // before the model saw it. A workflow injector is never expanded, so a
+ // surviving `$ARGUMENTS` or `!`-prefixed line would reach the model verbatim
+ // as an instruction nobody wrote.
+ it("carries no unexpanded command substitutions in any injector", () => {
+ for (const { id, injector } of workflowCatalogue()) {
+ expect(injector, `${id} still references $ARGUMENTS`).not.toContain("$ARGUMENTS");
+ expect(injector, `${id} still uses a command file's !\`…\` shell interpolation`).not.toMatch(/^!`/mu);
+ }
+ });
+
+ it("tells the standup workflow to gather its own data and to expect a Plan denial", () => {
+ const preset = workflowCatalogue().find((workflow) => workflow.id === "standup")!;
+ expect(preset.injector).toContain("Nothing is pre-fetched for you");
+ expect(preset.injector).toContain("git log --all --author=");
+ expect(preset.injector).toContain("gh pr list");
+ expect(preset.injector).toMatch(/Plan session bash is\nlikely denied/u);
+ });
+});
+
+describe("genericWorkflowPrompt and genericWorkflowValid", () => {
+ const withArgument = { argument: { label: "Objective", required: true, maxLength: 10 } };
+ const optional = { argument: { label: "Scope", required: false, maxLength: 10 } };
+ const fixed = { prompt: "Do the fixed thing." };
+
+ it("uses the typed text as the prompt, trimmed", () => {
+ expect(genericWorkflowPrompt(withArgument, " ship it ")).toBe("ship it");
+ expect(genericWorkflowPrompt(fixed, "ignored")).toBe("Do the fixed thing.");
+ expect(genericWorkflowPrompt({}, "ignored")).toBe("");
+ });
+
+ it("requires text for a required field and enforces the declared bound", () => {
+ expect(genericWorkflowValid(withArgument, "ship it")).toBe(true);
+ expect(genericWorkflowValid(withArgument, " ")).toBe(false);
+ expect(genericWorkflowValid(withArgument, "x".repeat(11))).toBe(false);
+ });
+
+ // An optional field left blank still has to produce something to send, so
+ // this refuses rather than submitting a message that is only the injector.
+ it("refuses anything that would send an empty prompt", () => {
+ expect(genericWorkflowValid(optional, "")).toBe(false);
+ expect(genericWorkflowValid({}, "")).toBe(false);
+ expect(genericWorkflowValid(fixed, "")).toBe(true);
+ expect(genericWorkflowValid({ ...optional, prompt: "fallback" }, "typed")).toBe(true);
+ });
+});
+
describe("isValidWorkflowId", () => {
it("accepts kebab-case ids and rejects everything else", () => {
expect(isValidWorkflowId("playwright-ui-review")).toBe(true);
diff --git a/vite.config.ts b/vite.config.ts
index 3e348e86..7013b5ba 100644
--- a/vite.config.ts
+++ b/vite.config.ts
@@ -50,7 +50,6 @@ export default defineConfig(() => {
resolve(process.cwd(), "docs"),
resolve(process.cwd(), "deploy/README.md"),
resolve(process.cwd(), "reminders/README.md"),
- resolve(process.cwd(), "agent-skills"),
resolve(process.cwd(), "README.md"),
resolve(process.cwd(), "CONTRIBUTING.md"),
resolve(process.cwd(), "AGENTS.md"),
diff --git a/vitest.config.ts b/vitest.config.ts
index f800fb95..055b491b 100644
--- a/vitest.config.ts
+++ b/vitest.config.ts
@@ -4,7 +4,7 @@ import { defineConfig } from "vitest/config";
// tests live in tests/ at the repo root and exercise server code too.
export default defineConfig({
test: {
- include: ["tests/**/*.test.ts", "agent-skills/src/lib/**/*.test.ts"],
+ include: ["tests/**/*.test.ts"],
environment: "node",
},
});
From 923aa5631ad49b064554c6fb9c01238277ea761b Mon Sep 17 00:00:00 2001
From: Leon Cheng
Date: Sun, 30 Aug 2026 00:30:48 -0400
Subject: [PATCH 2/4] Cut the workflow catalogue to the eight requested
conversions
The user reviewed the converted list and kept eight: hand off to another
session, complete an objective autonomously, operate DCA sessions over the
API, build system design artifacts, preview documentation, write a mini
design doc, leaving now, and write today's standup.
Removes red-team, review-learning, worktree-up, verify, manager-children,
native-worktree-subagents, deep-research and research-handoff from the
catalogue. They are NOT restored as commands: the command model is gone
either way. Four keep an equivalent reminder (deep-research ->
deep-research-subagents, native-worktree-subagents ->
native-worktree-subagents, research-handoff -> parallel-research-handoff,
verify -> human-verification-steps). The other four leave the app entirely
and that is intended.
The Investigate group loses both members and goes with them, leaving five
groups over fourteen workflows.
Trims the reminder-to-workflow join to the two links that still resolve,
rather than leaving four pointing at deleted pages.
---
AGENTS.md | 43 ++
client/components/workflow-dialog.tsx | 13 +-
client/components/workflow-picker.tsx | 111 +++-
client/lib/reminderWorkflows.ts | 18 +-
client/lib/workflows.ts | 30 +-
client/pages/playbooks.module.css | 7 +-
client/simulator/publicSimulator.ts | 21 +-
server/workflows/workflows.ts | 849 ++++++--------------------
tests/e2e/smoke.ui.spec.ts | 20 +
tests/e2e/workflows.ui.spec.ts | 83 ++-
tests/workflows.test.ts | 22 +-
11 files changed, 444 insertions(+), 773 deletions(-)
diff --git a/AGENTS.md b/AGENTS.md
index 182a9c9a..af7be9be 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -491,6 +491,38 @@ several decisions below.
(`agent: plan` for the read-only ones). A workflow carries no declarative mode, and
adding one was deliberately rejected as out of scope — so the guarantee is gone,
and the UI has to say so rather than let a reader assume it survived.
+21d. **A 22-item catalogue is scanned, not read, so the picker is a tile grid.**
+ Full-width rows carrying a two-to-three line description fit ~4.5 of 22 workflows
+ on a desktop screen and ~4 on a phone — four screenfuls to the last item, with at
+ most 1.5 group headings visible at once, so the grouping organised nothing for the
+ reader. The picker now uses the **same title-only tile grid as the reminder
+ picker** (`grid grid-cols-2 sm:grid-cols-3`, `min-h-14` tiles), which puts over
+ half the catalogue and several headings on one screen; an E2E assertion counts
+ tiles above the panel fold on both form factors so the regression is caught rather
+ than re-measured. The description is not lost, it is relocated: it stays on the
+ form's preview stage where it is read before sending, and remains on the tile as
+ `aria-description` and `title` so a screen reader and a hover still get it.
+ Two consequences follow from the tile. **Every shipped workflow needs its own
+ icon**: with a two-line title as the only text, the icon rail is the primary
+ scanning aid, and a shared `Circle` fallback for 16 of 22 would carry no
+ information — `Circle` now means only "a workflow this build has never heard of".
+ And **search covers title and id only**. Matching descriptions was harmless at six
+ workflows and is not at 22: "review" is a substring of "pre*view*" and of
+ "*review*ing", so it surfaced "Send an update to another session" and "Start a DCA
+ session" — tiles whose visible text did not contain what was typed. Matching is
+ still a plain substring (the two pickers must not diverge), so "preview" still
+ matches "review"; the property gained is that every hit now contains the typed text
+ in the title the tile displays, which makes the result set explicable from screen.
+21e. **The injector window is sized to the longest injector, not the shortest.**
+ `max-h-48` (192px) was chosen when the longest shipped injector was 19 lines. The
+ ported procedures run to ~160, so it showed roughly 5% of
+ `system-design-artifacts`: technically scrollable, but not the read-before-send
+ that decision 21 makes the entire trust story. It is now `max-h-[60dvh]` — still
+ bounded, because the dialog must not become one unbroken page, and in `dvh` so a
+ mobile URL bar cannot shrink it below what it promises. The Playbooks card's
+ injector disclosure is bounded at `18rem` for the same reason in reverse: an
+ unbounded preview turned one opened card into most of the page, and the detail
+ modal is where a long injector is meant to be read end to end.
21c. **The repository command catalogue is retired; its procedures are workflows.**
23 commands under `agent-skills/commands/` became 16 workflows and 7 deletions.
The 7 — `background`, `build-waves`, `handoff`, `duck-mode`, `grill-me`,
@@ -947,6 +979,17 @@ several decisions below.
workflow landing there is a placement bug that reads identically in the UI. A
detail route reports absence only after a successful catalogue load, never during
loading or after a failure.
+ The six groups are **Review · Execute · Delegate · Coordinate · Investigate ·
+ Document**, and the server catalogue is authored in that same order so one file
+ tells the truth about both. Two seams were redrawn once 22 items made them
+ load-bearing: "Coordinate" had grown to six and was quietly two ideas — bringing a
+ new session into being versus messaging one that already exists — so **Delegate**
+ took the five that create an agent and Coordinate kept the two that report to
+ something already there (one machine, one human). "Ship" held two whose seam with
+ Execute was soft; verifying and wrapping up are the end of doing work here, not a
+ separate act, so they folded into **Execute** rather than keeping a heading they
+ had not earned. Prefer folding a two-item group into an adjacent one over keeping
+ a heading that only names a coincidence.
Runtime reminders under root `reminders/` stay separate: application-owned
per-message prompt bodies, never sourced from a workflow. The runtime `/skill`
Catalog panel still reports whatever external skills and commands the connected
diff --git a/client/components/workflow-dialog.tsx b/client/components/workflow-dialog.tsx
index 9d7b7c48..bf5bb8f6 100644
--- a/client/components/workflow-dialog.tsx
+++ b/client/components/workflow-dialog.tsx
@@ -678,9 +678,20 @@ export function WorkflowDialog({
Exact prompt
{generatedPrompt}
+ {/*
+ * The injector window is the whole trust story (decision 21): the
+ * contract is that the user can read the exact trusted content
+ * before submitting. `max-h-48` was sized when the longest shipped
+ * injector was 19 lines; the ported procedures run to ~160, so it
+ * showed about 5% of `system-design-artifacts` — technically
+ * scrollable, but not read-before-send in any honest sense. It
+ * stays scrollable and bounded (the dialog must not become one
+ * unbroken page), in `dvh` so a mobile URL bar cannot shrink it
+ * below what it promises.
+ */}
Trusted injector — server-resolved from id "{workflow.id}"
-
{workflow.injector}
+
{workflow.injector}
Appended by the server exactly as shown. The browser only names the workflow id.
{sendsIntoThisSession && <>
diff --git a/client/components/workflow-picker.tsx b/client/components/workflow-picker.tsx
index ee46e642..e99973b9 100644
--- a/client/components/workflow-picker.tsx
+++ b/client/components/workflow-picker.tsx
@@ -1,5 +1,5 @@
import { useEffect, useRef, useState, type KeyboardEvent } from "react";
-import { Check, ChevronDown, Circle, GitPullRequest, Image, MessageSquareText, MonitorCheck, Search, Send, Split, SquarePlus, X, type LucideIcon } from "lucide-react";
+import { ArrowRightLeft, Blocks, BookOpen, Check, ChevronDown, Circle, ClipboardList, FolderGit2, GitBranch, GitPullRequest, GraduationCap, Image, ListChecks, LogOut, Megaphone, MessageSquareText, MonitorCheck, PencilRuler, Search, Split, SquarePlus, Swords, Target, Telescope, Terminal, Users, X, type LucideIcon } from "lucide-react";
import { createPortal } from "react-dom";
import type { WorkflowSummary } from "../lib/api.js";
@@ -7,18 +7,50 @@ import { groupWorkflows } from "../lib/workflows.js";
const LISTBOX_ID = "composer-workflow-listbox";
+/**
+ * Search title and id only.
+ *
+ * The description used to be searched too, which was harmless at six workflows
+ * and is not at 22: a plain substring match over 22 descriptions turns "review"
+ * into a hit on "Start a DCA session" (its description says *reviewing* its
+ * mode) and on "Send an update to another session" (whose description contains
+ * "pre**view**"). Both surface a tile whose visible text does not contain what
+ * was typed, which reads as a bug rather than as a feature. The description is
+ * no longer on the tile either, so matching on it would be matching on text the
+ * reader cannot see.
+ */
function matches(workflow: WorkflowSummary, query: string): boolean {
const needle = query.trim().toLowerCase();
- return !needle || `${workflow.title} ${workflow.description} ${workflow.id}`.toLowerCase().includes(needle);
+ return !needle || `${workflow.title} ${workflow.id}`.toLowerCase().includes(needle);
}
+/**
+ * One distinct symbol per workflow. In a 22-tile grid whose tiles carry only a
+ * two-line title, the icon rail is the primary scanning aid — a shared fallback
+ * would carry zero information for most of the catalogue, so every shipped id
+ * is named here. `Circle` remains only for a workflow a newer server ships that
+ * this build has never heard of.
+ */
const WORKFLOW_ICONS: Record = {
+ // Review
"playwright-ui-review": MonitorCheck,
"pr-snippet-review": GitPullRequest,
- "session-update": MessageSquareText,
+ // Execute
+ goal: Target,
+ dca: Terminal,
+ "leaving-now-wrap-up": LogOut,
+ // Delegate
"managed-child": Split,
"start-dca-session": SquarePlus,
+ "session-handoff": ArrowRightLeft,
+ // Coordinate
+ "session-update": MessageSquareText,
+ standup: Megaphone,
+ // Document
"design-doc-prototype": Image,
+ "docs-preview": BookOpen,
+ "mini-design-doc": PencilRuler,
+ "system-design-artifacts": Blocks,
};
/**
@@ -177,33 +209,52 @@ export function WorkflowPicker({
)}
- {groupedWorkflows.map(({ label, workflows }) =>
+ {/*
+ * A title-only tile grid, matching the reminder picker exactly.
+ * The catalogue is 22 workflows, and one full-width row carrying a
+ * two-to-three line description fit roughly 4.5 of them on a
+ * desktop screen — four screenfuls to reach the last item, and at
+ * most one and a half group headings visible at once, so the
+ * grouping never actually organised anything for the reader. The
+ * description is not lost: it is on the form's preview stage,
+ * where it is read before sending rather than while scanning.
+ */}
+ {groupedWorkflows.map(({ label, workflows }) =>
@@ -217,7 +268,7 @@ export function WorkflowPicker({
function WorkflowIcon({ workflow }: { workflow: WorkflowSummary }) {
const Icon = WORKFLOW_ICONS[workflow.id] ?? Circle;
- return
+ return ;
}
diff --git a/client/lib/reminderWorkflows.ts b/client/lib/reminderWorkflows.ts
index 135ef07d..8edc871e 100644
--- a/client/lib/reminderWorkflows.ts
+++ b/client/lib/reminderWorkflows.ts
@@ -7,21 +7,21 @@
*
* It replaces the reminder-to-command join that lived in
* `agent-skills/src/lib/reminderCommands.ts`. That map covered twelve reminders
- * because every reminder had a same-subject command beside it. Only six survive
- * here, and that is the honest number: the other six commands were deleted
- * rather than converted, precisely because the reminder already said everything
- * they said. Linking those to an unrelated workflow would invent a relationship
- * to keep a symmetry that no longer exists.
+ * because every reminder had a same-subject command beside it. Only two survive
+ * here, and that is the honest number: the other ten commands were deleted
+ * rather than converted, because the reminder already said everything they said
+ * or the capability was retired outright. Linking those to an unrelated workflow
+ * would invent a relationship to keep a symmetry that no longer exists.
*
* A reminder with no entry simply renders no details link, which is already how
* the picker treats an unmapped reminder.
+ *
+ * This join is a stopgap. Once every reminder has its own Playbooks detail page,
+ * the picker should link a reminder to ITSELF rather than to a workflow that
+ * merely shares its subject, and this module should be deleted.
*/
export const REMINDER_WORKFLOWS: Readonly> = Object.freeze({
- "deep-research-subagents": "deep-research",
"docs-and-diagram-tooling": "docs-preview",
- "human-verification-steps": "verify",
- "native-worktree-subagents": "native-worktree-subagents",
- "parallel-research-handoff": "research-handoff",
"session-handoff": "session-handoff",
});
diff --git a/client/lib/workflows.ts b/client/lib/workflows.ts
index 1d973c2c..d74c8805 100644
--- a/client/lib/workflows.ts
+++ b/client/lib/workflows.ts
@@ -36,31 +36,37 @@ export const START_DCA_SESSION_WORKFLOW_ID = "start-dca-session";
export const PR_SNIPPET_REVIEW_WORKFLOW_ID = "pr-snippet-review";
export const DESIGN_DOC_PROTOTYPE_WORKFLOW_ID = "design-doc-prototype";
-// Six semantic groups covering the whole catalogue. The sixteen ported
-// procedures are named as literals rather than as exported constants: unlike
-// the five above they are not referenced anywhere else, because they need no
-// bespoke branch — the generic argument field renders all of them.
+// Six semantic groups covering the whole catalogue, in catalogue order. The
+// sixteen ported procedures are named as literals rather than as exported
+// constants: unlike the five above they are not referenced anywhere else,
+// because they need no bespoke branch — the generic argument field renders all
+// of them.
+//
+// Two seams were redrawn after the 22-item catalogue made them load-bearing.
+// "Coordinate" had grown to six and was quietly two ideas — creating a new
+// agent and messaging one that already exists — so **Delegate** now owns
+// everything that brings a new session into being, and Coordinate keeps only
+// the two that report to something already there (one machine, one human).
+// "Ship" held two whose seam with Execute was soft: verifying and wrapping up
+// are the end of doing the work here, not a separate act, so they fold into
+// **Execute** rather than keeping a heading they did not earn.
//
// Nothing should land in the "Other" bucket `groupWorkflows` appends. That
// bucket exists for a workflow a newer server ships that this build has never
// heard of, not as a home for one this build forgot to place.
export const WORKFLOW_GROUPS = [
- { label: "Review", ids: [PLAYWRIGHT_REVIEW_WORKFLOW_ID, PR_SNIPPET_REVIEW_WORKFLOW_ID, "red-team", "review-learning"] },
+ { label: "Review", ids: [PLAYWRIGHT_REVIEW_WORKFLOW_ID, PR_SNIPPET_REVIEW_WORKFLOW_ID] },
+ { label: "Execute", ids: ["goal", "dca", "leaving-now-wrap-up"] },
{
- label: "Coordinate",
+ label: "Delegate",
ids: [
- SESSION_UPDATE_WORKFLOW_ID,
MANAGED_CHILD_WORKFLOW_ID,
START_DCA_SESSION_WORKFLOW_ID,
- "manager-children",
- "native-worktree-subagents",
"session-handoff",
],
},
- { label: "Execute", ids: ["goal", "dca", "worktree-up"] },
- { label: "Investigate", ids: ["deep-research", "research-handoff"] },
+ { label: "Coordinate", ids: [SESSION_UPDATE_WORKFLOW_ID, "standup"] },
{ label: "Document", ids: [DESIGN_DOC_PROTOTYPE_WORKFLOW_ID, "docs-preview", "mini-design-doc", "system-design-artifacts"] },
- { label: "Ship", ids: ["verify", "leaving-now-wrap-up", "standup"] },
] as const;
export interface WorkflowGroup {
diff --git a/client/pages/playbooks.module.css b/client/pages/playbooks.module.css
index 8be200b2..82434265 100644
--- a/client/pages/playbooks.module.css
+++ b/client/pages/playbooks.module.css
@@ -78,7 +78,12 @@
.empty { margin-top: 1.5rem; border: 2px solid var(--playbooks-line); background: var(--playbooks-surface); padding: 1rem; color: var(--playbooks-muted); }
.workflowState { margin-top: 1.5rem; }
.workflowGroup { margin-top: 1.6rem; }.workflowGroup > h3 { margin: 0 0 .65rem; color: var(--playbooks-muted); font: 800 .7rem ui-monospace, monospace; letter-spacing: .12em; text-transform: uppercase; }
-.injectorPreview { margin-top: .8rem; color: var(--playbooks-muted); font: .62rem ui-monospace, monospace; }.injectorPreview summary { cursor: pointer; }.injectorPreview pre, .injectorDetail pre { max-width: 100%; overflow: auto; margin: .65rem 0 0; border: 1px solid var(--playbooks-line); background: var(--playbooks-terminal); color: var(--playbooks-terminal-ink); padding: .75rem; white-space: pre-wrap; overflow-wrap: anywhere; }
+.injectorPreview { margin-top: .8rem; color: var(--playbooks-muted); font: .62rem ui-monospace, monospace; }.injectorPreview summary { cursor: pointer; }
+/* A ported procedure runs to ~160 lines, so an unbounded card preview turns one
+ opened disclosure into most of the page. The detail modal is where a long
+ injector is meant to be read end to end. */
+.injectorPreview pre { max-height: 18rem; }
+.injectorPreview pre, .injectorDetail pre { max-width: 100%; overflow: auto; margin: .65rem 0 0; border: 1px solid var(--playbooks-line); background: var(--playbooks-terminal); color: var(--playbooks-terminal-ink); padding: .75rem; white-space: pre-wrap; overflow-wrap: anywhere; }
.locations { margin-top: 4rem; }.locations p { max-width: 43rem; color: var(--playbooks-muted); line-height: 1.5; }.tableWrap { overflow-x: auto; border: 2px solid var(--playbooks-line); background: var(--playbooks-surface); box-shadow: 4px 4px 0 var(--playbooks-shadow); }.table { min-width: 40rem; width: 100%; border-collapse: collapse; font-size: .75rem; }.table th, .table td { border-bottom: 1px solid var(--playbooks-line); padding: .75rem; text-align: left; vertical-align: top; }.table th { font-family: ui-monospace, monospace; font-weight: 700; }.table td span { display: block; margin-top: .25rem; color: var(--playbooks-muted); }
diff --git a/client/simulator/publicSimulator.ts b/client/simulator/publicSimulator.ts
index 6666352e..96a2c2a2 100644
--- a/client/simulator/publicSimulator.ts
+++ b/client/simulator/publicSimulator.ts
@@ -476,30 +476,21 @@ export function createPublicSimulator(): typeof fetch {
if (path === "/api/permission-requests") return response({ requests: permissions });
const permissionRoute = routeMatch(path, /^\/api\/permission-requests\/([^/]+)\/reply$/u);
if (permissionRoute && method === "POST") { permissions = permissions.filter((item) => item.id !== decodeURIComponent(permissionRoute[1])); return response({ replied: true }); }
- if (path === "/api/reminders") return response({ reminders: [{ id: "verify", title: "Verify changes", description: "Run focused verification before reporting completion.", triggers: ["verify", "test"], tags: ["verification"] }, { id: "review", title: "Review implementation", description: "Review behavior, risks, and missing tests.", triggers: ["review"], tags: ["critique"] }] });
if (path === "/api/workflows") return response({ workflows: [
{ id: "playwright-ui-review", title: "Review a UI change with Playwright", description: "Drive a focused Playwright pass over one route or component and bring back targeted evidence, without a full deployment or a complete screenshot regeneration.", injector: "Drive the named route with Playwright and exercise exactly the requested state or interaction. Report what was verified, what failed, and where any evidence was written." },
{ id: "pr-snippet-review", title: "Post a snippet-by-snippet PR review", description: "Walk one pull request as an ordered sequence of explained snippets and post it as a single GitHub comment. Takes only the pull request number; the repository comes from this project directory.", injector: "Produce one ordered snippet-by-snippet GitHub review comment. Resolve the repository from this session's project directory and pin every link to the pull request head SHA." },
- { id: "red-team", title: "Red-team the work just produced", description: "Argue against the plan or diff that was just produced, grounding every objection in evidence and closing with exactly one verdict.", injector: "Argue against the work above. Ground every objection in file:line or command output and close with exactly one verdict.", argument: { label: "Target", placeholder: "your most recent output, docs/rfc.md, or the plan above", hint: "Name the file or plan to argue against. Say \"your most recent output\" to red-team the work already in this transcript.", required: true, maxLength: 4_000 } },
- { id: "review-learning", title: "Review code with learning excerpts", description: "Review a diff or pull request findings-first as a senior engineer, then teach from two to five exact, cited source excerpts.", injector: "Return findings first, ordered by severity, then teach from two to five exact cited excerpts.", argument: { label: "Review target", placeholder: "the current diff, pull request #253, or server/routes/sessions.ts", hint: "Name the diff, pull request, file, or subsystem to review. Nothing is changed, posted, or persisted.", required: true, maxLength: 4_000 } },
- { id: "session-update", title: "Send an update to another session", description: "Deliver a hand-off message to another session in this project after an explicit preview of the target and the exact prompt.", injector: "Treat this as new input from the user's other workstream. Delivery is asynchronous: accepted does not mean completed." },
- { id: "managed-child", title: "Launch a Managed Child", description: "Start an independent child session with its own transcript and a Plan or Build policy fixed at creation time. No native task card is created and no automatic hand-back occurs.", injector: "Complete the objective in this independent managed-child transcript and report outcomes, risks, and next steps." },
- { id: "start-dca-session", title: "Start a DCA session", description: "Start an independent root session in this project or an isolated worktree, after reviewing its Plan/Build mode, model, assignment, and trusted instructions.", injector: "Work only on the assignment under the selected mode. This is an independent root session with no parent or automatic hand-back." },
- { id: "manager-children", title: "Run a wave of manager children", description: "Dispatch isolated OpenCode workers into sibling git worktrees and manage their status files, branches, and pull request wave.", injector: "Dispatch each disjoint assignment to its own worktree, branch, and status file, then integrate one reviewed branch at a time.", argument: { label: "Wave to manage", placeholder: "the four independent notification fixes in issues #288, #290, #291, #294", hint: "Describe the disjoint assignments. Each child gets its own worktree, branch, and status file.", required: true, maxLength: 20_000 } },
- { id: "native-worktree-subagents", title: "Delegate to native worktree subagents", description: "Delegate disjoint edits to native Task children confined to a sibling worktree, with fail-closed containment guards before every mutation.", injector: "Confine every child to its assigned worktree with absolute paths and a pwd guard before any mutation.", argument: { label: "Work to delegate", placeholder: "the three independent client fixes on this branch", hint: "Describe the disjoint edits. Children stay scoped to this session's OpenCode directory regardless of their assigned worktree.", required: true, maxLength: 20_000 } },
- { id: "session-handoff", title: "Hand off to a standalone session", description: "Carry the current task into one explicitly configured standalone OpenCode session, with a self-contained packet and explicit CLI flags.", injector: "Nothing is inherited automatically. Carry settings through explicit CLI flags and a self-contained handoff packet.", argument: { label: "Task to hand off", placeholder: "finish the epic hierarchy work on feat/planning-epic-hierarchy", hint: "Nothing is inherited automatically, so describe the task the way the receiving session will need to read it.", required: true, maxLength: 20_000 } },
{ id: "goal", title: "Complete an objective autonomously", description: "Work an objective through research, implementation, fixes, and final verification as one sustained run with durable checkpoints.", injector: "Work the objective through to completion as one sustained run, with a durable checkpoint updated at every boundary.", argument: { label: "Objective", placeholder: "make the notification badge match the server's global unresolved count", hint: "The run continues without asking whether to proceed, so state the acceptance criteria you actually want.", required: true, maxLength: 100_000 } },
{ id: "dca", title: "Operate DCA sessions over the BFF API", description: "Create, prompt, and observe DCA sessions through the agent-facing BFF HTTP API, distinct from the human-only composer workflows.", injector: "Use the BFF HTTP API to create, prompt, and observe sessions. Accepted is not completed; poll for the finished turn.", argument: { label: "Session work", placeholder: "launch a read-only child that audits export error handling and report its findings", hint: "Describe the session lifecycle work. The procedure below calls already-authorized BFF endpoints directly.", required: true, maxLength: 20_000 } },
- { id: "worktree-up", title: "Create an isolated worktree", description: "Cut a sibling git worktree from the remote default branch, install its own dependencies, and prove a green baseline before any edits.", injector: "Fetch first, cut a sibling worktree from the remote default branch, install its dependencies, and prove a green baseline.", argument: { label: "Topic", placeholder: "planning-epic-hierarchy (issue #241)", hint: "Becomes the kebab-case worktree directory and branch name. Include the issue number when one exists.", required: true, maxLength: 200 } },
- { id: "deep-research", title: "Fan out deep research", description: "Split one broad research question across concurrent read-only agents, then reconcile their cited evidence into a single answer.", injector: "Split the question into non-overlapping axes, launch read-only agents concurrently, then synthesize rather than concatenate.", argument: { label: "Research question", placeholder: "how does notification suppression interact with grouping, badges, and Web Push?", hint: "Worth delegating only with three or more independent unknowns. A needle lookup should stay inline.", required: true, maxLength: 20_000 } },
- { id: "research-handoff", title: "Research, then compile handoff prompts", description: "Research several independent tasks read-only, compile decision-closed handoff prompts, and stop for review before launching anything.", injector: "Research read-only, compile decision-closed prompts, and stop for review before launching anything.", argument: { label: "Tasks to research", placeholder: "one task per line: the epic hierarchy feed, the badge count, the push rotation fix", hint: "List the independent tasks. Nothing is launched until you answer the two questions at the end.", required: true, maxLength: 100_000 } },
+ { id: "leaving-now-wrap-up", title: "Wrap up and leave an accurate status", description: "Stop this run safely, push authorized progress, refresh every status artifact, and end with one accurate merge verdict.", injector: "Stop this run's work, preserve authorized progress, refresh every status artifact, and end with one accurate merge verdict.", argument: { label: "Wrap-up note", placeholder: "branch feat/planning-epic-hierarchy, PR #312 — post the update there", hint: "May name the branch, pull request, issue, or human update destination. No destination is invented for you.", required: true, maxLength: 4_000 } },
+ { id: "managed-child", title: "Launch a Managed Child", description: "Start an independent child session with its own transcript and a Plan or Build policy fixed at creation time. No native task card is created and no automatic hand-back occurs.", injector: "Complete the objective in this independent managed-child transcript and report outcomes, risks, and next steps." },
+ { id: "start-dca-session", title: "Start a DCA session", description: "Start an independent root session in this project or an isolated worktree, after reviewing its Plan/Build mode, model, assignment, and trusted instructions.", injector: "Work only on the assignment under the selected mode. This is an independent root session with no parent or automatic hand-back." },
+ { id: "session-handoff", title: "Hand off to another session", description: "Carry the current task into one explicitly configured standalone OpenCode session, with a self-contained packet and explicit CLI flags.", injector: "Nothing is inherited automatically. Carry settings through explicit CLI flags and a self-contained handoff packet.", argument: { label: "Task to hand off", placeholder: "finish the epic hierarchy work on feat/planning-epic-hierarchy", hint: "Nothing is inherited automatically, so describe the task the way the receiving session will need to read it.", required: true, maxLength: 20_000 } },
+ { id: "session-update", title: "Send an update to another session", description: "Deliver a hand-off message to another session in this project after an explicit preview of the target and the exact prompt.", injector: "Treat this as new input from the user's other workstream. Delivery is asynchronous: accepted does not mean completed." },
+ { id: "standup", title: "Write today's standup", description: "Gather the last day's commits and pull requests, then turn them into a three-section standup update written to be read aloud.", injector: "Gather the commit and pull request data yourself, then write a three-section standup update from what the commands actually returned.", argument: { label: "Scope", placeholder: "everything, or just the notification work", hint: "Narrows the update to one project or topic. The commit and pull request data is gathered by the agent, not pre-fetched.", required: true, maxLength: 2_000 } },
{ id: "design-doc-prototype", title: "Capture a Durable Design Prototype", description: "Mock up an unbuilt UI change as fast static HTML, screenshot it, and publish it into a dated engineering-design document — no fields to fill in, just confirm and send.", injector: "Build a self-contained static HTML mockup, capture desktop and mobile screenshots, and publish the durable design writeup for review.", prompt: "Capture a durable design prototype for this proposal and publish it for review." },
{ id: "docs-preview", title: "Choose, render, and preview documentation", description: "Pick the documentation medium from where the reader opens it, render it with a real renderer, and preview the result before claiming completion.", injector: "Choose the medium from where the reader opens it, render it with a real renderer, and preview the result before claiming completion.", argument: { label: "What to document", placeholder: "the notification suppression pipeline, as a diagram for the README", hint: "Name the subject and, if you already know it, where the reader will open it.", required: true, maxLength: 20_000 } },
{ id: "mini-design-doc", title: "Write a mini design doc", description: "Produce a transcript-first technical design narrative that takes under five minutes to read and ends with one clear recommendation.", injector: "Write a transcript-first design narrative under five minutes long, ending with one clear recommendation.", argument: { label: "Subject", placeholder: "whether session status should be joined into the notification centre", hint: "One medium-sized decision. The result stays in this transcript and creates no files.", required: true, maxLength: 20_000 } },
{ id: "system-design-artifacts", title: "Build a system-design review package", description: "Assemble an evidence-led senior system-design review package, with every load-bearing claim tagged by its evidence class.", injector: "Audit evidence first, tag every load-bearing claim with an evidence class, and select artifacts that each answer a distinct review question.", argument: { label: "Subject", placeholder: "the notification delivery and suppression system, current-state", hint: "Name the system or proposal, and say current-state, target-state, or mixed if you already know which you want.", required: true, maxLength: 20_000 } },
- { id: "verify", title: "Run the checks, then write verification steps", description: "Discover and run this repository's real verification contract, then write numbered human verification steps with expected results and failure signals.", injector: "Discover this repository's real verification contract, run it, then write numbered human verification steps with expected results and failure signals.", argument: { label: "Surface to verify", placeholder: "the notification popover on mobile", hint: "Scopes the human checklist. The automated checks are discovered from the repository either way.", required: true, maxLength: 4_000 } },
- { id: "leaving-now-wrap-up", title: "Wrap up and leave an accurate status", description: "Stop this run safely, push authorized progress, refresh every status artifact, and end with one accurate merge verdict.", injector: "Stop this run's work, preserve authorized progress, refresh every status artifact, and end with one accurate merge verdict.", argument: { label: "Wrap-up note", placeholder: "branch feat/planning-epic-hierarchy, PR #312 — post the update there", hint: "May name the branch, pull request, issue, or human update destination. No destination is invented for you.", required: true, maxLength: 4_000 } },
- { id: "standup", title: "Write a standup update", description: "Gather the last day's commits and pull requests, then turn them into a three-section standup update written to be read aloud.", injector: "Gather the commit and pull request data yourself, then write a three-section standup update from what the commands actually returned.", argument: { label: "Scope", placeholder: "everything, or just the notification work", hint: "Narrows the update to one project or topic. The commit and pull request data is gathered by the agent, not pre-fetched.", required: true, maxLength: 2_000 } },
] });
// -----------------------------------------------------------------------
diff --git a/server/workflows/workflows.ts b/server/workflows/workflows.ts
index 9f9cb9fe..ac899158 100644
--- a/server/workflows/workflows.ts
+++ b/server/workflows/workflows.ts
@@ -53,6 +53,7 @@ export interface WorkflowPreset {
// Catalogue order follows the picker's semantic groups: Review, Coordinate,
// Execute, Investigate, Document, Ship.
const RAW_CATALOGUE: WorkflowPreset[] = [
+ // ── Review — judge work that already exists ───────────────────────────────────
{
id: "playwright-ui-review",
title: "Review a UI change with Playwright",
@@ -93,364 +94,7 @@ const RAW_CATALOGUE: WorkflowPreset[] = [
"Post exactly one comment. Report the resulting comment URL when you are done.",
].join("\n"),
},
- {
- id: "red-team",
- title: "Red-team the work just produced",
- description:
- "Argue against the plan or diff that was just produced, grounding every objection in evidence and closing with exactly one verdict.",
- argument: {
- label: "Target",
- placeholder: "your most recent output, docs/rfc.md, or the plan above",
- hint: "Name the file or plan to argue against. Say \"your most recent output\" to red-team the work already in this transcript.",
- required: true,
- maxLength: 4_000,
- },
- injector: [
- "Red-team what you just produced. If the note above names a file or a plan, target",
- "that; otherwise target your own most recent output.",
- "",
- "Open with the side-switch, verbatim, as the first line:",
- "",
- "> Red-teaming the work above. I am arguing against it.",
- "",
- "Then:",
- "",
- "1. Work all six objection classes — wrong problem, cheaper alternative, hidden",
- " coupling, operational cost, reversibility, the unchecked assumption. Say so",
- " explicitly when a class yields nothing.",
- "2. Ground every objection in `file:line`, pasted command output, or a doc URL.",
- " Grep for the other callers rather than reasoning about them.",
- "3. Keep objections you cannot ground in a separate, clearly labelled",
- " speculative bucket. Never mix them with grounded ones.",
- "4. Rank the grounded objections by likelihood x cost-if-true x cheapness-to-check",
- " and present them as a table.",
- "5. Close with the single cheapest experiment that would kill the work, and a",
- " verdict of exactly one of `proceed`, `proceed-with-change`, or `stop`.",
- "",
- "Do not re-litigate the work's merits. The case for it has already been made.",
- "Prefer a fresh subtask context for anything larger than a small diff so the",
- "reviewer does not inherit the author's sunk cost.",
- "",
- "Score likelihood, cost if true, and inverted cost-to-check from 1-5; sort by",
- "their product. Cheap checks on plausible expensive failures should rise first.",
- "",
- "| Failure | Response |",
- "|---|---|",
- "| Review hedges or praises the work | Re-state the side switch and delete the defense |",
- "| Concerns are plausible but ungrounded | Run the grep/curl/read or move them to speculative |",
- "| Highest concern cannot be acted on | Include cost-to-check and identify one experiment |",
- "| Same blind spot survives | Move the artifact into a fresh subtask context |",
- "| Review ends with concerns but no decision | Emit one of the three exact verdicts |",
- ].join("\n"),
- },
- {
- id: "review-learning",
- title: "Review code with learning excerpts",
- description:
- "Review a diff or pull request findings-first as a senior engineer, then teach from two to five exact, cited source excerpts.",
- argument: {
- label: "Review target",
- placeholder: "the current diff, pull request #253, or server/routes/sessions.ts",
- hint: "Name the diff, pull request, file, or subsystem to review. Nothing is changed, posted, or persisted.",
- required: true,
- maxLength: 4_000,
- },
- injector: [
- "Review the target above as a senior engineer and turn the result into a concise",
- "learning walkthrough. If no target is supplied, review the current diff. Honor",
- "any requested subsystem or boundary first, such as authentication or an",
- "external-runtime integration.",
- "",
- "This procedure is self-contained. Do not load or defer to a skill. Do not change",
- "code, post comments, open issues, or persist the walkthrough unless the user",
- "explicitly asks after the review.",
- "",
- "## Review before teaching",
- "",
- "Inspect the complete diff and enough surrounding code, callers, tests,",
- "configuration, and contracts to understand behavior across layers. For a PR,",
- "review the whole change at its pinned head revision rather than only the latest",
- "commit. Reproduce or run focused checks when safe and feasible.",
- "",
- "Return findings first, ordered by severity. Each finding must contain:",
- "",
- "- severity and a precise title;",
- "- repository-relative `path:line-line` at the defect or risky behavior;",
- "- concrete failure mode and affected user/system;",
- "- evidence establishing the claim;",
- "- the smallest useful remediation direction, without implementing it.",
- "",
- "Separate `Verified findings` from `Unverified risks`. A verified finding is",
- "supported by implementation, a reproducer, test output, or an authoritative",
- "contract. An unverified risk names exactly what evidence is missing and the",
- "cheapest check. Do not inflate educational observations into findings.",
- "",
- "If there are no findings, say that explicitly before teaching and name residual",
- "risks and test gaps. \"No findings\" never means \"proved correct\".",
- "",
- "## Then teach from a small evidence set",
- "",
- "After findings, select two to five high-value excerpts. Choose code that explains",
- "an invariant, boundary, state transition, failure strategy, or non-obvious",
- "tradeoff. Do not quote routine plumbing, the whole diff, or several snippets",
- "that teach the same lesson.",
- "",
- "For every excerpt:",
- "",
- "1. give an exact repository-relative `path:line-line` verified against the",
- " reviewed revision;",
- "2. quote the smallest contiguous source range that is independently readable;",
- "3. state whether it is `Finding evidence` or `Educational`;",
- "4. explain the engineering lesson and why the implementation shape matters;",
- "5. connect it to the preceding and following layer;",
- "6. state what the excerpt does not prove.",
- "",
- "Use this output shape:",
- "",
- "```text",
- "Verified findings",
- "1. [severity] Title — path:line-line",
- " Failure, evidence, remediation direction.",
- "",
- "Unverified risks",
- "- Risk — missing evidence; cheapest verification.",
- "",
- "Layer map",
- "HTTP route -> service/pool -> sidecar protocol -> external SDK",
- "",
- "Learning excerpts",
- "1. path:line-line — Finding evidence | Educational",
- " ",
- " Lesson: ...",
- " Connection: previous layer -> this code -> next layer.",
- " Does not prove: ...",
- "",
- "Residual risks and test gaps",
- "- ...",
- "```",
- "",
- "Adapt the layer map to the repository. Distinguish control flow from authority:",
- "a route calling a service does not prove the service may trust browser input,",
- "and a mock sidecar does not prove the external SDK behaves the same way. Keep",
- "quoted lines exact, explanations concise, and educational value subordinate to",
- "the correctness review.",
- ].join("\n"),
- },
- {
- id: "session-update",
- title: "Send an update to another session",
- description:
- "Deliver a hand-off message to another session in this project after an explicit preview of the target and the exact prompt.",
- injector: [
- 'You are receiving the "Send an update to another session" workflow.',
- "This message was composed in another session in the same project and delivered here after an explicit preview and confirmation. Treat it as new input from the user's other workstream: read it, reconcile it with your current task, and continue accordingly.",
- "Delivery is asynchronous (POST /session/{id}/prompt_async answers 204 for accepted, not completed), so do not assume the sender is watching this transcript live.",
- ].join("\n"),
- },
- {
- id: "managed-child",
- title: "Launch a Managed Child",
- description:
- "Start an independent child session with its own transcript and a Plan or Build policy fixed at creation time. No native task card is created and no automatic hand-back occurs.",
- injector: [
- 'You are a managed child session started by the "Launch a Managed Child" workflow.',
- "- You run in your own independent transcript with the Plan or Build policy fixed at creation time.",
- "- No native task card exists in the parent and no automatic hand-back will occur: the human reads your results here, in this transcript.",
- "- Complete the objective below and end with a clear summary of outcomes, remaining risks, and suggested next steps.",
- ].join("\n"),
- },
- {
- id: "start-dca-session",
- title: "Start a DCA session",
- description:
- "Start an independent root session in this project or an isolated worktree, after reviewing its Plan/Build mode, model, assignment, and trusted instructions.",
- injector: [
- 'You are an independent root session started by the "Start a DCA session" workflow.',
- "- You have no parent session, task card, Managed Child relationship, automatic hand-back, or provenance link to the session that started you.",
- "- Work only on the assignment below under the Plan or Build mode selected at launch.",
- "- End with a clear summary of outcomes, verification performed, remaining risks, and suggested next steps.",
- ].join("\n"),
- },
- {
- id: "manager-children",
- title: "Run a wave of manager children",
- description:
- "Dispatch isolated OpenCode workers into sibling git worktrees and manage their status files, branches, and pull request wave.",
- argument: {
- label: "Wave to manage",
- placeholder: "the four independent notification fixes in issues #288, #290, #291, #294",
- hint: "Describe the disjoint assignments. Each child gets its own worktree, branch, and status file.",
- required: true,
- maxLength: 20_000,
- },
- injector: [
- "Manage the work described above through separate OpenCode workers in sibling git",
- "worktrees and unfocused CMUX workspaces.",
- "",
- "Before dispatch:",
- "",
- "1. Write a durable wave plan with ownership, dependencies, integration order,",
- " and final verification.",
- "2. State explicitly that standalone children continue after this turn but do",
- " not automatically resume the manager.",
- "3. Create one worktree and branch per disjoint assignment from the current",
- " remote default branch.",
- "4. Require status files, heartbeats, verified commits, pushed branches, and PRs.",
- "",
- "Standalone children cannot resume this manager by writing status, pushing a PR,",
- "changing a badge, or running `cmux notify`. Those are durable evidence or human",
- "alerts, not an OpenCode callback. The manager resumes only on a user message, an",
- "in-process task result, or a separately tested supervisor prompt. Never promise",
- "unattended progression without that wake channel.",
- "",
- "Cut waves on complete artifacts and disjoint files. Parallel workers may read",
- "shared files but must not edit the same integration file, lockfile, migration,",
- "generated output, port, or database. Cap a wave around five children. Create",
- "each sibling worktree after fetching the remote default branch, verify its path,",
- "branch, clean status, dependencies, baseline, and fixed-port ownership.",
- "",
- "Every cold-start assignment must name the absolute worktree and branch,",
- "objective, owned and forbidden files, sibling workers, settled contracts,",
- "non-goals, permission posture, model, exact tests, definition of done, and the",
- "rule that children never push the default branch. Require a gitignored",
- "`.agent-status.json` with phases `assigned`, `working`, `verifying`, `pushed`,",
- "`pr-open`, `blocked`, and `done`; UTC timestamps come from `date -u`, update on",
- "every transition, and heartbeat at least every ten minutes.",
- "",
- "Before launching, ask the user which model to use when neither the repository",
- "nor the current session has an explicit model contract that settles the choice.",
- "Do not guess from availability, cost, or a previous unrelated child. Never add",
- "`--auto` or any equivalent broad permission-approval mode unless the user",
- "explicitly authorizes it for these workers. A copied launch command is not",
- "authorization. Record both decisions in each assignment and launch command.",
- "",
- "Monitor in this order: status file, Git/remote/PR/CI evidence, then the child",
- "screen only when evidence is stale or contradictory. A stale heartbeat or a",
- "`done` badge without a pushed branch is not delivery proof.",
- "",
- "On every resumed turn, read the durable plan and status files, fetch remotes,",
- "inspect exact-head checks, reconcile ownership, then continue the queued action.",
- "If nothing is ready, report that and stop rather than busy-waiting. Integrate one",
- "reviewed branch at a time, run the full suite after each merge, and clean a",
- "worktree only after merge and after confirming no follow-up needs it.",
- "",
- "| Failure | Response |",
- "|---|---|",
- "| Manager stops after dispatch | Wait for a real inbound turn; restore durable state |",
- "| Notification produces no manager action | Correct the claim: it notified a human only |",
- "| Two children edit one seam | Stop one writer and sequence ownership |",
- "| Tracker says done but no PR exists | Inspect Git and require push/PR evidence |",
- "| Child works in the wrong checkout | Stop and relaunch with absolute containment |",
- "| Model was not specified or contractually settled | Ask before launch |",
- "| Launch template contains `--auto` without explicit authorization | Remove it and ask; do not broaden permissions by convenience |",
- "| Automated wake duplicates turns | Disable it until idle checks, dedupe, and serialization are proven |",
- "| Manager loses the next action | Restore the synchronized plan and task queue |",
- ].join("\n"),
- },
- {
- id: "native-worktree-subagents",
- title: "Delegate to native worktree subagents",
- description:
- "Delegate disjoint edits to native Task children confined to a sibling worktree, with fail-closed containment guards before every mutation.",
- argument: {
- label: "Work to delegate",
- placeholder: "the three independent client fixes on this branch",
- hint: "Describe the disjoint edits. Children stay scoped to this session's OpenCode directory regardless of their assigned worktree.",
- required: true,
- maxLength: 20_000,
- },
- injector: [
- "Delegate the work described above only after confirming a fresh Build-only parent and a",
- "dedicated sibling worktree from fresh `origin/main`. A parent that previously",
- "activated Plan may pass historical denies to children even after its own Build",
- "tools return. If the child cannot pass preflight, stop; never weaken policy or",
- "substitute an unrelated root session.",
- "",
- "The child remains scoped to the parent's OpenCode directory. External-directory",
- "permission does not change relative path resolution, shell CWD, LSP/VCS scope,",
- "or event directory. Its cold-start prompt must require:",
- "",
- "1. The absolute worktree path and branch, with edits allowed only there.",
- "2. Every Bash call setting `workdir` there or using `git -C `.",
- "3. Every read, edit, and patch using an absolute path inside the worktree.",
- "4. Exclusive file ownership, non-goals, exact verification, commit/push rules,",
- " and the final report.",
- "5. This guard before edits, tests, commit, and push:",
- "",
- " pwd",
- " git rev-parse --show-toplevel",
- " git status --short --branch",
- "",
- "The child must stop without mutation unless both `pwd` and Git top-level equal",
- "the assignment. Never fall back to the parent checkout, force-push, or push the",
- "default branch. Parallel children must not share a lockfile, migration, port,",
- "database, generated artifact, or integration file. Review the diff and checks",
- "at hand-back before presenting or merging it, and never duplicate its work.",
- "",
- "| Failure | Response |",
- "|---|---|",
- "| Child resolves paths in the parent checkout | Stop and relaunch with absolute containment rules |",
- "| Tool remains denied in Build | Use a fresh Build-only parent; do not weaken rules |",
- "| Separate branches still conflict | Sequence shared ownership or give it to one owner |",
- "| Hand-back is unclear | Fix the prompt's deliverable and verification contract before launch |",
- ].join("\n"),
- },
- {
- id: "session-handoff",
- title: "Hand off to a standalone session",
- description:
- "Carry the current task into one explicitly configured standalone OpenCode session, with a self-contained packet and explicit CLI flags.",
- argument: {
- label: "Task to hand off",
- placeholder: "finish the epic hierarchy work on feat/planning-epic-hierarchy",
- hint: "Nothing is inherited automatically, so describe the task the way the receiving session will need to read it.",
- required: true,
- maxLength: 20_000,
- },
- injector: [
- "Prepare one standalone OpenCode session for the task described above.",
- "",
- "Nothing is inherited automatically. Carry settings through explicit CLI flags",
- "and a self-contained handoff packet.",
- "",
- "1. Choose the mechanism: fresh interactive TUI for steerable work,",
- " `opencode run` for scripted work, `--session --fork` only when history",
- " must be copied, or the task tool when this is actually a subagent job.",
- "2. Inspect `opencode agent list`, `opencode models`, the repository root, branch,",
- " worktree state, and baseline. Mark anything you cannot verify as UNVERIFIED.",
- "3. Write a prompt file outside the worktree containing the absolute path,",
- " branch, objective, progress, settled decisions with rationale, owned and",
- " forbidden files, requested agent/model/variant, permission posture,",
- " verification commands, stop condition, and unverified assumptions.",
- "4. Show the packet and exact launch command before executing it. Use",
- " `--agent plan` or `--agent build`; do not express mode as prompt prose.",
- " Use `--variant` for provider-specific reasoning effort. `--thinking` controls",
- " display only. Never add `--auto` unless the user explicitly requested it.",
- "5. Keep secrets out of the packet and process arguments. Verify the target",
- " branch before allowing edits.",
- "6. After launch, verify the working directory, branch, selected agent and model,",
- " and that the first reply understood the packet. A successful process start",
- " proves what was requested, not what the provider accepted.",
- "",
- "Use cmux only as an optional presentation wrapper and never steal focus. Store",
- "the packet outside every worktree and keep secrets out of it: prompt text can",
- "surface in shell history or process arguments. Require the child's first reply",
- "to restate its path, branch, objective, ownership, agent, model, variant,",
- "permission posture, and stop condition before work begins.",
- "",
- "| Failure | Response |",
- "|---|---|",
- "| Child edits instead of planning | Relaunch with `--agent plan`; prose is not mode control |",
- "| History was copied unexpectedly | Use a fresh TUI or run without continuation/fork flags |",
- "| Child opens the wrong repository | Pass an absolute project path or `--dir` |",
- "| Parent and child overwrite one another | Assign disjoint ownership and stop one writer |",
- "| Secret appears in process arguments | Stop, remove it, and rotate the exposed credential |",
- "| Provider ignores the reasoning variant | Mark acceptance UNVERIFIED until reported |",
- "",
- "After launch, report what started and end the parent turn. Do not begin the",
- "child's assigned work.",
- ].join("\n"),
- },
+ // ── Execute — do and finish the work in this session ──────────────────────────
{
id: "goal",
title: "Complete an objective autonomously",
@@ -656,169 +300,211 @@ const RAW_CATALOGUE: WorkflowPreset[] = [
].join("\n"),
},
{
- id: "worktree-up",
- title: "Create an isolated worktree",
+ id: "leaving-now-wrap-up",
+ title: "Wrap up and leave an accurate status",
description:
- "Cut a sibling git worktree from the remote default branch, install its own dependencies, and prove a green baseline before any edits.",
+ "Stop this run safely, push authorized progress, refresh every status artifact, and end with one accurate merge verdict.",
argument: {
- label: "Topic",
- placeholder: "planning-epic-hierarchy (issue #241)",
- hint: "Becomes the kebab-case worktree directory and branch name. Include the issue number when one exists.",
+ label: "Wrap-up note",
+ placeholder: "branch feat/planning-epic-hierarchy, PR #312 — post the update there",
+ hint: "May name the branch, pull request, issue, or human update destination. No destination is invented for you.",
required: true,
- maxLength: 200,
+ maxLength: 4_000,
},
injector: [
- "Create a git worktree for the topic named above now.",
- "",
- "1. Find the repository root and origin's default branch. Run `git fetch origin`",
- " before branching; start from `origin/`, never a stale local branch.",
- "2. Create the worktree beside the repository at",
- " `.worktrees/`, on branch `/`. Use kebab-case and",
- " include the issue number when one exists.",
- "3. Install dependencies in the new worktree because gitignored directories such",
- " as `node_modules` and `.venv` are not shared.",
- "4. Copy required gitignored local configuration such as `.env`; make any paths",
- " inside it absolute when they refer back to state in the original checkout.",
- "5. Establish a green baseline with the repository's typecheck, tests, and build",
- " before writing code.",
- "6. Before starting anything on fixed ports, run",
- " `lsof -nP -iTCP: -sTCP:LISTEN` and confirm the owning PID. Only one",
- " worktree may run a fixed-port stack at a time.",
- "7. Report the absolute worktree path, branch, base revision, dependency status,",
- " baseline result, and any port conflict.",
- "",
- "Do not work around Git's worktree safety checks. If `origin/HEAD` is unset, run",
- "`git remote set-head origin --auto` and re-read it. If the branch exists, omit",
- "`-b`; if it is checked out elsewhere, find and use that worktree instead.",
- "",
- "Worktrees must be siblings, never nested under the clone. On cleanup, inspect",
- "for uncommitted work before `git worktree remove`; use `--force` only when that",
- "work is explicitly disposable. Run `git worktree prune` for registrations whose",
- "directories were deleted outside Git.",
+ "Wrap up the current run now. The note above may name the branch, PR, issue, or",
+ "human update destination. Complete these steps in order; do not stop after a",
+ "chat-only summary.",
"",
- "| Failure | Response |",
- "|---|---|",
- "| New branch is already behind | Fetch and recreate it from the remote default branch |",
- "| Branch is already checked out | Use `git worktree list`; do not bypass the refusal |",
- "| New worktree commands fail immediately | Install its own dependencies |",
- "| App has no local config | Copy required ignored config and fix relative paths |",
- "| Server behaves like another branch | Verify the listening PID and its worktree |",
- "| Two stacks share writable state | Stop one stack; use stack-free verification tiers |",
- "| Deleted path remains registered | Prune stale worktrees, then confirm the list |",
+ "1. Stop work owned by this run. Cancel this run's background agents, test",
+ " runners, preview servers, and other processes that could keep mutating state.",
+ " Verify they stopped. Never kill a process merely because its name or port",
+ " looks familiar: establish that this run started or owns it. Leave unrelated",
+ " user and sibling-worktree processes alone.",
+ "2. Inspect the repository root, branch, worktree, status, diff, staged diff, and",
+ " recent commits. Separate this run's intended changes from pre-existing or",
+ " unrelated user changes. Scan intended paths for credentials and generated",
+ " secret files. Never stage or commit secrets, and never absorb unrelated",
+ " changes just to make the tree clean.",
+ "3. Run the most relevant affordable verification for this run's changes. Record",
+ " each check as green, red, or not run, including the command and concise reason.",
+ " A red check does not justify hiding otherwise useful progress.",
+ "4. Preserve real progress under the repository's permissions and the user's",
+ " explicit commit/push instructions. If authorized, stage only owned files,",
+ " commit them, and push the current feature branch even when verification is",
+ " red, with the failure stated plainly. Never push `main` or another protected",
+ " default branch. Do not force-push. If commit or push is not authorized, is",
+ " rejected, or would include unsafe/unrelated content, leave it uncommitted or",
+ " local and report that fact as a blocker instead of bypassing the gate.",
+ "5. Refresh every project-defined status artifact with the current UTC timestamp,",
+ " exact branch/commit/push state, verification results, blockers, and next",
+ " action. Supersede stale claims. Reconcile the task list immediately: completed",
+ " work is completed, intentionally dropped work is cancelled, and only genuine",
+ " remaining work stays pending. If a status artifact is tracked, include its",
+ " final update in the authorized push, using a small follow-up commit when the",
+ " progress commit already exists. Do not let status sync create new local-only",
+ " state.",
+ "6. Post one human-readable update to each destination the project explicitly",
+ " defines and this run is authorized to use, such as its PR or issue. Do not",
+ " invent a destination or leak details to a public channel. Include what is",
+ " green and red, why anything is red, what is committed and pushed versus only",
+ " local, open decisions requiring a human, and the next command or owner.",
+ "7. If there is a PR, end the posted update and final response with exactly one",
+ " accurate verdict: `SAFE TO MERGE` only when the intended change is pushed,",
+ " required checks are green, and no blocking decision remains; otherwise `DO",
+ " NOT MERGE`, followed by the blocking facts. Re-read remote/PR state after the",
+ " push before choosing the verdict.",
+ "",
+ "Finish with the branch, pushed commit (or `not pushed`), stopped-work evidence,",
+ "verification matrix, status destinations updated, open decisions, and merge",
+ "verdict. Accuracy is more important than making the wrap-up look green.",
+ ].join("\n"),
+ },
+ // ── Delegate — create a new agent or session and hand it work ─────────────────
+ {
+ id: "managed-child",
+ title: "Launch a Managed Child",
+ description:
+ "Start an independent child session with its own transcript and a Plan or Build policy fixed at creation time. No native task card is created and no automatic hand-back occurs.",
+ injector: [
+ 'You are a managed child session started by the "Launch a Managed Child" workflow.',
+ "- You run in your own independent transcript with the Plan or Build policy fixed at creation time.",
+ "- No native task card exists in the parent and no automatic hand-back will occur: the human reads your results here, in this transcript.",
+ "- Complete the objective below and end with a clear summary of outcomes, remaining risks, and suggested next steps.",
+ ].join("\n"),
+ },
+ {
+ id: "start-dca-session",
+ title: "Start a DCA session",
+ description:
+ "Start an independent root session in this project or an isolated worktree, after reviewing its Plan/Build mode, model, assignment, and trusted instructions.",
+ injector: [
+ 'You are an independent root session started by the "Start a DCA session" workflow.',
+ "- You have no parent session, task card, Managed Child relationship, automatic hand-back, or provenance link to the session that started you.",
+ "- Work only on the assignment below under the Plan or Build mode selected at launch.",
+ "- End with a clear summary of outcomes, verification performed, remaining risks, and suggested next steps.",
].join("\n"),
},
{
- id: "deep-research",
- title: "Fan out deep research",
+ id: "session-handoff",
+ title: "Hand off to another session",
description:
- "Split one broad research question across concurrent read-only agents, then reconcile their cited evidence into a single answer.",
+ "Carry the current task into one explicitly configured standalone OpenCode session, with a self-contained packet and explicit CLI flags.",
argument: {
- label: "Research question",
- placeholder: "how does notification suppression interact with grouping, badges, and Web Push?",
- hint: "Worth delegating only with three or more independent unknowns. A needle lookup should stay inline.",
+ label: "Task to hand off",
+ placeholder: "finish the epic hierarchy work on feat/planning-epic-hierarchy",
+ hint: "Nothing is inherited automatically, so describe the task the way the receiving session will need to read it.",
required: true,
maxLength: 20_000,
},
injector: [
- "Research the question above properly.",
- "",
- "First decide whether delegation pays. Use this procedure only when the question",
- "contains at least three independent unknowns, spans unrelated files or sources,",
- "or would require roughly twenty tool calls. For a needle lookup, do it inline.",
- "",
- "If it qualifies:",
- "",
- "1. Split it into 3–5 non-overlapping axes. Name what each agent owns and what it",
- " must not read so they do not converge on the first grep hit.",
- "2. Launch all agents concurrently in one message. Use `explore` with",
- " \"very thorough\" for codebase research; use `general` read-only only when bash",
- " or another unavailable tool is genuinely required.",
- "3. Every prompt starts and ends with READ-ONLY, supplies the absolute repository",
- " path, asks numbered questions, requires `file:line` or verbatim URL evidence,",
- " asks what does **not** exist, and ends with `UNVERIFIED:`.",
- "4. Spot-check one load-bearing claim from each report yourself.",
- "5. Synthesize rather than concatenate: answer the original question first,",
- " reconcile disagreements by reading the cited evidence, merge all unverified",
- " items, preserve citations, and say what surprised you.",
- "",
- "Do not dispatch sequential questions whose later shape depends on an earlier",
- "answer, and do not let research agents mutate state.",
- "",
- "Use a flat fan-out: the usual subagent depth is one. Cap the batch at five;",
- "above that overlap and synthesis cost usually erase the gain. A `general` agent",
- "must be told READ-ONLY at both the start and end of its prompt; prefer the",
- "enforced read-only `explore` agent whenever its tools are sufficient.",
- "",
- "Specify a bounded deliverable, such as one answer-first section per numbered",
- "question under 800 words. If a live API is in scope, allow GET only and request",
- "the verbatim schema rather than a paraphrase.",
+ "Prepare one standalone OpenCode session for the task described above.",
+ "",
+ "Nothing is inherited automatically. Carry settings through explicit CLI flags",
+ "and a self-contained handoff packet.",
+ "",
+ "1. Choose the mechanism: fresh interactive TUI for steerable work,",
+ " `opencode run` for scripted work, `--session --fork` only when history",
+ " must be copied, or the task tool when this is actually a subagent job.",
+ "2. Inspect `opencode agent list`, `opencode models`, the repository root, branch,",
+ " worktree state, and baseline. Mark anything you cannot verify as UNVERIFIED.",
+ "3. Write a prompt file outside the worktree containing the absolute path,",
+ " branch, objective, progress, settled decisions with rationale, owned and",
+ " forbidden files, requested agent/model/variant, permission posture,",
+ " verification commands, stop condition, and unverified assumptions.",
+ "4. Show the packet and exact launch command before executing it. Use",
+ " `--agent plan` or `--agent build`; do not express mode as prompt prose.",
+ " Use `--variant` for provider-specific reasoning effort. `--thinking` controls",
+ " display only. Never add `--auto` unless the user explicitly requested it.",
+ "5. Keep secrets out of the packet and process arguments. Verify the target",
+ " branch before allowing edits.",
+ "6. After launch, verify the working directory, branch, selected agent and model,",
+ " and that the first reply understood the packet. A successful process start",
+ " proves what was requested, not what the provider accepted.",
+ "",
+ "Use cmux only as an optional presentation wrapper and never steal focus. Store",
+ "the packet outside every worktree and keep secrets out of it: prompt text can",
+ "surface in shell history or process arguments. Require the child's first reply",
+ "to restate its path, branch, objective, ownership, agent, model, variant,",
+ "permission posture, and stop condition before work begins.",
"",
"| Failure | Response |",
"|---|---|",
- "| Reports repeat each other | Split by artifact or directory and name exclusions |",
- "| Reports are essays without evidence | Ask numbered questions and require citations |",
- "| A report contains invented certainty | Merge `UNVERIFIED` lists and spot-check its load-bearing claim |",
- "| A child mutates state | Stop it; use enforced read-only delegation |",
- "| Calls ran sequentially | Relaunch independent axes concurrently or keep the work inline |",
- "| Synthesis is longer than the reports | Answer first, reconcile conflicts, preserve only decisive evidence |",
+ "| Child edits instead of planning | Relaunch with `--agent plan`; prose is not mode control |",
+ "| History was copied unexpectedly | Use a fresh TUI or run without continuation/fork flags |",
+ "| Child opens the wrong repository | Pass an absolute project path or `--dir` |",
+ "| Parent and child overwrite one another | Assign disjoint ownership and stop one writer |",
+ "| Secret appears in process arguments | Stop, remove it, and rotate the exposed credential |",
+ "| Provider ignores the reasoning variant | Mark acceptance UNVERIFIED until reported |",
+ "",
+ "After launch, report what started and end the parent turn. Do not begin the",
+ "child's assigned work.",
].join("\n"),
},
+ // ── Coordinate — tell someone who already exists what is happening ────────────
{
- id: "research-handoff",
- title: "Research, then compile handoff prompts",
+ id: "session-update",
+ title: "Send an update to another session",
description:
- "Research several independent tasks read-only, compile decision-closed handoff prompts, and stop for review before launching anything.",
+ "Deliver a hand-off message to another session in this project after an explicit preview of the target and the exact prompt.",
+ injector: [
+ 'You are receiving the "Send an update to another session" workflow.',
+ "This message was composed in another session in the same project and delivered here after an explicit preview and confirmation. Treat it as new input from the user's other workstream: read it, reconcile it with your current task, and continue accordingly.",
+ "Delivery is asynchronous (POST /session/{id}/prompt_async answers 204 for accepted, not completed), so do not assume the sender is watching this transcript live.",
+ ].join("\n"),
+ },
+ {
+ id: "standup",
+ title: "Write today's standup",
+ description:
+ "Gather the last day's commits and pull requests, then turn them into a three-section standup update written to be read aloud.",
argument: {
- label: "Tasks to research",
- placeholder: "one task per line: the epic hierarchy feed, the badge count, the push rotation fix",
- hint: "List the independent tasks. Nothing is launched until you answer the two questions at the end.",
+ label: "Scope",
+ placeholder: "everything, or just the notification work",
+ hint: "Narrows the update to one project or topic. The commit and pull request data is gathered by the agent, not pre-fetched.",
required: true,
- maxLength: 100_000,
+ maxLength: 2_000,
},
injector: [
- "Research and prepare handoffs for the tasks listed above.",
- "",
- "Three phases, in order:",
- "",
- "1. **Read-only research.** Split the supplied tasks into independent axes and",
- " launch one read-only agent per task concurrently. Require structure, prior",
- " art, the nearest analogue, concrete integration points, live API truth when",
- " reachable by GET, testing conventions, explicit gaps, `file:line` evidence,",
- " and a final `UNVERIFIED:` list.",
- "2. **Compile prompts.** Turn each report into a decision-closed prompt containing",
- " the absolute worktree/branch state, docs to read first,",
- " `PRE-RESEARCHED - DO NOT RE-DERIVE`, settled decisions with rationale,",
- " `GOTCHA:` lines, numbered build steps, reasoned exclusions, constraints,",
- " exact verification, `SHARED-RESOURCE RULE`, and the report-back contract.",
- " Store prompt files outside every git worktree.",
- "3. **Stop for review.** Show every prompt before launching. Ask whether the",
- " receiving sessions should plan first or edit immediately, and whether they",
- " should open PRs or leave local commits. Do not fire anything until those two",
- " choices are answered.",
- "",
- "Use plain ASCII in prompt files, and do not recreate a baseline from a stale",
- "local default branch. Research axes must be independent, read-only, and large",
- "enough to justify delegation; a needle lookup stays inline. State READ-ONLY at",
- "both ends of each prompt. Restrict live API probes to GET and preserve verbatim",
- "response shapes.",
- "",
- "For later launch, fetch first, create sibling worktrees from the remote default",
- "branch, install dependencies in each, and prove a baseline before allowing",
- "edits. Enumerate fixed ports, writable state, databases, generated output, and",
- "lockfiles in every receiving prompt. Only one worker may own a shared resource.",
- "Never steal focus when creating a session.",
+ "Turn the last day's work into a standup update.",
"",
- "| Failure | Response |",
- "|---|---|",
- "| Research agent starts implementing | Stop it and strengthen the read-only boundary |",
- "| Receiving agent re-greps everything | Add `file:line` evidence and explicit negative findings |",
- "| Scope is relitigated | Preserve the decision's rationale and evidence |",
- "| Both workers bind one port or state directory | Keep stack-free checks parallel; serialize the shared stack |",
- "| First test is red | Establish whether baseline or worker caused it before proceeding |",
- "| Prompt is mangled | Store plain-ASCII text in a file; do not inline multiline shell arguments |",
- "| Live feature silently no-ops | Probe its actual gate or API before writing the handoff |",
+ "Nothing is pre-fetched for you. Gather the data yourself by running these three",
+ "commands, then use only what they actually returned:",
+ "",
+ " git log --all --author=\"$(git config user.email)\" --since=\"24 hours ago\" --pretty=format:'%h %s' --no-merges",
+ " gh pr list --author \"@me\" --state merged --limit 10 --json number,title,mergedAt -q '.[] | \"#\\(.number) \\(.title)\"'",
+ " gh pr list --author \"@me\" --state open --limit 10 --json number,title,isDraft -q '.[] | \"#\\(.number)\\(if .isDraft then \" DRAFT\" else \"\" end) \\(.title)\"'",
+ "",
+ "This workflow is sent in this session's current mode. In a Plan session bash is",
+ "likely denied, so these commands may not run at all. If they do not, say so",
+ "plainly and stop rather than reconstructing a standup from memory or from this",
+ "transcript and presenting it as today's commits. If `gh` is missing or",
+ "unauthenticated, report the pull request sections as unavailable instead of",
+ "inventing them.",
+ "",
+ "Write the standup update from that output, in three sections:",
+ "",
+ "**Yesterday** — what actually landed, grouped by theme rather than listed per",
+ "commit. Say what it does for a reader, not what the diff touched.",
+ "",
+ "**Today** — what the open work implies is next. Mark anything that is a guess.",
+ "",
+ "**Blocked** — only genuine blockers. If there are none, say \"nothing blocked\"",
+ "rather than inventing one.",
+ "",
+ "Rules:",
+ "",
+ "- Three to six bullets per section. This gets read aloud.",
+ "- No commit hashes and no branch names unless someone would need to go find it.",
+ "- If the log is empty, say so plainly. Do not pad it out of the PR list.",
+ "- The scope note above may narrow this to one project or topic; if it does,",
+ " drop everything else.",
+ "",
+ "This is deliberately self-contained and adds no retrieval context until a human",
+ "invokes it.",
].join("\n"),
},
+ // ── Document — produce something a reader keeps ───────────────────────────────
{
id: "design-doc-prototype",
title: "Capture a Durable Design Prototype",
@@ -1140,185 +826,6 @@ const RAW_CATALOGUE: WorkflowPreset[] = [
"and does not claim that opening the PR deploys or operates the system.",
].join("\n"),
},
- {
- id: "verify",
- title: "Run the checks, then write verification steps",
- description:
- "Discover and run this repository's real verification contract, then write numbered human verification steps with expected results and failure signals.",
- argument: {
- label: "Surface to verify",
- placeholder: "the notification popover on mobile",
- hint: "Scopes the human checklist. The automated checks are discovered from the repository either way.",
- required: true,
- maxLength: 4_000,
- },
- injector: [
- "Discover this repository's verification contract before running anything:",
- "",
- "1. Read its agent instructions and contribution guide.",
- "2. Inspect changed files, package or task manifests, lockfiles, CI workflows,",
- " and adjacent tests to identify the language, package manager, and required",
- " checks. Do not infer npm merely because this is an OpenCode project.",
- "3. Prefer documented repository commands. When none exist, choose the narrowest",
- " standard checks supported by the detected tooling and state why.",
- "4. Run the relevant focused checks, then the repository's required aggregate",
- " typecheck, test, lint, and build checks when practical. Record every exact",
- " command, exit status, and useful result.",
- "5. Inspect `git status --short`, the diff against the actual base branch, and the",
- " diff stat before writing the checklist.",
- "",
- "Never claim a check ran based on old CI, prompt text, or a remembered convention.",
- "If any required check is red, stop and report the failure. Do not send a human",
- "to verify a build that is already broken.",
- "",
- "If everything is green, write the human verification checklist for the change",
- "shown in the diff, scoped to the surface named above when one is named:",
- "",
- "- 5 to 12 numbered steps, each with the action, the expected result, and the",
- " failure signal.",
- "- Name the exact URL, command, viewport, theme, or test data each step needs.",
- "- Check what automated tests cannot: visual layout, interaction, keyboard",
- " access, responsive behaviour, deployed behaviour.",
- "- Cover a boundary, not only the happy path — empty state, error state, narrow",
- " viewport, or reduced motion, whichever the diff actually touches.",
- "- Separate VERIFIED, FAILED, and UNVERIFIED. Never report an unreachable",
- " surface as passing.",
- "- End with a disposition: Ready to ship, Fixes required, Partially verified, or",
- " Blocked on human access.",
- "",
- "Research the changed routes, help text, API contracts, tests, start commands,",
- "ports, fixtures, roles, flags, and deployment target yourself. Ask the user only",
- "for access or product intent that the repository cannot establish. Do not use",
- "implementation-detail checks; verify what a user can see or accomplish.",
- "",
- "A screenshot proves one visual instant, not focus movement, persistence, time,",
- "keyboard operation, or error handling. Exercise those behaviors directly.",
- "",
- "After execution, list `VERIFIED`, `FAILED`, and `UNVERIFIED` separately, keeping",
- "empty categories visible. End with exactly one disposition: **Ready to ship**,",
- "**Fixes required**, **Partially verified**, or **Blocked on human access**.",
- "",
- "| Failure | Response |",
- "|---|---|",
- "| Repository has no documented check commands | Inspect its manifests and CI, run only checks supported by detected tooling, and explain the choice |",
- "| Automation is red | Stop with Fixes required |",
- "| Infrastructure prevents a check | Mark it UNVERIFIED, never passed |",
- "| A step lacks URL, data, viewport, role, or expected output | Add the missing setup before handing it to a human |",
- "| Checklist exceeds 12 steps | Remove automated or low-information duplication |",
- ].join("\n"),
- },
- {
- id: "leaving-now-wrap-up",
- title: "Wrap up and leave an accurate status",
- description:
- "Stop this run safely, push authorized progress, refresh every status artifact, and end with one accurate merge verdict.",
- argument: {
- label: "Wrap-up note",
- placeholder: "branch feat/planning-epic-hierarchy, PR #312 — post the update there",
- hint: "May name the branch, pull request, issue, or human update destination. No destination is invented for you.",
- required: true,
- maxLength: 4_000,
- },
- injector: [
- "Wrap up the current run now. The note above may name the branch, PR, issue, or",
- "human update destination. Complete these steps in order; do not stop after a",
- "chat-only summary.",
- "",
- "1. Stop work owned by this run. Cancel this run's background agents, test",
- " runners, preview servers, and other processes that could keep mutating state.",
- " Verify they stopped. Never kill a process merely because its name or port",
- " looks familiar: establish that this run started or owns it. Leave unrelated",
- " user and sibling-worktree processes alone.",
- "2. Inspect the repository root, branch, worktree, status, diff, staged diff, and",
- " recent commits. Separate this run's intended changes from pre-existing or",
- " unrelated user changes. Scan intended paths for credentials and generated",
- " secret files. Never stage or commit secrets, and never absorb unrelated",
- " changes just to make the tree clean.",
- "3. Run the most relevant affordable verification for this run's changes. Record",
- " each check as green, red, or not run, including the command and concise reason.",
- " A red check does not justify hiding otherwise useful progress.",
- "4. Preserve real progress under the repository's permissions and the user's",
- " explicit commit/push instructions. If authorized, stage only owned files,",
- " commit them, and push the current feature branch even when verification is",
- " red, with the failure stated plainly. Never push `main` or another protected",
- " default branch. Do not force-push. If commit or push is not authorized, is",
- " rejected, or would include unsafe/unrelated content, leave it uncommitted or",
- " local and report that fact as a blocker instead of bypassing the gate.",
- "5. Refresh every project-defined status artifact with the current UTC timestamp,",
- " exact branch/commit/push state, verification results, blockers, and next",
- " action. Supersede stale claims. Reconcile the task list immediately: completed",
- " work is completed, intentionally dropped work is cancelled, and only genuine",
- " remaining work stays pending. If a status artifact is tracked, include its",
- " final update in the authorized push, using a small follow-up commit when the",
- " progress commit already exists. Do not let status sync create new local-only",
- " state.",
- "6. Post one human-readable update to each destination the project explicitly",
- " defines and this run is authorized to use, such as its PR or issue. Do not",
- " invent a destination or leak details to a public channel. Include what is",
- " green and red, why anything is red, what is committed and pushed versus only",
- " local, open decisions requiring a human, and the next command or owner.",
- "7. If there is a PR, end the posted update and final response with exactly one",
- " accurate verdict: `SAFE TO MERGE` only when the intended change is pushed,",
- " required checks are green, and no blocking decision remains; otherwise `DO",
- " NOT MERGE`, followed by the blocking facts. Re-read remote/PR state after the",
- " push before choosing the verdict.",
- "",
- "Finish with the branch, pushed commit (or `not pushed`), stopped-work evidence,",
- "verification matrix, status destinations updated, open decisions, and merge",
- "verdict. Accuracy is more important than making the wrap-up look green.",
- ].join("\n"),
- },
- {
- id: "standup",
- title: "Write a standup update",
- description:
- "Gather the last day's commits and pull requests, then turn them into a three-section standup update written to be read aloud.",
- argument: {
- label: "Scope",
- placeholder: "everything, or just the notification work",
- hint: "Narrows the update to one project or topic. The commit and pull request data is gathered by the agent, not pre-fetched.",
- required: true,
- maxLength: 2_000,
- },
- injector: [
- "Turn the last day's work into a standup update.",
- "",
- "Nothing is pre-fetched for you. Gather the data yourself by running these three",
- "commands, then use only what they actually returned:",
- "",
- " git log --all --author=\"$(git config user.email)\" --since=\"24 hours ago\" --pretty=format:'%h %s' --no-merges",
- " gh pr list --author \"@me\" --state merged --limit 10 --json number,title,mergedAt -q '.[] | \"#\\(.number) \\(.title)\"'",
- " gh pr list --author \"@me\" --state open --limit 10 --json number,title,isDraft -q '.[] | \"#\\(.number)\\(if .isDraft then \" DRAFT\" else \"\" end) \\(.title)\"'",
- "",
- "This workflow is sent in this session's current mode. In a Plan session bash is",
- "likely denied, so these commands may not run at all. If they do not, say so",
- "plainly and stop rather than reconstructing a standup from memory or from this",
- "transcript and presenting it as today's commits. If `gh` is missing or",
- "unauthenticated, report the pull request sections as unavailable instead of",
- "inventing them.",
- "",
- "Write the standup update from that output, in three sections:",
- "",
- "**Yesterday** — what actually landed, grouped by theme rather than listed per",
- "commit. Say what it does for a reader, not what the diff touched.",
- "",
- "**Today** — what the open work implies is next. Mark anything that is a guess.",
- "",
- "**Blocked** — only genuine blockers. If there are none, say \"nothing blocked\"",
- "rather than inventing one.",
- "",
- "Rules:",
- "",
- "- Three to six bullets per section. This gets read aloud.",
- "- No commit hashes and no branch names unless someone would need to go find it.",
- "- If the log is empty, say so plainly. Do not pad it out of the PR list.",
- "- The scope note above may narrow this to one project or topic; if it does,",
- " drop everything else.",
- "",
- "This is deliberately self-contained and adds no retrieval context until a human",
- "invokes it.",
- ].join("\n"),
- },
];
/**
diff --git a/tests/e2e/smoke.ui.spec.ts b/tests/e2e/smoke.ui.spec.ts
index f5b10a7a..07d27283 100644
--- a/tests/e2e/smoke.ui.spec.ts
+++ b/tests/e2e/smoke.ui.spec.ts
@@ -1130,6 +1130,26 @@ test.describe("composer", () => {
await expect(page.getByTestId("composer-workflow-option")).toHaveCount(1);
await expect(page.getByTestId("composer-workflow-option")).toContainText("Post a snippet-by-snippet PR review");
+ // Search covers title and id only. Matching descriptions was harmless at
+ // six workflows and is not at 22: "review" used to surface "Send an update
+ // to another session" and "Start a DCA session", whose descriptions say
+ // "pre*view*" and "*review*ing" — tiles whose visible text does not contain
+ // what was typed, which reads as a bug.
+ //
+ // Matching is still a plain substring, so "preview" genuinely does contain
+ // "review" and "Choose, render, and preview documentation" is still a hit.
+ // That is the property being locked in, not a leftover: every result now
+ // contains the typed text in the title the tile actually shows, so the
+ // result set is explicable from the screen. Neither of the two false hits
+ // above is here.
+ await page.getByTestId("composer-workflow-search").fill("review");
+ const reviewHits = page.getByTestId("composer-workflow-option");
+ await expect(reviewHits).toHaveCount(5);
+ for (const text of await reviewHits.allInnerTexts()) expect(text.toLowerCase()).toContain("review");
+ for (const id of ["session-update", "start-dca-session"]) {
+ await expect(page.locator(`[data-testid="composer-workflow-option"][data-workflow-id="${id}"]`)).toHaveCount(0);
+ }
+
// A zero-match query must not divide by zero in the index math.
await page.getByTestId("composer-workflow-search").fill("zzzz-no-such-workflow");
await expect(page.getByTestId("composer-workflow-empty")).toBeVisible();
diff --git a/tests/e2e/workflows.ui.spec.ts b/tests/e2e/workflows.ui.spec.ts
index 6d0cff01..8cdf5fa4 100644
--- a/tests/e2e/workflows.ui.spec.ts
+++ b/tests/e2e/workflows.ui.spec.ts
@@ -41,26 +41,18 @@ test.describe("workflow catalogue API", () => {
expect(payload.workflows.map((workflow) => workflow.id)).toEqual([
"playwright-ui-review",
"pr-snippet-review",
- "red-team",
- "review-learning",
- "session-update",
+ "goal",
+ "dca",
+ "leaving-now-wrap-up",
"managed-child",
"start-dca-session",
- "manager-children",
- "native-worktree-subagents",
"session-handoff",
- "goal",
- "dca",
- "worktree-up",
- "deep-research",
- "research-handoff",
+ "session-update",
+ "standup",
"design-doc-prototype",
"docs-preview",
"mini-design-doc",
"system-design-artifacts",
- "verify",
- "leaving-now-wrap-up",
- "standup",
]);
// The four core keys are required of every workflow; `argument` and
// `prompt` are optional additions. This is deliberately not a bare superset
@@ -160,18 +152,37 @@ test.describe("workflow picker UI", () => {
await expect(options).toHaveCount(22);
const groups = page.getByTestId("composer-workflow-group");
await expect(groups).toHaveCount(6);
- for (const [index, label] of ["Review", "Coordinate", "Execute", "Investigate", "Document", "Ship"].entries()) {
+ for (const [index, label] of ["Review", "Execute", "Delegate", "Coordinate", "Document"].entries()) {
await expect(groups.nth(index)).toHaveAccessibleName(label);
}
await expect(page.getByTestId("composer-workflow-icon")).toHaveCount(22);
await expect(options.nth(0)).toContainText("Review a UI change with Playwright");
await expect(options.nth(1)).toContainText("Post a snippet-by-snippet PR review");
await expect(options.nth(2)).toContainText("Red-team the work just produced");
- await expect(options.nth(4)).toContainText("Send an update to another session");
- await expect(options.nth(5)).toContainText("Launch a Managed Child");
- await expect(options.nth(6)).toContainText("Start a DCA session");
- await expect(options.nth(15)).toContainText("Capture a Durable Design Prototype");
- await expect(options.nth(21)).toContainText("Write a standup update");
+ await expect(options.nth(4)).toContainText("Create an isolated worktree");
+ await expect(options.nth(9)).toContainText("Launch a Managed Child");
+ await expect(options.nth(10)).toContainText("Start a DCA session");
+ await expect(options.nth(14)).toContainText("Send an update to another session");
+ await expect(options.nth(18)).toContainText("Capture a Durable Design Prototype");
+ await expect(options.nth(21)).toContainText("Build a system-design review package");
+
+ // A title-only tile grid, matching the reminder picker. A 22-item catalogue
+ // in full-width description rows put ~4.5 tiles on screen and never more
+ // than 1.5 group headings, so the grouping organised nothing. The tile must
+ // stay a real touch target, must not carry the description, and multiple
+ // tiles must share a row.
+ const [first, second] = await Promise.all([options.nth(0).boundingBox(), options.nth(1).boundingBox()]);
+ expect(first!.height, "tile is a real touch target").toBeGreaterThanOrEqual(44);
+ expect(Math.abs(first!.y - second!.y), "tiles share a row").toBeLessThanOrEqual(2);
+ await expect(options.nth(0)).not.toContainText("without a full deployment");
+ // The description leaves the tile but not the accessible name tree.
+ await expect(options.nth(0)).toHaveAttribute("title", /without a full deployment/u);
+ // Enough of the catalogue is reachable without scrolling that the group
+ // structure is actually visible: the old rows could not show two headings.
+ const panel = (await page.getByTestId("composer-workflow-panel").boundingBox())!;
+ const visibleTiles = await options.evaluateAll((nodes, bottom) =>
+ nodes.filter((node) => node.getBoundingClientRect().bottom <= bottom).length, panel.y + panel.height);
+ expect(visibleTiles, "at least half the catalogue fits one screen").toBeGreaterThanOrEqual(11);
// Choosing a workflow opens its form — it never sends.
await page.locator('[data-testid="composer-workflow-option"][data-workflow-id="playwright-ui-review"]').click();
@@ -268,6 +279,40 @@ test.describe("workflow picker UI", () => {
expect(text).not.toContain("$ARGUMENTS");
});
+ // The visible injector is the whole trust story (decision 21). The ported
+ // procedures run to ~160 lines against a ceiling sized for 19, so this pins
+ // the window well above the old `max-h-48` (192px) on both form factors.
+ for (const [name, width, height] of [["desktop", 1280, 800], ["mobile", 390, 740]] as const) {
+ test(`a long injector is genuinely reviewable on ${name}`, async ({ page }) => {
+ await page.setViewportSize({ width, height });
+ await page.goto(mainSession);
+ await page.getByTestId("composer-workflow-select").click();
+ // The tile grid has to survive the narrow sheet too: more than the ~4
+ // rows the old full-width description layout managed.
+ const options = page.getByTestId("composer-workflow-option");
+ await expect(options).toHaveCount(22);
+ const panel = (await page.getByTestId("composer-workflow-panel").boundingBox())!;
+ const visibleTiles = await options.evaluateAll((nodes, bottom) =>
+ nodes.filter((node) => node.getBoundingClientRect().bottom <= bottom).length, panel.y + panel.height);
+ expect(visibleTiles, `${name}: tiles visible without scrolling`).toBeGreaterThanOrEqual(width < 700 ? 8 : 11);
+
+ await page.locator('[data-testid="composer-workflow-option"][data-workflow-id="system-design-artifacts"]').click();
+ const dialog = page.getByTestId("composer-workflow-dialog");
+ await dialog.getByTestId("composer-workflow-field-argument").fill("the notification delivery system, current-state");
+ await dialog.getByTestId("composer-workflow-preview").click();
+
+ const body = dialog.getByTestId("composer-workflow-injector-body");
+ const { visible, total } = await body.evaluate((node) => ({ visible: node.clientHeight, total: node.scrollHeight }));
+ expect(total, "this injector is genuinely long").toBeGreaterThan(1_000);
+ expect(visible, `${name}: window is far above the old 192px ceiling`).toBeGreaterThan(280);
+ // Still bounded and still scrollable — the dialog must not become one
+ // unbroken page — and the end of the trusted text must be reachable.
+ expect(visible).toBeLessThan(total);
+ await body.evaluate((node) => { node.scrollTop = node.scrollHeight; });
+ expect(await body.evaluate((node) => node.scrollTop + node.clientHeight >= node.scrollHeight - 2)).toBe(true);
+ });
+ }
+
test("design prototype: no fields, fixed prompt, sends into this session", async ({ page }) => {
// The prompt is a fixed constant rather than a marker-bearing draft, so
// this counts matching payloads instead of asserting absence — a retry of
diff --git a/tests/workflows.test.ts b/tests/workflows.test.ts
index dc49dd7d..1a180483 100644
--- a/tests/workflows.test.ts
+++ b/tests/workflows.test.ts
@@ -32,33 +32,25 @@ import {
describe("workflow catalogue", () => {
it("shares semantic picker groups and sends unknown workflows to Other", () => {
const catalogue = [...workflowCatalogue(), { id: "future-workflow", title: "Future", description: "New server workflow", injector: "Do the future work." }];
- expect(groupWorkflows(catalogue).map(({ label }) => label)).toEqual(["Review", "Coordinate", "Execute", "Investigate", "Document", "Ship", "Other"]);
+ expect(groupWorkflows(catalogue).map(({ label }) => label)).toEqual(["Review", "Execute", "Delegate", "Coordinate", "Document", "Other"]);
expect(groupWorkflows(catalogue).at(-1)?.workflows.map(({ id }) => id)).toEqual(["future-workflow"]);
});
it("contains the shipped workflows, in picker order", () => {
expect(workflowCatalogue().map((workflow) => workflow.id)).toEqual([
PLAYWRIGHT_REVIEW_WORKFLOW_ID,
PR_SNIPPET_REVIEW_WORKFLOW_ID,
- "red-team",
- "review-learning",
- SESSION_UPDATE_WORKFLOW_ID,
+ "goal",
+ "dca",
+ "leaving-now-wrap-up",
MANAGED_CHILD_WORKFLOW_ID,
START_DCA_SESSION_WORKFLOW_ID,
- "manager-children",
- "native-worktree-subagents",
"session-handoff",
- "goal",
- "dca",
- "worktree-up",
- "deep-research",
- "research-handoff",
+ SESSION_UPDATE_WORKFLOW_ID,
+ "standup",
DESIGN_DOC_PROTOTYPE_WORKFLOW_ID,
"docs-preview",
"mini-design-doc",
"system-design-artifacts",
- "verify",
- "leaving-now-wrap-up",
- "standup",
]);
});
@@ -128,7 +120,7 @@ describe("generic argument workflows", () => {
const generic = workflowCatalogue().filter((workflow) => workflow.argument);
it("declares a usable, server-bounded field for every argument workflow", () => {
- expect(generic.length).toBeGreaterThanOrEqual(16);
+ expect(generic.length).toBeGreaterThanOrEqual(8);
for (const { id, argument } of generic) {
expect(argument!.label.trim(), `${id} has a blank label`).not.toBe("");
expect(argument!.label.length).toBeLessThanOrEqual(60);
From bea9685d1b1424b6448b6a559744b089ac6d8d5a Mon Sep 17 00:00:00 2001
From: Leon Cheng
Date: Sun, 30 Aug 2026 00:35:46 -0400
Subject: [PATCH 3/4] Align the workflow picker header and footer with the
reminder picker
Collapses the header to one row - search left, close right - and moves the
"nothing is sent until you confirm" promise into the footer hint. The
heading band cost a whole fold of a panel whose scarce resource is vertical
space, and the promise is still permanently on screen where it now sits.
Records why there is no tag chip row: workflows have no tags field, the five
group headings already carry the taxonomy, and a second one would have to be
authored, served and kept true. That is the one structural difference between
the two pickers and it is deliberate.
---
client/components/workflow-picker.tsx | 65 +++++++++++++++++++--------
1 file changed, 47 insertions(+), 18 deletions(-)
diff --git a/client/components/workflow-picker.tsx b/client/components/workflow-picker.tsx
index e99973b9..aad2429c 100644
--- a/client/components/workflow-picker.tsx
+++ b/client/components/workflow-picker.tsx
@@ -163,15 +163,16 @@ export function WorkflowPicker({
-
-
-
-
Workflows
-
Choosing a workflow opens a form. Nothing is sent or launched until you confirm.
-
-
-
-
+ {/*
+ * One header row — search on the left, close on the right — exactly
+ * as the reminder picker builds it. The heading and the promise that
+ * used to sit above the search cost a whole band of a panel whose
+ * scarce resource is vertical space; the promise moves to the footer
+ * hint, where it is still on screen at all times and no longer
+ * competes with the catalogue for the first fold.
+ */}
+
+
+
+ {/*
+ * No tag chip row here, deliberately — see the PR body. Workflows
+ * have no `tags` field, the five group headings already sit in the
+ * scroll flow, and a second taxonomy would have to be authored,
+ * served and kept true. This is the one place the two pickers
+ * differ structurally, and it is a choice rather than an omission.
+ */}
{attachedWorkflow && (
-
Up/Down to navigate - Enter to open the form - Esc to close
+ {/*
+ * The reminder picker's footer leads with what attaching a reminder
+ * actually does ("applies to the next message only") before the key
+ * hints. The workflow equivalent has to lead with the same kind of
+ * fact, and the honest one is that Enter opens a form: this picker
+ * cannot send or launch anything, which is the promise that used to
+ * live in the header.
+ */}
+
Choosing a workflow opens its form. Nothing is sent or launched until you confirm. Up/Down to navigate - Enter to open the form - Esc to close
,
document.body,
From 69aaafb48da1347bfbeef1443f1a4f9968404a59 Mon Sep 17 00:00:00 2001
From: Leon Cheng
Date: Sun, 30 Aug 2026 00:52:01 -0400
Subject: [PATCH 4/4] Re-fixture the E2E suite against the fourteen-workflow
catalogue
The cut left six specs asserting the old shape: card and option counts of 22,
a six-group chooser, and `nth()` spot-checks naming workflows that no longer
exist. It also left three specs anchored on the deleted `verify` id - the
Playbooks detail fixture, the generic-argument dialog test, and the reminder
details link.
Re-anchors the detail fixture on system-design-artifacts, which is now the
longest injector and so the honest subject for the full-procedure, light-mode
and phone-width tests. Re-anchors the generic-argument test on goal.
Recomputes the two derived numbers rather than guessing: the chooser's
"review" search matches four titles-or-ids, not five, and the order
spot-checks now name the first entry of each of the five groups so a workflow
moving between groups fails loudly instead of shifting an opaque index.
The reminder round-trip now exercises session-handoff, one of the two joins
that survives, and asserts that human-verification-steps renders no link at
all - its command was deleted rather than converted.
---
tests/e2e/playbooks.ui.spec.ts | 14 ++++----
tests/e2e/smoke.ui.spec.ts | 28 ++++++++-------
tests/e2e/workflows.ui.spec.ts | 41 +++++++++++-----------
tests/preview-e2e/public-simulator.spec.ts | 2 +-
4 files changed, 44 insertions(+), 41 deletions(-)
diff --git a/tests/e2e/playbooks.ui.spec.ts b/tests/e2e/playbooks.ui.spec.ts
index 49a3ada2..aab2a968 100644
--- a/tests/e2e/playbooks.ui.spec.ts
+++ b/tests/e2e/playbooks.ui.spec.ts
@@ -5,7 +5,7 @@ import { expect, test } from "@playwright/test";
// open a command detail page to get one. The command catalogue is retired, so
// they open a workflow detail page instead; the behaviour under test is the
// same modal component.
-const FIXTURE = "/playbooks/workflows/verify";
+const FIXTURE = "/playbooks/workflows/system-design-artifacts";
test.describe("Playbooks", () => {
test("is first-class navigation on the bar beside Planning", async ({ page }) => {
@@ -21,8 +21,8 @@ test.describe("Playbooks", () => {
// catalogue's own contract. It is deliberately exact: the retired command
// inventory used to be derived from disk, but this one has to match what
// the server actually served.
- await expect(page.getByTestId("opencode-playbook-workflow-card")).toHaveCount(22);
- await expect(page.getByTestId("opencode-playbook-workflow-verify")).toBeVisible();
+ await expect(page.getByTestId("opencode-playbook-workflow-card")).toHaveCount(14);
+ await expect(page.getByTestId("opencode-playbook-workflow-goal")).toBeVisible();
await expect(page.getByTestId("opencode-playbook-workflow-system-design-artifacts")).toBeVisible();
});
@@ -56,8 +56,8 @@ test.describe("Playbooks", () => {
test("renders live workflows by shared semantic group and shows the exact injector", async ({ page }) => {
await page.goto("/playbooks/workflows");
await expect(page).toHaveTitle("Workflows | Playbooks | DCA");
- await expect(page.getByTestId("opencode-playbook-workflow-group")).toHaveCount(6);
- await expect(page.getByTestId("opencode-playbook-workflow-card")).toHaveCount(22);
+ await expect(page.getByTestId("opencode-playbook-workflow-group")).toHaveCount(5);
+ await expect(page.getByTestId("opencode-playbook-workflow-card")).toHaveCount(14);
await page.getByTestId("opencode-playbook-workflow-start-dca-session").click();
await expect(page).toHaveTitle("Workflow | Playbooks | DCA");
await expect(page.getByTestId("opencode-playbook-workflow-injector")).toContainText("independent root session");
@@ -193,9 +193,9 @@ test.describe("Playbooks", () => {
await page.goto(FIXTURE);
await expect(page).toHaveTitle("Workflow | Playbooks | DCA");
- await expect(page.getByTestId("opencode-playbook-dialog").getByRole("heading", { name: "Run the checks, then write verification steps" })).toBeVisible();
+ await expect(page.getByTestId("opencode-playbook-dialog").getByRole("heading", { name: "Build a system-design review package" })).toBeVisible();
// The full ported procedure is present, tables and all.
- await expect(page.getByTestId("opencode-playbook-dialog")).toContainText("Failure");
+ await expect(page.getByTestId("opencode-playbook-dialog")).toContainText("Evidence is absent, contradictory, inaccessible, or intentionally unprobed");
expect(await page.evaluate(() => document.documentElement.scrollWidth)).toBeLessThanOrEqual(390);
// Full-screen on a phone, not a centred card with unreachable backdrop.
diff --git a/tests/e2e/smoke.ui.spec.ts b/tests/e2e/smoke.ui.spec.ts
index 07d27283..3869744f 100644
--- a/tests/e2e/smoke.ui.spec.ts
+++ b/tests/e2e/smoke.ui.spec.ts
@@ -1119,19 +1119,19 @@ test.describe("composer", () => {
await page.goto(`/sessions/ses_mock_done?directory=${encodeURIComponent(DIR)}`);
await page.getByTestId("composer-workflow-select").click();
await expect(page.getByTestId("composer-workflow-search")).toBeFocused();
- await expect(page.getByTestId("composer-workflow-option")).toHaveCount(22);
+ await expect(page.getByTestId("composer-workflow-option")).toHaveCount(14);
// Decision 21: this promise must survive the header gaining a search box.
await expect(page.getByTestId("composer-workflow-panel")).toContainText("Nothing is sent or launched until you confirm.");
- // Search matters more now that the catalogue is 22 entries rather than six,
- // so the needle has to be specific: "pull request" alone also describes the
- // review-learning workflow.
+ // Search matters more now that the catalogue is 14 entries rather than six,
+ // so the needle has to be specific: "PR review" alone also describes the
+ // Playwright UI review workflow.
await page.getByTestId("composer-workflow-search").fill("snippet-by-snippet");
await expect(page.getByTestId("composer-workflow-option")).toHaveCount(1);
await expect(page.getByTestId("composer-workflow-option")).toContainText("Post a snippet-by-snippet PR review");
// Search covers title and id only. Matching descriptions was harmless at
- // six workflows and is not at 22: "review" used to surface "Send an update
+ // six workflows and is not at 14: "review" used to surface "Send an update
// to another session" and "Start a DCA session", whose descriptions say
// "pre*view*" and "*review*ing" — tiles whose visible text does not contain
// what was typed, which reads as a bug.
@@ -1144,7 +1144,7 @@ test.describe("composer", () => {
// above is here.
await page.getByTestId("composer-workflow-search").fill("review");
const reviewHits = page.getByTestId("composer-workflow-option");
- await expect(reviewHits).toHaveCount(5);
+ await expect(reviewHits).toHaveCount(4);
for (const text of await reviewHits.allInnerTexts()) expect(text.toLowerCase()).toContain("review");
for (const id of ["session-update", "start-dca-session"]) {
await expect(page.locator(`[data-testid="composer-workflow-option"][data-workflow-id="${id}"]`)).toHaveCount(0);
@@ -1181,17 +1181,19 @@ test.describe("composer", () => {
await expect(page.getByTestId("composer-reminder-icon")).toHaveCount(13);
const humanVerification = page.locator('[data-testid="composer-reminder-option"][data-reminder-id="human-verification-steps"]');
await expect(humanVerification).toHaveAccessibleName("Attach Write Human Verification Steps");
- const details = page.locator('[data-testid="composer-reminder-details"][data-reminder-id="human-verification-steps"]');
- // The retired command catalogue used to host this documentation; the
- // ported workflow does now, and the link has to follow it rather than keep
- // resolving to a route that no longer exists.
- await expect(details).toHaveAttribute("href", "/playbooks/workflows/verify");
+ // Only the two reminders whose same-subject command was actually converted
+ // into a workflow carry a details link. `human-verification-steps` is not
+ // one of them: the `verify` command was deleted rather than ported, so this
+ // tile renders no link at all rather than one pointing at a dead route.
+ await expect(page.locator('[data-testid="composer-reminder-tile"][data-reminder-id="human-verification-steps"]').getByTestId("composer-reminder-details")).toHaveCount(0);
+ const details = page.locator('[data-testid="composer-reminder-details"][data-reminder-id="session-handoff"]');
+ await expect(details).toHaveAttribute("href", "/playbooks/workflows/session-handoff");
await expect(details).toHaveAttribute("target", "_blank");
- await expect(details).toHaveAccessibleName("Open Write Human Verification Steps details in a new tab");
+ await expect(details).toHaveAccessibleName("Open Hand Off to a New Session details in a new tab");
// The details link is a real touch target in its own right (matching
// this app's usual ~44px convention) but must still stay a minority of
// the tile's width -- the button beside it is the large target, not this.
- const tile = page.locator('[data-testid="composer-reminder-tile"][data-reminder-id="human-verification-steps"]');
+ const tile = page.locator('[data-testid="composer-reminder-tile"][data-reminder-id="session-handoff"]');
const [detailsBox, tileBox] = await Promise.all([details.boundingBox(), tile.boundingBox()]);
expect(detailsBox?.width, "details link is a real touch target").toBeGreaterThanOrEqual(40);
expect(detailsBox?.height, "details link is a real touch target").toBeGreaterThanOrEqual(40);
diff --git a/tests/e2e/workflows.ui.spec.ts b/tests/e2e/workflows.ui.spec.ts
index 8cdf5fa4..ae471b4f 100644
--- a/tests/e2e/workflows.ui.spec.ts
+++ b/tests/e2e/workflows.ui.spec.ts
@@ -149,22 +149,23 @@ test.describe("workflow picker UI", () => {
// workflow placed — nothing may fall into the "Other" bucket, which exists
// for a workflow a newer server ships and not for one this build forgot.
const options = page.getByTestId("composer-workflow-option");
- await expect(options).toHaveCount(22);
+ await expect(options).toHaveCount(14);
const groups = page.getByTestId("composer-workflow-group");
- await expect(groups).toHaveCount(6);
+ await expect(groups).toHaveCount(5);
for (const [index, label] of ["Review", "Execute", "Delegate", "Coordinate", "Document"].entries()) {
await expect(groups.nth(index)).toHaveAccessibleName(label);
}
- await expect(page.getByTestId("composer-workflow-icon")).toHaveCount(22);
+ await expect(page.getByTestId("composer-workflow-icon")).toHaveCount(14);
+ // Spot-check the first entry of every group, so a workflow silently moving
+ // between groups fails here rather than only shifting an opaque index.
await expect(options.nth(0)).toContainText("Review a UI change with Playwright");
- await expect(options.nth(1)).toContainText("Post a snippet-by-snippet PR review");
- await expect(options.nth(2)).toContainText("Red-team the work just produced");
- await expect(options.nth(4)).toContainText("Create an isolated worktree");
- await expect(options.nth(9)).toContainText("Launch a Managed Child");
- await expect(options.nth(10)).toContainText("Start a DCA session");
- await expect(options.nth(14)).toContainText("Send an update to another session");
- await expect(options.nth(18)).toContainText("Capture a Durable Design Prototype");
- await expect(options.nth(21)).toContainText("Build a system-design review package");
+ await expect(options.nth(2)).toContainText("Complete an objective autonomously");
+ await expect(options.nth(5)).toContainText("Launch a Managed Child");
+ await expect(options.nth(7)).toContainText("Hand off to another session");
+ await expect(options.nth(8)).toContainText("Send an update to another session");
+ await expect(options.nth(9)).toContainText("Write today's standup");
+ await expect(options.nth(10)).toContainText("Capture a Durable Design Prototype");
+ await expect(options.nth(13)).toContainText("Build a system-design review package");
// A title-only tile grid, matching the reminder picker. A 22-item catalogue
// in full-width description rows put ~4.5 tiles on screen and never more
@@ -240,16 +241,16 @@ test.describe("workflow picker UI", () => {
const marker = `WF-ARG-${Date.now()}`;
await page.goto(mainSession);
await page.getByTestId("composer-workflow-select").click();
- await page.locator('[data-testid="composer-workflow-option"][data-workflow-id="verify"]').click();
+ await page.locator('[data-testid="composer-workflow-option"][data-workflow-id="goal"]').click();
const dialog = page.getByTestId("composer-workflow-dialog");
- await expect(dialog).toHaveAttribute("data-workflow", "verify");
+ await expect(dialog).toHaveAttribute("data-workflow", "goal");
// One generic field, focused, described by the server's own spec — no
// bespoke branch exists for this workflow in the dialog.
const field = dialog.getByTestId("composer-workflow-field-argument");
await expect(field).toBeFocused();
- await expect(field).toHaveAttribute("placeholder", /notification popover/u);
- await expect(dialog.getByTestId("composer-workflow-argument-hint")).toContainText("Scopes the human checklist");
+ await expect(field).toHaveAttribute("placeholder", /notification badge/u);
+ await expect(dialog.getByTestId("composer-workflow-argument-hint")).toContainText("acceptance criteria");
// Required means required: an empty field cannot reach the preview.
await expect(dialog.getByTestId("composer-workflow-preview")).toBeDisabled();
await field.fill(` ${marker} `);
@@ -258,8 +259,8 @@ test.describe("workflow picker UI", () => {
// The typed text IS the prompt, trimmed and otherwise untouched.
await expect(dialog.getByTestId("composer-workflow-prompt-preview")).toHaveText(marker);
- await expect(dialog.getByTestId("composer-workflow-injector")).toContainText('server-resolved from id "verify"');
- await expect(dialog.getByTestId("composer-workflow-injector")).toContainText("Ready to ship");
+ await expect(dialog.getByTestId("composer-workflow-injector")).toContainText('server-resolved from id "goal"');
+ await expect(dialog.getByTestId("composer-workflow-injector")).toContainText("Complete the objective above as one sustained run.");
// The ported command pinned its own agent in frontmatter; a workflow cannot,
// so the preview has to say what governs instead of dropping it silently.
await expect(dialog.getByTestId("composer-workflow-mode-note")).toContainText("Sent in this session's current mode");
@@ -273,8 +274,8 @@ test.describe("workflow picker UI", () => {
expect(payload!.sessionID).toBe(MAIN);
const text = promptText(payload!);
expect(text).toContain(marker);
- expect(text).toContain('');
- expect(text).toContain("Ready to ship");
+ expect(text).toContain('');
+ expect(text).toContain("Complete the objective above as one sustained run.");
// The command's own substitution token must not survive the port.
expect(text).not.toContain("$ARGUMENTS");
});
@@ -290,7 +291,7 @@ test.describe("workflow picker UI", () => {
// The tile grid has to survive the narrow sheet too: more than the ~4
// rows the old full-width description layout managed.
const options = page.getByTestId("composer-workflow-option");
- await expect(options).toHaveCount(22);
+ await expect(options).toHaveCount(14);
const panel = (await page.getByTestId("composer-workflow-panel").boundingBox())!;
const visibleTiles = await options.evaluateAll((nodes, bottom) =>
nodes.filter((node) => node.getBoundingClientRect().bottom <= bottom).length, panel.y + panel.height);
diff --git a/tests/preview-e2e/public-simulator.spec.ts b/tests/preview-e2e/public-simulator.spec.ts
index 5a28c2a1..7bc999d9 100644
--- a/tests/preview-e2e/public-simulator.spec.ts
+++ b/tests/preview-e2e/public-simulator.spec.ts
@@ -19,7 +19,7 @@ test("serves an interactive, credential-free PR simulator", async ({ page }) =>
await expect(page.getByTestId("opencode-todo-list")).toContainText("Review the PR deployment");
await page.goto("./#/playbooks/workflows");
- await expect(page.getByTestId("opencode-playbook-workflow-card")).toHaveCount(6);
+ await expect(page.getByTestId("opencode-playbook-workflow-card")).toHaveCount(14);
await page.getByTestId("opencode-playbook-workflow-start-dca-session").click();
await expect(page.getByTestId("opencode-playbook-workflow-injector")).toContainText("independent root session");
await page.getByTestId("opencode-playbook-close").click();