From 3797c63d4aec9886711a0e8513b30549319bd1da Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 15:42:32 +0800 Subject: [PATCH 01/18] fix(executor): enforce workspace isolation for every tool call and scrub secrets MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Claude executor only path-checked Write/Edit/NotebookEdit, so a Bash call could write anywhere on the worker; the containment test even asserted Bash was allowed. The check also used `startsWith(writeRoot)`, so a sibling directory `-evil` passed. `workspace.access: readOnly` was declared in the template schema and plumbed to checkout but never enforced, so "read-only" templates still had full write + shell. `settingSources: ["project"]` loaded `.claude/settings.json` (hooks, permission overrides) and `.mcp.json` straight from the checked-out — untrusted — repository, and the SDK subprocess inherited the full worker environment including AGRIPPA_SECRET_KEY (the master key that decrypts every stored credential) and the datastore URLs, all reachable from a Bash `env`. The worker container also ran as root on a shared volume. Introduce one execution-isolation seam in executor-core (`evaluateToolCall`, `isWithin`, `buildScrubbedEnv`) that the SDK adapter must route every decision through, rather than reimplementing containment inline: - read-only workspaces deny shell and confine writes to `.agrippa/artifacts`; read-write workspaces allow shell (OS-sandboxed when available) and confine writes to the workspace, boundary-safe against the sibling-prefix bug; - `ToolPolicy.access` is now required and wired from the template workspace spec; - the adapter enables the SDK `sandbox` (bubblewrap, graceful when absent), passes `strictMcpConfig`, an explicit `skills` allowlist, and a scrubbed subprocess env that drops the master key and datastore URLs while keeping the Anthropic auth vars the SDK needs; - the worker strips repo-supplied `.claude`/`.mcp.json` after checkout so the project setting source can only load platform-controlled skills; - the worker image runs as the non-root `bun` user. Full Bash containment in read-write workspaces still relies on the OS sandbox / non-root worker — the static layer cannot bound arbitrary shell writes; this is called out for the execution-isolation deep module (docs/design/03). --- apps/worker/src/deps/workspace.ts | 16 +++ infra/Dockerfile.worker | 6 + packages/executor-claude/src/executor.test.ts | 74 ++++++---- packages/executor-claude/src/executor.ts | 46 ++++--- packages/executor-core/src/index.ts | 1 + packages/executor-core/src/isolation.test.ts | 75 +++++++++++ packages/executor-core/src/isolation.ts | 127 ++++++++++++++++++ packages/executor-core/src/types.ts | 7 + packages/orchestration/src/engine/engine.ts | 7 +- 9 files changed, 317 insertions(+), 42 deletions(-) create mode 100644 packages/executor-core/src/isolation.test.ts create mode 100644 packages/executor-core/src/isolation.ts diff --git a/apps/worker/src/deps/workspace.ts b/apps/worker/src/deps/workspace.ts index d064c16..e73590d 100644 --- a/apps/worker/src/deps/workspace.ts +++ b/apps/worker/src/deps/workspace.ts @@ -7,6 +7,21 @@ import { eq } from "drizzle-orm"; const WORKSPACE_ROOT = process.env.WORKSPACE_ROOT ?? path.join(tmpdir(), "agrippa-workspaces"); +/** + * Repo-supplied agent configuration that would otherwise be honored by the SDK + * project setting source: hooks run shell, settings grant tool permissions, and + * .mcp.json wires servers. A checked-out repo is untrusted, so these are removed + * before any agent runs. Registry skills are re-materialized into .claude/skills + * afterwards (docs/design/03 §Sandboxing). + */ +const REPO_CONFIG_TO_STRIP = [".claude", ".mcp.json"]; + +async function sanitizeWorkspace(dir: string): Promise { + for (const entry of REPO_CONFIG_TO_STRIP) { + await rm(path.join(dir, entry), { recursive: true, force: true }); + } +} + async function git(args: string[], cwd?: string): Promise { const proc = Bun.spawn(["git", ...args], { cwd, @@ -74,6 +89,7 @@ export class GitWorkspaceManager implements WorkspaceManager { await git(["clone", "--depth", "50", "--branch", ref, cloneUrl, dir]); // scrub the credential from the remote before any agent code runs await git(["remote", "set-url", "origin", connection.url], dir); + await sanitizeWorkspace(dir); } async diff(runId: string): Promise { diff --git a/infra/Dockerfile.worker b/infra/Dockerfile.worker index 2c3bfc3..fc90d08 100644 --- a/infra/Dockerfile.worker +++ b/infra/Dockerfile.worker @@ -18,4 +18,10 @@ ENV NODE_ENV=production ENV AGRIPPA_TEMPLATES_DIR=/app/templates ENV WORKSPACE_ROOT=/work/runs ENV ARTIFACT_STORAGE_ROOT=/work/artifacts + +# Run agent code as a non-root user so a shell tool call cannot reach the +# container filesystem or other roots (docs/design/03 §Sandboxing). Chowning +# /work here makes a freshly-created named volume inherit `bun` ownership. +RUN mkdir -p /work/runs /work/artifacts && chown -R bun:bun /app /work +USER bun CMD ["bun", "apps/worker/src/index.ts"] diff --git a/packages/executor-claude/src/executor.test.ts b/packages/executor-claude/src/executor.test.ts index c06c093..e07bfb1 100644 --- a/packages/executor-claude/src/executor.test.ts +++ b/packages/executor-claude/src/executor.test.ts @@ -36,7 +36,7 @@ function makeRequest(overrides: Partial = {}): StepExecuti mcpServers: [ { slug: "github", transport: "http", url: "https://mcp.example/", headers: { a: "b" } }, ], - toolPolicy: { writeRoot: workspaceDir }, + toolPolicy: { writeRoot: workspaceDir, access: "readWrite" }, limits: { maxTurns: 50 }, workspaceDir, priorContext: [{ stepId: "earlier", output: "earlier result", artifactKeys: [] }], @@ -99,30 +99,58 @@ describe("claude executor option mapping (docs/design/03)", () => { expect(prompt).toContain(".agrippa/artifacts/fix-report.md"); }); - it("canUseTool denies writes outside the workspace and allows inside", async () => { - const req = makeRequest(); - const { options } = buildQueryArgs(req, makeCtx(), new AbortController()); - const canUseTool = options.canUseTool; - if (!canUseTool) throw new Error("canUseTool missing"); - - const denied = await canUseTool("Write", { file_path: "/etc/passwd" }, { - signal: new AbortController().signal, - suggestions: [], - } as never); - expect(denied?.behavior).toBe("deny"); - - const allowed = await canUseTool( - "Write", - { file_path: path.join(req.workspaceDir, "src/x.ts") }, - { signal: new AbortController().signal, suggestions: [] } as never, + it("canUseTool contains writes and shell per the workspace policy", async () => { + const ctx = { signal: new AbortController().signal, suggestions: [] } as never; + const rwReq = makeRequest(); + const rw = buildQueryArgs(rwReq, makeCtx(), new AbortController()).options.canUseTool; + if (!rw) throw new Error("canUseTool missing"); + + // read-write: writes inside the workspace allowed, escaping writes denied + expect((await rw("Write", { file_path: "/etc/passwd" }, ctx))?.behavior).toBe("deny"); + expect( + (await rw("Write", { file_path: path.join(rwReq.workspaceDir, "src/x.ts") }, ctx))?.behavior, + ).toBe("allow"); + // sibling-prefix directory must not pass the containment check + expect((await rw("Write", { file_path: `${rwReq.workspaceDir}-evil/x` }, ctx))?.behavior).toBe( + "deny", ); - expect(allowed?.behavior).toBe("allow"); + // read-write: shell is permitted (OS-sandboxed when available) + expect((await rw("Bash", { command: "ls" }, ctx))?.behavior).toBe("allow"); + + // read-only: shell denied, repo writes denied, artifact writes allowed + const roReq = makeRequest(); + roReq.toolPolicy = { writeRoot: roReq.workspaceDir, access: "readOnly" }; + const ro = buildQueryArgs(roReq, makeCtx(), new AbortController()).options.canUseTool; + if (!ro) throw new Error("canUseTool missing"); + expect((await ro("Bash", { command: "ls" }, ctx))?.behavior).toBe("deny"); + expect( + (await ro("Write", { file_path: path.join(roReq.workspaceDir, "src/x.ts") }, ctx))?.behavior, + ).toBe("deny"); + expect( + ( + await ro( + "Write", + { file_path: path.join(roReq.workspaceDir, ".agrippa/artifacts/report.md") }, + ctx, + ) + )?.behavior, + ).toBe("allow"); + }); - const bash = await canUseTool("Bash", { command: "ls" }, { - signal: new AbortController().signal, - suggestions: [], - } as never); - expect(bash?.behavior).toBe("allow"); + it("scrubs platform secrets from the agent subprocess env", () => { + const req = makeRequest(); + const prev = process.env.AGRIPPA_SECRET_KEY; + process.env.AGRIPPA_SECRET_KEY = "master-key"; + try { + const { options } = buildQueryArgs(req, makeCtx(), new AbortController()); + expect(options.env?.AGRIPPA_SECRET_KEY).toBeUndefined(); + expect(options.env?.DATABASE_URL).toBeUndefined(); + // strict MCP + no repo settings/hooks honored + expect(options.strictMcpConfig).toBe(true); + } finally { + if (prev === undefined) delete process.env.AGRIPPA_SECRET_KEY; + else process.env.AGRIPPA_SECRET_KEY = prev; + } }); }); diff --git a/packages/executor-claude/src/executor.ts b/packages/executor-claude/src/executor.ts index 041d56d..cf27885 100644 --- a/packages/executor-claude/src/executor.ts +++ b/packages/executor-claude/src/executor.ts @@ -1,11 +1,13 @@ import { readdirSync } from "node:fs"; import path from "node:path"; import type { ArtifactKind } from "@agrippa/core"; -import type { - ExecutionContext, - Executor, - ExecutorEvent, - StepExecutionRequest, +import { + buildScrubbedEnv, + type ExecutionContext, + type Executor, + type ExecutorEvent, + evaluateToolCall, + type StepExecutionRequest, } from "@agrippa/executor-core"; import { type Options, type SDKMessage, query as sdkQuery } from "@anthropic-ai/claude-agent-sdk"; @@ -83,15 +85,27 @@ export function buildQueryArgs( } } - const writeRoot = path.resolve(req.toolPolicy.writeRoot); + // Only the skills we materialized are enabled — never whatever the checked-out + // repo happens to ship under .claude/skills (docs/design/03 §Sandboxing). + const skillNames = req.skills.map((s) => s.slug.split("/").pop() as string); const options: Options = { cwd: req.workspaceDir, model: req.model.providerModelId, systemPrompt: { type: "preset", preset: "claude_code", append: req.systemPrompt }, agents: Object.keys(agents).length > 0 ? agents : undefined, mcpServers: Object.keys(mcpServers).length > 0 ? mcpServers : undefined, - // skills load from /.claude/skills via project settings + // skills load from /.claude/skills; 'project' also pulls CLAUDE.md. + // The worker strips repo-supplied .claude settings/hooks before this runs, so + // 'project' cannot load attacker-controlled hooks or permission overrides. settingSources: ["project"], + skills: skillNames.length > 0 ? skillNames : undefined, + // ignore any .mcp.json in the checked-out repo — only our resolved servers + strictMcpConfig: true, + // OS-level command isolation when the host supports it (bubblewrap); degrade + // gracefully elsewhere (e.g. macOS dev) rather than refusing to run + sandbox: { enabled: true, failIfUnavailable: false }, + // secrets (master key, datastore URLs) must not reach the agent subprocess + env: buildScrubbedEnv(), maxTurns: req.limits.maxTurns, includePartialMessages: true, resume: req.resumeSessionId, @@ -100,17 +114,13 @@ export function buildQueryArgs( abortController, permissionMode: "acceptEdits", canUseTool: async (toolName, input) => { - // deny writes escaping the run workspace; everything else proceeds - const target = (input.file_path ?? input.path ?? input.notebook_path) as string | undefined; - if (target && ["Write", "Edit", "NotebookEdit"].includes(toolName)) { - const resolved = path.resolve(req.workspaceDir, target); - if (!resolved.startsWith(writeRoot)) { - return { - behavior: "deny", - message: `writes outside the run workspace are not permitted (${target})`, - }; - } - } + const decision = evaluateToolCall( + req.toolPolicy, + req.workspaceDir, + toolName, + input as Record, + ); + if (decision.behavior === "deny") return decision; return { behavior: "allow", updatedInput: input }; }, }; diff --git a/packages/executor-core/src/index.ts b/packages/executor-core/src/index.ts index c2cc7cc..23359f1 100644 --- a/packages/executor-core/src/index.ts +++ b/packages/executor-core/src/index.ts @@ -1,3 +1,4 @@ export * from "./budget"; export * from "./fake-executor"; +export * from "./isolation"; export * from "./types"; diff --git a/packages/executor-core/src/isolation.test.ts b/packages/executor-core/src/isolation.test.ts new file mode 100644 index 0000000..e199348 --- /dev/null +++ b/packages/executor-core/src/isolation.test.ts @@ -0,0 +1,75 @@ +import { describe, expect, it } from "bun:test"; +import path from "node:path"; +import { buildScrubbedEnv, evaluateToolCall, isWithin } from "./isolation"; + +const ROOT = "/work/runs/run-1"; +const rw = { access: "readWrite" as const, writeRoot: ROOT }; +const ro = { access: "readOnly" as const, writeRoot: ROOT }; + +describe("isWithin", () => { + it("accepts the root and nested paths, rejects siblings and escapes", () => { + expect(isWithin(ROOT, ROOT)).toBe(true); + expect(isWithin(ROOT, `${ROOT}/src/a.ts`)).toBe(true); + // the sibling-prefix bug: /work/runs/run-1 must NOT contain /work/runs/run-1-evil + expect(isWithin(ROOT, `${ROOT}-evil/a.ts`)).toBe(false); + expect(isWithin(ROOT, "/etc/passwd")).toBe(false); + expect(isWithin(ROOT, `${ROOT}/../run-2/a.ts`)).toBe(false); + }); +}); + +describe("evaluateToolCall — read-write workspace", () => { + it("allows in-workspace writes and shell, denies escaping writes", () => { + expect(evaluateToolCall(rw, ROOT, "Write", { file_path: `${ROOT}/a.ts` }).behavior).toBe( + "allow", + ); + expect(evaluateToolCall(rw, ROOT, "Bash", { command: "ls" }).behavior).toBe("allow"); + expect(evaluateToolCall(rw, ROOT, "Write", { file_path: "/etc/cron.d/x" }).behavior).toBe( + "deny", + ); + // relative path resolves against the workspace, escaping is denied + expect(evaluateToolCall(rw, ROOT, "Edit", { file_path: "../run-2/a" }).behavior).toBe("deny"); + }); +}); + +describe("evaluateToolCall — read-only workspace", () => { + it("denies shell and repo writes, permits artifact writes", () => { + expect(evaluateToolCall(ro, ROOT, "Bash", { command: "ls" }).behavior).toBe("deny"); + expect(evaluateToolCall(ro, ROOT, "Write", { file_path: `${ROOT}/src/a.ts` }).behavior).toBe( + "deny", + ); + expect( + evaluateToolCall(ro, ROOT, "Write", { + file_path: path.join(ROOT, ".agrippa/artifacts/report.md"), + }).behavior, + ).toBe("allow"); + // reads are always fine + expect(evaluateToolCall(ro, ROOT, "Read", { file_path: `${ROOT}/src/a.ts` }).behavior).toBe( + "allow", + ); + }); +}); + +describe("buildScrubbedEnv", () => { + it("drops platform secrets but keeps Anthropic auth and system vars", () => { + const env = buildScrubbedEnv({ + PATH: "/usr/bin", + HOME: "/home/bun", + ANTHROPIC_API_KEY: "sk-ant-xxx", + AGRIPPA_SECRET_KEY: "master", + DATABASE_URL: "postgres://secret", + BETTER_AUTH_SECRET: "s", + REDIS_URL: "redis://x", + GITHUB_TOKEN: "ghp_x", + SOME_PASSWORD: "p", + }); + expect(env.PATH).toBe("/usr/bin"); + expect(env.HOME).toBe("/home/bun"); + expect(env.ANTHROPIC_API_KEY).toBe("sk-ant-xxx"); + expect(env.AGRIPPA_SECRET_KEY).toBeUndefined(); + expect(env.DATABASE_URL).toBeUndefined(); + expect(env.BETTER_AUTH_SECRET).toBeUndefined(); + expect(env.REDIS_URL).toBeUndefined(); + expect(env.GITHUB_TOKEN).toBeUndefined(); + expect(env.SOME_PASSWORD).toBeUndefined(); + }); +}); diff --git a/packages/executor-core/src/isolation.ts b/packages/executor-core/src/isolation.ts new file mode 100644 index 0000000..9017aa9 --- /dev/null +++ b/packages/executor-core/src/isolation.ts @@ -0,0 +1,127 @@ +import path from "node:path"; + +/** + * Execution-isolation seam (docs/design/03 §Sandboxing, ADR-0005). + * + * One enforceable place for the workspace containment rules the SDK adapter + * applies to every tool call, plus the environment-scrubbing rule for the + * subprocess it spawns. Kept pure and synchronous so the same logic backs the + * adapter and its tests — the adapter must not re-implement any of it. + * + * What this layer can and cannot do: it statically contains the file-writing + * tools (Write/Edit/NotebookEdit) and refuses shell in read-only workspaces. + * It does **not** contain arbitrary writes a shell command makes in a + * read-write workspace — that requires OS-level isolation (the SDK `sandbox` + * option / a non-root worker / a container), layered on top by the adapter. + */ + +export type WorkspaceAccess = "readOnly" | "readWrite"; + +/** Artifact convention directory (relative to the workspace root). */ +export const ARTIFACT_SUBDIR = ".agrippa/artifacts"; + +/** Tools that write to the filesystem through a file_path / path arg. */ +const WRITE_TOOLS = new Set(["Write", "Edit", "MultiEdit", "NotebookEdit"]); +/** Tools that execute arbitrary commands (uncontainable by static rules). */ +const EXEC_TOOLS = new Set(["Bash", "BashOutput", "KillShell", "KillBash"]); + +export type ToolDecision = { behavior: "allow" } | { behavior: "deny"; message: string }; + +/** + * Is `child` the same path as, or nested under, `parent`? Boundary-safe: + * `/work/run` does NOT contain `/work/run-evil` (the naive `startsWith` + * prefix test does, which is the bug this replaces). + */ +export function isWithin(parent: string, child: string): boolean { + const p = path.resolve(parent); + const c = path.resolve(child); + return c === p || c.startsWith(p + path.sep); +} + +/** + * Decide a single tool call against the workspace policy. + * + * - read-only workspace: file writes are confined to the artifact directory + * (the agent still has to emit its declared artifacts) and shell is denied; + * - read-write workspace: file writes must stay within the workspace root; + * shell is allowed (contained by the OS sandbox when available). + */ +export function evaluateToolCall( + policy: { access: WorkspaceAccess; writeRoot: string }, + workspaceDir: string, + toolName: string, + input: Record, +): ToolDecision { + const writeRoot = path.resolve(policy.writeRoot); + + if (EXEC_TOOLS.has(toolName)) { + if (policy.access === "readOnly") { + return { + behavior: "deny", + message: `shell commands are not permitted in a read-only workspace (${toolName})`, + }; + } + return { behavior: "allow" }; + } + + if (WRITE_TOOLS.has(toolName)) { + const target = (input.file_path ?? input.path ?? input.notebook_path) as string | undefined; + if (target === undefined) return { behavior: "allow" }; + const resolved = path.resolve(workspaceDir, target); + if (!isWithin(writeRoot, resolved)) { + return { + behavior: "deny", + message: `writes outside the run workspace are not permitted (${target})`, + }; + } + if ( + policy.access === "readOnly" && + !isWithin(path.join(writeRoot, ARTIFACT_SUBDIR), resolved) + ) { + return { + behavior: "deny", + message: `read-only workspace: writes are confined to ${ARTIFACT_SUBDIR} (${target})`, + }; + } + } + + return { behavior: "allow" }; +} + +/** + * Environment variables that must never reach the agent subprocess: leaking + * `AGRIPPA_SECRET_KEY` decrypts every stored credential, and the datastore + * URLs grant direct access to run/tenant data. The Anthropic/Claude vars the + * SDK needs to authenticate are preserved. + */ +const SECRET_ENV_KEYS = new Set([ + "AGRIPPA_SECRET_KEY", + "DATABASE_URL", + "TEST_DATABASE_URL", + "BETTER_AUTH_SECRET", + "REDIS_URL", +]); + +/** Heuristic secret-name match, minus the Anthropic/Claude auth vars we keep. */ +function looksSecret(key: string): boolean { + if (/^(ANTHROPIC_|CLAUDE_)/.test(key)) return false; + return /(SECRET|PASSWORD|PRIVATE_KEY|_TOKEN$|_KEY$)/i.test(key); +} + +/** + * Build the subprocess environment for the agent, dropping platform secrets. + * The SDK's `env` option REPLACES the child environment wholesale, so we start + * from the worker env and remove what the agent must not see, rather than + * allow-listing (which would starve the CLI of PATH/HOME/locale it needs). + */ +export function buildScrubbedEnv( + source: Record = process.env, +): Record { + const out: Record = {}; + for (const [key, value] of Object.entries(source)) { + if (value === undefined) continue; + if (SECRET_ENV_KEYS.has(key) || looksSecret(key)) continue; + out[key] = value; + } + return out; +} diff --git a/packages/executor-core/src/types.ts b/packages/executor-core/src/types.ts index ca67088..f1cb522 100644 --- a/packages/executor-core/src/types.ts +++ b/packages/executor-core/src/types.ts @@ -1,4 +1,5 @@ import type { ArtifactKind } from "@agrippa/core"; +import type { WorkspaceAccess } from "./isolation"; /** * The Executor contract (docs/design/03-executor-abstraction.md, ADR-0005). @@ -44,6 +45,12 @@ export type ToolPolicy = { disallowedTools?: string[]; /** Absolute path writes must stay within (the run workspace). */ writeRoot: string; + /** + * Repo access declared by the template workspace. `readOnly` confines writes + * to the artifact directory and forbids shell; `readWrite` allows both within + * the workspace. Enforced by evaluateToolCall in ./isolation. + */ + access: WorkspaceAccess; }; export type PriorStepSummary = { diff --git a/packages/orchestration/src/engine/engine.ts b/packages/orchestration/src/engine/engine.ts index 91c6104..d35f71c 100644 --- a/packages/orchestration/src/engine/engine.ts +++ b/packages/orchestration/src/engine/engine.ts @@ -552,7 +552,12 @@ class RunEngine { subagents, skills, mcpServers, - toolPolicy: { writeRoot: this.workspaceDir }, + // no workspace repo → scratch dir with nothing to protect (readWrite); + // a repo checkout carries the template's declared access (default readOnly) + toolPolicy: { + writeRoot: this.workspaceDir, + access: this.template.spec.workspace?.access ?? "readWrite", + }, limits: { maxTurns: 50 }, workspaceDir: this.workspaceDir, resumeSessionId: row.executorSessionId ?? undefined, From e97263a06cb77a5417981921dd6b0b50d4e5570b Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 15:43:55 +0800 Subject: [PATCH 02/18] fix(worker): contain artifact ingestion against symlink escapes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The artifact store resolved an executor-controlled source path lexically and read it whole, following symlinks. An agent could `ln -s /proc/self/environ` (or another run's files under the shared /work volume) into `.agrippa/artifacts/`, and the linked bytes were ingested as this run's artifact and served by the download endpoint to any run viewer — a filesystem and secret disclosure. Resolve the source through realpath and require the real target to sit inside the workspace (reusing the isolation seam's `isWithin`); missing/broken sources yield no content instead of a zero-byte row. Adds the first worker-adapter tests: normal file stored, escaping symlink rejected, missing file is not an artifact. --- apps/worker/src/deps/artifacts.test.ts | 70 ++++++++++++++++++++++++++ apps/worker/src/deps/artifacts.ts | 32 ++++++++++-- 2 files changed, 99 insertions(+), 3 deletions(-) create mode 100644 apps/worker/src/deps/artifacts.test.ts diff --git a/apps/worker/src/deps/artifacts.test.ts b/apps/worker/src/deps/artifacts.test.ts new file mode 100644 index 0000000..e069071 --- /dev/null +++ b/apps/worker/src/deps/artifacts.test.ts @@ -0,0 +1,70 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import { mkdtempSync, rmSync, symlinkSync, writeFileSync } from "node:fs"; +import { mkdir } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { DiskArtifactStore } from "./artifacts"; + +const store = new DiskArtifactStore(); +const dirs: string[] = []; + +function freshWorkspace(): string { + const dir = mkdtempSync(path.join(tmpdir(), "agrippa-ws-")); + dirs.push(dir); + return dir; +} + +afterEach(() => { + for (const dir of dirs.splice(0)) rmSync(dir, { recursive: true, force: true }); +}); + +describe("DiskArtifactStore path containment", () => { + it("stores a normal in-workspace file inline", async () => { + const ws = freshWorkspace(); + await mkdir(path.join(ws, ".agrippa/artifacts"), { recursive: true }); + writeFileSync(path.join(ws, ".agrippa/artifacts/report.md"), "# ok"); + + const stored = await store.store( + "run-1", + "report", + "markdown", + { + path: ".agrippa/artifacts/report.md", + }, + ws, + ); + expect(stored.inline).toBe("# ok"); + expect(stored.size).toBeGreaterThan(0); + }); + + it("rejects a symlink escaping the workspace", async () => { + const ws = freshWorkspace(); + await mkdir(path.join(ws, ".agrippa/artifacts"), { recursive: true }); + // secret file outside the workspace, reachable only via the symlink + const outside = mkdtempSync(path.join(tmpdir(), "agrippa-secret-")); + dirs.push(outside); + const secret = path.join(outside, "environ"); + writeFileSync(secret, "AGRIPPA_SECRET_KEY=master"); + symlinkSync(secret, path.join(ws, ".agrippa/artifacts/leak.md")); + + await expect( + store.store("run-1", "leak", "markdown", { path: ".agrippa/artifacts/leak.md" }, ws), + ).rejects.toThrow(/escapes the run workspace/); + }); + + it("treats a missing file as no content, not a zero-byte artifact", async () => { + const ws = freshWorkspace(); + const stored = await store.store( + "run-1", + "nope", + "markdown", + { + path: ".agrippa/artifacts/nope.md", + }, + ws, + ); + expect(stored.inline).toBeNull(); + expect(stored.storageRef).toBeNull(); + expect(stored.size).toBe(0); + }); +}); diff --git a/apps/worker/src/deps/artifacts.ts b/apps/worker/src/deps/artifacts.ts index 5e1bb8e..151642b 100644 --- a/apps/worker/src/deps/artifacts.ts +++ b/apps/worker/src/deps/artifacts.ts @@ -1,12 +1,36 @@ -import { mkdir } from "node:fs/promises"; +import { mkdir, realpath } from "node:fs/promises"; import { tmpdir } from "node:os"; import path from "node:path"; import type { ArtifactKind } from "@agrippa/core"; +import { isWithin } from "@agrippa/executor-core"; import type { ArtifactStore, StoredArtifact } from "@agrippa/orchestration"; const STORAGE_ROOT = process.env.ARTIFACT_STORAGE_ROOT ?? path.join(tmpdir(), "agrippa-artifacts"); const INLINE_LIMIT = 64 * 1024; +const EMPTY: StoredArtifact = { inline: null, storageRef: null, size: 0, mime: null }; + +/** + * Resolve a workspace-relative artifact source to a real path that is provably + * inside the workspace. Following symlinks (via realpath) is the point: an + * agent can `ln -s /proc/self/environ .agrippa/artifacts/leak.md`, and a purely + * lexical containment check would pass while the read escaped. Returns null when + * the source does not exist (missing files are not artifacts). + */ +async function resolveContainedPath(workspaceDir: string, rel: string): Promise { + const root = await realpath(workspaceDir); + let real: string; + try { + real = await realpath(path.resolve(workspaceDir, rel)); + } catch { + return null; // missing / broken symlink + } + if (!isWithin(root, real)) { + throw new Error(`artifact source escapes the run workspace: ${rel}`); + } + return real; +} + /** ≤64 KB inline in Postgres; larger content on the artifacts volume. */ export class DiskArtifactStore implements ArtifactStore { async store( @@ -23,13 +47,15 @@ export class DiskArtifactStore implements ArtifactStore { content = typeof source.inline === "string" ? source.inline : JSON.stringify(source.inline); mime = kind === "json" ? "application/json" : "text/markdown"; } else if (source.path) { - const file = Bun.file(path.resolve(workspaceDir, source.path)); + const real = await resolveContainedPath(workspaceDir, source.path); + if (real === null) return EMPTY; + const file = Bun.file(real); if (await file.exists()) { content = await file.text(); mime = file.type || null; } } - if (content === null) return { inline: null, storageRef: null, size: 0, mime: null }; + if (content === null) return EMPTY; const size = Buffer.byteLength(content); if (size <= INLINE_LIMIT) { From 6749bcb2fbab734a1fc1b52d54c56e3a72b25cdc Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 15:54:36 +0800 Subject: [PATCH 03/18] fix(authz): pin an authorized run manifest and scope repo connections to the project MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two cross-tenant authorization holes shared one root cause: the worker re-resolved mutable global resources that submission never fully authorized. - repoRef was only shape-validated (a UUID), and the worker loaded the repo connection by raw id with no project predicate. A member of project A could submit project B's repoConnectionId and the worker would clone B's private repo with B's stored token. Submission now rejects a repoConnectionId that is not owned by the project (verifyRepoRefs), and the worker loads the connection scoped to the run's projectId as defence in depth. - optional skills/MCP skipped the grant check at submit, and the worker resolved them from the global registry with the platform credential — a project with no GitHub grant still received the shared GitHub token when autoOpenPr enabled the optional server. Submission now pins an authorized resource manifest (required grants enforced, optional resources included only when granted) onto the run; the engine resolves skills/MCP only from that manifest, never the global registry. Ungranted optional resources are treated as unavailable, so the dependent step is skipped. Adds runs.resource_manifest (migration 0002) and regression tests: a cross-project repoConnectionId is refused, and an ungranted optional MCP server is never resolved even when it exists in the registry. --- apps/api/src/routes/execution.ts | 11 +- .../src/test/execution.integration.test.ts | 39 +- apps/worker/src/deps/workspace.ts | 11 +- .../db/drizzle/0002_run-resource-manifest.sql | 1 + packages/db/drizzle/meta/0002_snapshot.json | 3068 +++++++++++++++++ packages/db/drizzle/meta/_journal.json | 7 + packages/db/src/schema/runs.ts | 6 + packages/orchestration/src/engine/deps.ts | 2 + .../src/engine/engine.integration.test.ts | 27 + packages/orchestration/src/engine/engine.ts | 43 +- packages/orchestration/src/resolve.ts | 83 +- 11 files changed, 3272 insertions(+), 26 deletions(-) create mode 100644 packages/db/drizzle/0002_run-resource-manifest.sql create mode 100644 packages/db/drizzle/meta/0002_snapshot.json diff --git a/apps/api/src/routes/execution.ts b/apps/api/src/routes/execution.ts index 496a2c6..cedfb43 100644 --- a/apps/api/src/routes/execution.ts +++ b/apps/api/src/routes/execution.ts @@ -18,11 +18,12 @@ import { templateVersions, } from "@agrippa/db"; import { + authorizeResources, buildParamsValidator, resolveModelRoles, SubmitError, type TemplateDoc, - verifyResourceGrants, + verifyRepoRefs, } from "@agrippa/orchestration"; import { and, asc, desc, eq, gt, max } from "drizzle-orm"; import { Hono } from "hono"; @@ -108,11 +109,15 @@ export const executionRoutes = new Hono() await assertQuotaHeadroom(db, projectId); try { + // every repoRef must reference a connection owned by this project + await verifyRepoRefs(db, projectId, compiled.spec.inputs, parsed.data); + const skillRows = await db.select({ id: skills.id, slug: skills.slug }).from(skills); const mcpRows = await db .select({ id: mcpServers.id, slug: mcpServers.slug }) .from(mcpServers); - await verifyResourceGrants(db, projectId, compiled, { + // required grants enforced; optional resources pinned only when granted + const resourceManifest = await authorizeResources(db, projectId, compiled, { skillIdBySlug: new Map(skillRows.map((s) => [s.slug, s.id])), mcpIdBySlug: new Map(mcpRows.map((m) => [m.slug, m.id])), }); @@ -142,6 +147,7 @@ export const executionRoutes = new Hono() executorId: DEFAULT_EXECUTOR, paramsSnapshot: parsed.data, modelResolution, + resourceManifest, budget: compiled.spec.budgets as unknown as Record, createdBy: user.id, }) @@ -239,6 +245,7 @@ export const executionRoutes = new Hono() executorId: latest.executorId, paramsSnapshot: latest.paramsSnapshot, modelResolution: latest.modelResolution, + resourceManifest: latest.resourceManifest, budget: latest.budget, createdBy: c.var.user.id, }) diff --git a/apps/api/src/test/execution.integration.test.ts b/apps/api/src/test/execution.integration.test.ts index 157d3bf..9bccd05 100644 --- a/apps/api/src/test/execution.integration.test.ts +++ b/apps/api/src/test/execution.integration.test.ts @@ -1,6 +1,6 @@ import { beforeAll, describe, expect, it } from "bun:test"; import type { RunQueue } from "@agrippa/core"; -import { runs } from "@agrippa/db"; +import { repoConnections, runs } from "@agrippa/db"; import { FakeExecutor, type FakeStepBehavior } from "@agrippa/executor-core"; import { type EngineDeps, @@ -43,6 +43,7 @@ describe.skipIf(!dbUp)("execution api (submit → engine → approve → artifac let admin: TestClient; let viewer: TestClient; let projectId: string; + let repoConnectionId: string; let taskTypeId: string; let taskId: string; let runId: string; @@ -85,6 +86,13 @@ describe.skipIf(!dbUp)("execution api (submit → engine → approve → artifac json: { email: "vera@example.com", role: "viewer" }, }); + // a repo connection owned by this project — submissions must reference one + const [conn] = await db + .insert(repoConnections) + .values({ projectId, provider: "github", url: "https://github.com/acme/widget.git" }) + .returning(); + repoConnectionId = conn?.id as string; + const types = await jsonOf>( await admin.request("/api/v1/scenarios/software-development/task-types"), ); @@ -96,7 +104,7 @@ describe.skipIf(!dbUp)("execution api (submit → engine → approve → artifac title: "Fix the widget", params: { bugReport: "It crashes", - repo: { repoConnectionId: Bun.randomUUIDv7() }, + repo: { repoConnectionId }, }, }); @@ -141,6 +149,33 @@ describe.skipIf(!dbUp)("execution api (submit → engine → approve → artifac expect(res.status).toBe(403); }); + it("refuses a repoConnectionId belonging to another project", async () => { + // a connection owned by a different project — cross-tenant IDOR attempt + const otherId = ( + await jsonOf<{ id: string }>( + await admin.request("/api/v1/projects", { + method: "POST", + json: { slug: "other", name: "Other" }, + }), + ) + ).id; + const [foreign] = await db + .insert(repoConnections) + .values({ projectId: otherId, provider: "github", url: "https://github.com/acme/secret.git" }) + .returning(); + + const res = await admin.request(`/api/v1/projects/${projectId}/tasks`, { + method: "POST", + json: { + taskTypeId, + title: "IDOR", + params: { bugReport: "It crashes", repo: { repoConnectionId: foreign?.id } }, + }, + }); + expect(res.status).toBe(400); + expect((await jsonOf<{ code: string }>(res)).code).toBe("repo_not_in_project"); + }); + it("accepts a valid submission and enqueues the run", async () => { const res = await admin.request(`/api/v1/projects/${projectId}/tasks`, { method: "POST", diff --git a/apps/worker/src/deps/workspace.ts b/apps/worker/src/deps/workspace.ts index e73590d..548287c 100644 --- a/apps/worker/src/deps/workspace.ts +++ b/apps/worker/src/deps/workspace.ts @@ -3,7 +3,7 @@ import { tmpdir } from "node:os"; import path from "node:path"; import { type Db, decryptSecret, loadSecretKey, repoConnections, secrets } from "@agrippa/db"; import type { WorkspaceManager, WorkspaceSpec } from "@agrippa/orchestration"; -import { eq } from "drizzle-orm"; +import { and, eq } from "drizzle-orm"; const WORKSPACE_ROOT = process.env.WORKSPACE_ROOT ?? path.join(tmpdir(), "agrippa-workspaces"); @@ -61,10 +61,17 @@ export class GitWorkspaceManager implements WorkspaceManager { const repoRef = spec.repo as { repoConnectionId?: string } | null; if (!repoRef?.repoConnectionId) throw new Error("workspace.checkout: repoRef missing"); + // scope by project so a run can never clone another project's/tenant's repo + // even if its params carry a foreign repoConnectionId const [connection] = await this.db .select() .from(repoConnections) - .where(eq(repoConnections.id, repoRef.repoConnectionId)); + .where( + and( + eq(repoConnections.id, repoRef.repoConnectionId), + eq(repoConnections.projectId, spec.projectId), + ), + ); if (!connection) throw new Error("workspace.checkout: repo connection not found"); let cloneUrl = connection.url; diff --git a/packages/db/drizzle/0002_run-resource-manifest.sql b/packages/db/drizzle/0002_run-resource-manifest.sql new file mode 100644 index 0000000..e277ca4 --- /dev/null +++ b/packages/db/drizzle/0002_run-resource-manifest.sql @@ -0,0 +1 @@ +ALTER TABLE "runs" ADD COLUMN "resource_manifest" jsonb DEFAULT '{"mcpServers":[],"skills":[]}'::jsonb NOT NULL; \ No newline at end of file diff --git a/packages/db/drizzle/meta/0002_snapshot.json b/packages/db/drizzle/meta/0002_snapshot.json new file mode 100644 index 0000000..20444c5 --- /dev/null +++ b/packages/db/drizzle/meta/0002_snapshot.json @@ -0,0 +1,3068 @@ +{ + "id": "00a6f090-e151-4e07-a855-f65a5107ff60", + "prevId": "bda22417-cff0-4e34-b64c-779af5f45760", + "version": "7", + "dialect": "postgresql", + "tables": { + "public.api_keys": { + "name": "api_keys", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "name": { + "name": "name", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "key_hash": { + "name": "key_hash", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "prefix": { + "name": "prefix", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "scopes": { + "name": "scopes", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'[]'::jsonb" + }, + "created_by": { + "name": "created_by", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "expires_at": { + "name": "expires_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "revoked_at": { + "name": "revoked_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "last_used_at": { + "name": "last_used_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + } + }, + "indexes": {}, + "foreignKeys": { + "api_keys_org_id_orgs_id_fk": { + "name": "api_keys_org_id_orgs_id_fk", + "tableFrom": "api_keys", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "api_keys_project_id_projects_id_fk": { + "name": "api_keys_project_id_projects_id_fk", + "tableFrom": "api_keys", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "api_keys_created_by_users_id_fk": { + "name": "api_keys_created_by_users_id_fk", + "tableFrom": "api_keys", + "tableTo": "users", + "columnsFrom": ["created_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.audit_logs": { + "name": "audit_logs", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "actor_user_id": { + "name": "actor_user_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "actor_api_key_id": { + "name": "actor_api_key_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "action": { + "name": "action", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "resource_type": { + "name": "resource_type", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "resource_id": { + "name": "resource_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "payload": { + "name": "payload", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "ip": { + "name": "ip", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "audit_logs_org_time_idx": { + "name": "audit_logs_org_time_idx", + "columns": [ + { + "expression": "org_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "created_at", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "audit_logs_org_id_orgs_id_fk": { + "name": "audit_logs_org_id_orgs_id_fk", + "tableFrom": "audit_logs", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "audit_logs_project_id_projects_id_fk": { + "name": "audit_logs_project_id_projects_id_fk", + "tableFrom": "audit_logs", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "audit_logs_actor_user_id_users_id_fk": { + "name": "audit_logs_actor_user_id_users_id_fk", + "tableFrom": "audit_logs", + "tableTo": "users", + "columnsFrom": ["actor_user_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "audit_logs_actor_api_key_id_api_keys_id_fk": { + "name": "audit_logs_actor_api_key_id_api_keys_id_fk", + "tableFrom": "audit_logs", + "tableTo": "api_keys", + "columnsFrom": ["actor_api_key_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.accounts": { + "name": "accounts", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "account_id": { + "name": "account_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "provider_id": { + "name": "provider_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "access_token": { + "name": "access_token", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "refresh_token": { + "name": "refresh_token", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "id_token": { + "name": "id_token", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "access_token_expires_at": { + "name": "access_token_expires_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "refresh_token_expires_at": { + "name": "refresh_token_expires_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "scope": { + "name": "scope", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "password": { + "name": "password", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "accounts_user_id_users_id_fk": { + "name": "accounts_user_id_users_id_fk", + "tableFrom": "accounts", + "tableTo": "users", + "columnsFrom": ["user_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.sessions": { + "name": "sessions", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "token": { + "name": "token", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "expires_at": { + "name": "expires_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true + }, + "ip_address": { + "name": "ip_address", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "user_agent": { + "name": "user_agent", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "sessions_user_id_users_id_fk": { + "name": "sessions_user_id_users_id_fk", + "tableFrom": "sessions", + "tableTo": "users", + "columnsFrom": ["user_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "sessions_token_unique": { + "name": "sessions_token_unique", + "nullsNotDistinct": false, + "columns": ["token"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.users": { + "name": "users", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "name": { + "name": "name", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "email": { + "name": "email", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "email_verified": { + "name": "email_verified", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": false + }, + "image": { + "name": "image", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "locale": { + "name": "locale", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'en'" + }, + "org_role": { + "name": "org_role", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'org_member'" + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "users_org_id_orgs_id_fk": { + "name": "users_org_id_orgs_id_fk", + "tableFrom": "users", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "users_email_unique": { + "name": "users_email_unique", + "nullsNotDistinct": false, + "columns": ["email"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.verifications": { + "name": "verifications", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "identifier": { + "name": "identifier", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "value": { + "name": "value", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "expires_at": { + "name": "expires_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.orgs": { + "name": "orgs", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "slug": { + "name": "slug", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "name": { + "name": "name", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "orgs_slug_unique": { + "name": "orgs_slug_unique", + "nullsNotDistinct": false, + "columns": ["slug"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.project_members": { + "name": "project_members", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "role": { + "name": "role", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "project_members_uq": { + "name": "project_members_uq", + "columns": [ + { + "expression": "project_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "user_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "project_members_project_id_projects_id_fk": { + "name": "project_members_project_id_projects_id_fk", + "tableFrom": "project_members", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "project_members_user_id_users_id_fk": { + "name": "project_members_user_id_users_id_fk", + "tableFrom": "project_members", + "tableTo": "users", + "columnsFrom": ["user_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.project_quotas": { + "name": "project_quotas", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "period": { + "name": "period", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'monthly'" + }, + "token_limit": { + "name": "token_limit", + "type": "bigint", + "primaryKey": false, + "notNull": false + }, + "cost_limit_usd": { + "name": "cost_limit_usd", + "type": "numeric(12, 2)", + "primaryKey": false, + "notNull": false + }, + "hard_stop": { + "name": "hard_stop", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": true + }, + "current_period_start": { + "name": "current_period_start", + "type": "date", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "project_quotas_project_id_projects_id_fk": { + "name": "project_quotas_project_id_projects_id_fk", + "tableFrom": "project_quotas", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "project_quotas_project_id_unique": { + "name": "project_quotas_project_id_unique", + "nullsNotDistinct": false, + "columns": ["project_id"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.project_resource_grants": { + "name": "project_resource_grants", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "resource_type": { + "name": "resource_type", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "resource_id": { + "name": "resource_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "config_override": { + "name": "config_override", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "granted_by": { + "name": "granted_by", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "project_grants_uq": { + "name": "project_grants_uq", + "columns": [ + { + "expression": "project_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "resource_type", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "resource_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "project_resource_grants_project_id_projects_id_fk": { + "name": "project_resource_grants_project_id_projects_id_fk", + "tableFrom": "project_resource_grants", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "project_resource_grants_granted_by_users_id_fk": { + "name": "project_resource_grants_granted_by_users_id_fk", + "tableFrom": "project_resource_grants", + "tableTo": "users", + "columnsFrom": ["granted_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.projects": { + "name": "projects", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "slug": { + "name": "slug", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "name": { + "name": "name", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "description": { + "name": "description", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'active'" + }, + "settings": { + "name": "settings", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "created_by": { + "name": "created_by", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "archived_at": { + "name": "archived_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + } + }, + "indexes": { + "projects_org_slug_uq": { + "name": "projects_org_slug_uq", + "columns": [ + { + "expression": "org_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "slug", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "projects_org_id_orgs_id_fk": { + "name": "projects_org_id_orgs_id_fk", + "tableFrom": "projects", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "projects_created_by_users_id_fk": { + "name": "projects_created_by_users_id_fk", + "tableFrom": "projects", + "tableTo": "users", + "columnsFrom": ["created_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.repo_connections": { + "name": "repo_connections", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "provider": { + "name": "provider", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "url": { + "name": "url", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "default_branch": { + "name": "default_branch", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'main'" + }, + "credential_secret_ref": { + "name": "credential_secret_ref", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'active'" + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "repo_connections_project_id_projects_id_fk": { + "name": "repo_connections_project_id_projects_id_fk", + "tableFrom": "repo_connections", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "repo_connections_credential_secret_ref_secrets_id_fk": { + "name": "repo_connections_credential_secret_ref_secrets_id_fk", + "tableFrom": "repo_connections", + "tableTo": "secrets", + "columnsFrom": ["credential_secret_ref"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.fabri": { + "name": "fabri", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "slug": { + "name": "slug", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "name_i18n": { + "name": "name_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "persona_i18n": { + "name": "persona_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "system_prompt": { + "name": "system_prompt", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "avatar": { + "name": "avatar", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "default_model_role_policy": { + "name": "default_model_role_policy", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'active'" + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "fabri_org_id_orgs_id_fk": { + "name": "fabri_org_id_orgs_id_fk", + "tableFrom": "fabri", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "fabri_slug_unique": { + "name": "fabri_slug_unique", + "nullsNotDistinct": false, + "columns": ["slug"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.mcp_servers": { + "name": "mcp_servers", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "slug": { + "name": "slug", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "name_i18n": { + "name": "name_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "transport": { + "name": "transport", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "config": { + "name": "config", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "auth_secret_ref": { + "name": "auth_secret_ref", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "config_revision": { + "name": "config_revision", + "type": "integer", + "primaryKey": false, + "notNull": true, + "default": 1 + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'active'" + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "mcp_servers_org_id_orgs_id_fk": { + "name": "mcp_servers_org_id_orgs_id_fk", + "tableFrom": "mcp_servers", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "mcp_servers_auth_secret_ref_secrets_id_fk": { + "name": "mcp_servers_auth_secret_ref_secrets_id_fk", + "tableFrom": "mcp_servers", + "tableTo": "secrets", + "columnsFrom": ["auth_secret_ref"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "mcp_servers_slug_unique": { + "name": "mcp_servers_slug_unique", + "nullsNotDistinct": false, + "columns": ["slug"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.models": { + "name": "models", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "provider": { + "name": "provider", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "provider_model_id": { + "name": "provider_model_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "display_name": { + "name": "display_name", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "tier": { + "name": "tier", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "capabilities": { + "name": "capabilities", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "context_window": { + "name": "context_window", + "type": "integer", + "primaryKey": false, + "notNull": false + }, + "input_cost_per_mtok": { + "name": "input_cost_per_mtok", + "type": "numeric(12, 4)", + "primaryKey": false, + "notNull": false + }, + "output_cost_per_mtok": { + "name": "output_cost_per_mtok", + "type": "numeric(12, 4)", + "primaryKey": false, + "notNull": false + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'active'" + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "models_org_id_orgs_id_fk": { + "name": "models_org_id_orgs_id_fk", + "tableFrom": "models", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "models_provider_model_id_unique": { + "name": "models_provider_model_id_unique", + "nullsNotDistinct": false, + "columns": ["provider_model_id"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.orchestration_templates": { + "name": "orchestration_templates", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "slug": { + "name": "slug", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "scenario_id": { + "name": "scenario_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "name_i18n": { + "name": "name_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "latest_published_version_id": { + "name": "latest_published_version_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "created_by": { + "name": "created_by", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "orchestration_templates_org_id_orgs_id_fk": { + "name": "orchestration_templates_org_id_orgs_id_fk", + "tableFrom": "orchestration_templates", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "orchestration_templates_scenario_id_scenarios_id_fk": { + "name": "orchestration_templates_scenario_id_scenarios_id_fk", + "tableFrom": "orchestration_templates", + "tableTo": "scenarios", + "columnsFrom": ["scenario_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "orchestration_templates_latest_published_version_id_template_versions_id_fk": { + "name": "orchestration_templates_latest_published_version_id_template_versions_id_fk", + "tableFrom": "orchestration_templates", + "tableTo": "template_versions", + "columnsFrom": ["latest_published_version_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "orchestration_templates_created_by_users_id_fk": { + "name": "orchestration_templates_created_by_users_id_fk", + "tableFrom": "orchestration_templates", + "tableTo": "users", + "columnsFrom": ["created_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "orchestration_templates_slug_unique": { + "name": "orchestration_templates_slug_unique", + "nullsNotDistinct": false, + "columns": ["slug"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.scenarios": { + "name": "scenarios", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "slug": { + "name": "slug", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "name_i18n": { + "name": "name_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "description_i18n": { + "name": "description_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "icon": { + "name": "icon", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "sort_order": { + "name": "sort_order", + "type": "integer", + "primaryKey": false, + "notNull": true, + "default": 0 + }, + "enabled": { + "name": "enabled", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "scenarios_org_id_orgs_id_fk": { + "name": "scenarios_org_id_orgs_id_fk", + "tableFrom": "scenarios", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "scenarios_slug_unique": { + "name": "scenarios_slug_unique", + "nullsNotDistinct": false, + "columns": ["slug"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.skill_versions": { + "name": "skill_versions", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "skill_id": { + "name": "skill_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "version": { + "name": "version", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "content_ref": { + "name": "content_ref", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "manifest": { + "name": "manifest", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'active'" + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "skill_versions_uq": { + "name": "skill_versions_uq", + "columns": [ + { + "expression": "skill_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "version", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "skill_versions_skill_id_skills_id_fk": { + "name": "skill_versions_skill_id_skills_id_fk", + "tableFrom": "skill_versions", + "tableTo": "skills", + "columnsFrom": ["skill_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.skills": { + "name": "skills", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "slug": { + "name": "slug", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "name_i18n": { + "name": "name_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "description_i18n": { + "name": "description_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "source": { + "name": "source", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "latest_version_id": { + "name": "latest_version_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "skills_org_id_orgs_id_fk": { + "name": "skills_org_id_orgs_id_fk", + "tableFrom": "skills", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "skills_latest_version_id_skill_versions_id_fk": { + "name": "skills_latest_version_id_skill_versions_id_fk", + "tableFrom": "skills", + "tableTo": "skill_versions", + "columnsFrom": ["latest_version_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "skills_slug_unique": { + "name": "skills_slug_unique", + "nullsNotDistinct": false, + "columns": ["slug"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.task_types": { + "name": "task_types", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "scenario_id": { + "name": "scenario_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "slug": { + "name": "slug", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "name_i18n": { + "name": "name_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "description_i18n": { + "name": "description_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "template_id": { + "name": "template_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "default_faber_id": { + "name": "default_faber_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "enabled": { + "name": "enabled", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": true + }, + "sort_order": { + "name": "sort_order", + "type": "integer", + "primaryKey": false, + "notNull": true, + "default": 0 + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "task_types_uq": { + "name": "task_types_uq", + "columns": [ + { + "expression": "scenario_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "slug", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "task_types_scenario_id_scenarios_id_fk": { + "name": "task_types_scenario_id_scenarios_id_fk", + "tableFrom": "task_types", + "tableTo": "scenarios", + "columnsFrom": ["scenario_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "task_types_template_id_orchestration_templates_id_fk": { + "name": "task_types_template_id_orchestration_templates_id_fk", + "tableFrom": "task_types", + "tableTo": "orchestration_templates", + "columnsFrom": ["template_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "task_types_default_faber_id_fabri_id_fk": { + "name": "task_types_default_faber_id_fabri_id_fk", + "tableFrom": "task_types", + "tableTo": "fabri", + "columnsFrom": ["default_faber_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.template_versions": { + "name": "template_versions", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "template_id": { + "name": "template_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "version": { + "name": "version", + "type": "integer", + "primaryKey": false, + "notNull": true + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'draft'" + }, + "source_yaml": { + "name": "source_yaml", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "compiled": { + "name": "compiled", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "checksum": { + "name": "checksum", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "created_by": { + "name": "created_by", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "published_at": { + "name": "published_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + } + }, + "indexes": { + "template_versions_uq": { + "name": "template_versions_uq", + "columns": [ + { + "expression": "template_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "version", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "template_versions_template_id_orchestration_templates_id_fk": { + "name": "template_versions_template_id_orchestration_templates_id_fk", + "tableFrom": "template_versions", + "tableTo": "orchestration_templates", + "columnsFrom": ["template_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "template_versions_created_by_users_id_fk": { + "name": "template_versions_created_by_users_id_fk", + "tableFrom": "template_versions", + "tableTo": "users", + "columnsFrom": ["created_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.approvals": { + "name": "approvals", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "run_id": { + "name": "run_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "step_id": { + "name": "step_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "checkpoint_id": { + "name": "checkpoint_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'pending'" + }, + "payload": { + "name": "payload", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "requested_at": { + "name": "requested_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "decided_by": { + "name": "decided_by", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "decided_at": { + "name": "decided_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "comment": { + "name": "comment", + "type": "text", + "primaryKey": false, + "notNull": false + } + }, + "indexes": {}, + "foreignKeys": { + "approvals_run_id_runs_id_fk": { + "name": "approvals_run_id_runs_id_fk", + "tableFrom": "approvals", + "tableTo": "runs", + "columnsFrom": ["run_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "approvals_step_id_run_steps_id_fk": { + "name": "approvals_step_id_run_steps_id_fk", + "tableFrom": "approvals", + "tableTo": "run_steps", + "columnsFrom": ["step_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "approvals_decided_by_users_id_fk": { + "name": "approvals_decided_by_users_id_fk", + "tableFrom": "approvals", + "tableTo": "users", + "columnsFrom": ["decided_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.artifacts": { + "name": "artifacts", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "run_id": { + "name": "run_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "step_id": { + "name": "step_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "artifact_key": { + "name": "artifact_key", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "kind": { + "name": "kind", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "name": { + "name": "name", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "mime": { + "name": "mime", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "size": { + "name": "size", + "type": "integer", + "primaryKey": false, + "notNull": false + }, + "storage_ref": { + "name": "storage_ref", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "inline": { + "name": "inline", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "artifacts_run_id_runs_id_fk": { + "name": "artifacts_run_id_runs_id_fk", + "tableFrom": "artifacts", + "tableTo": "runs", + "columnsFrom": ["run_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "artifacts_step_id_run_steps_id_fk": { + "name": "artifacts_step_id_run_steps_id_fk", + "tableFrom": "artifacts", + "tableTo": "run_steps", + "columnsFrom": ["step_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.run_events": { + "name": "run_events", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "bigserial", + "primaryKey": true, + "notNull": true + }, + "run_id": { + "name": "run_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "step_id": { + "name": "step_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "seq": { + "name": "seq", + "type": "integer", + "primaryKey": false, + "notNull": true + }, + "type": { + "name": "type", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "payload": { + "name": "payload", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "run_events_run_seq_uq": { + "name": "run_events_run_seq_uq", + "columns": [ + { + "expression": "run_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "seq", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "run_events_run_id_runs_id_fk": { + "name": "run_events_run_id_runs_id_fk", + "tableFrom": "run_events", + "tableTo": "runs", + "columnsFrom": ["run_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "run_events_step_id_run_steps_id_fk": { + "name": "run_events_step_id_run_steps_id_fk", + "tableFrom": "run_events", + "tableTo": "run_steps", + "columnsFrom": ["step_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.run_steps": { + "name": "run_steps", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "run_id": { + "name": "run_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "phase_id": { + "name": "phase_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "step_id": { + "name": "step_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "attempt": { + "name": "attempt", + "type": "integer", + "primaryKey": false, + "notNull": true, + "default": 1 + }, + "seq": { + "name": "seq", + "type": "integer", + "primaryKey": false, + "notNull": true + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'pending'" + }, + "agent_ref": { + "name": "agent_ref", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "model_id": { + "name": "model_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "executor_session_id": { + "name": "executor_session_id", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "output": { + "name": "output", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "usage": { + "name": "usage", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "error": { + "name": "error", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "started_at": { + "name": "started_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "finished_at": { + "name": "finished_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + } + }, + "indexes": { + "run_steps_uq": { + "name": "run_steps_uq", + "columns": [ + { + "expression": "run_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "phase_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "step_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "attempt", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "run_steps_run_id_runs_id_fk": { + "name": "run_steps_run_id_runs_id_fk", + "tableFrom": "run_steps", + "tableTo": "runs", + "columnsFrom": ["run_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.runs": { + "name": "runs", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "task_id": { + "name": "task_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "number": { + "name": "number", + "type": "integer", + "primaryKey": false, + "notNull": true + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'queued'" + }, + "template_version_id": { + "name": "template_version_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "faber_id": { + "name": "faber_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "executor_id": { + "name": "executor_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "params_snapshot": { + "name": "params_snapshot", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "model_resolution": { + "name": "model_resolution", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "resource_manifest": { + "name": "resource_manifest", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{\"mcpServers\":[],\"skills\":[]}'::jsonb" + }, + "budget": { + "name": "budget", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "usage_totals": { + "name": "usage_totals", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "workspace_ref": { + "name": "workspace_ref", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "error": { + "name": "error", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "cancel_requested": { + "name": "cancel_requested", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": false + }, + "queued_at": { + "name": "queued_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "started_at": { + "name": "started_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "finished_at": { + "name": "finished_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "created_by": { + "name": "created_by", + "type": "uuid", + "primaryKey": false, + "notNull": true + } + }, + "indexes": { + "runs_task_number_uq": { + "name": "runs_task_number_uq", + "columns": [ + { + "expression": "task_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "number", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + }, + "runs_project_idx": { + "name": "runs_project_idx", + "columns": [ + { + "expression": "project_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "status", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "runs_task_id_tasks_id_fk": { + "name": "runs_task_id_tasks_id_fk", + "tableFrom": "runs", + "tableTo": "tasks", + "columnsFrom": ["task_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "runs_project_id_projects_id_fk": { + "name": "runs_project_id_projects_id_fk", + "tableFrom": "runs", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "runs_template_version_id_template_versions_id_fk": { + "name": "runs_template_version_id_template_versions_id_fk", + "tableFrom": "runs", + "tableTo": "template_versions", + "columnsFrom": ["template_version_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "runs_faber_id_fabri_id_fk": { + "name": "runs_faber_id_fabri_id_fk", + "tableFrom": "runs", + "tableTo": "fabri", + "columnsFrom": ["faber_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "runs_created_by_users_id_fk": { + "name": "runs_created_by_users_id_fk", + "tableFrom": "runs", + "tableTo": "users", + "columnsFrom": ["created_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.tasks": { + "name": "tasks", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "task_type_id": { + "name": "task_type_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "title": { + "name": "title", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "params": { + "name": "params", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "latest_run_id": { + "name": "latest_run_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "created_by": { + "name": "created_by", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "tasks_org_id_orgs_id_fk": { + "name": "tasks_org_id_orgs_id_fk", + "tableFrom": "tasks", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "tasks_project_id_projects_id_fk": { + "name": "tasks_project_id_projects_id_fk", + "tableFrom": "tasks", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "tasks_task_type_id_task_types_id_fk": { + "name": "tasks_task_type_id_task_types_id_fk", + "tableFrom": "tasks", + "tableTo": "task_types", + "columnsFrom": ["task_type_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "tasks_latest_run_id_runs_id_fk": { + "name": "tasks_latest_run_id_runs_id_fk", + "tableFrom": "tasks", + "tableTo": "runs", + "columnsFrom": ["latest_run_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "tasks_created_by_users_id_fk": { + "name": "tasks_created_by_users_id_fk", + "tableFrom": "tasks", + "tableTo": "users", + "columnsFrom": ["created_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.secrets": { + "name": "secrets", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "kind": { + "name": "kind", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "ciphertext": { + "name": "ciphertext", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "created_by": { + "name": "created_by", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "rotated_at": { + "name": "rotated_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + } + }, + "indexes": {}, + "foreignKeys": { + "secrets_org_id_orgs_id_fk": { + "name": "secrets_org_id_orgs_id_fk", + "tableFrom": "secrets", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "secrets_created_by_users_id_fk": { + "name": "secrets_created_by_users_id_fk", + "tableFrom": "secrets", + "tableTo": "users", + "columnsFrom": ["created_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.token_usage": { + "name": "token_usage", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "run_id": { + "name": "run_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "step_id": { + "name": "step_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "attempt": { + "name": "attempt", + "type": "integer", + "primaryKey": false, + "notNull": true, + "default": 1 + }, + "model_id": { + "name": "model_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "input_tokens": { + "name": "input_tokens", + "type": "bigint", + "primaryKey": false, + "notNull": true, + "default": 0 + }, + "output_tokens": { + "name": "output_tokens", + "type": "bigint", + "primaryKey": false, + "notNull": true, + "default": 0 + }, + "cache_read_tokens": { + "name": "cache_read_tokens", + "type": "bigint", + "primaryKey": false, + "notNull": true, + "default": 0 + }, + "cache_write_tokens": { + "name": "cache_write_tokens", + "type": "bigint", + "primaryKey": false, + "notNull": true, + "default": 0 + }, + "cost_usd": { + "name": "cost_usd", + "type": "numeric(12, 6)", + "primaryKey": false, + "notNull": true, + "default": "'0'" + }, + "occurred_at": { + "name": "occurred_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "token_usage_project_time_idx": { + "name": "token_usage_project_time_idx", + "columns": [ + { + "expression": "project_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "occurred_at", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "token_usage_run_idx": { + "name": "token_usage_run_idx", + "columns": [ + { + "expression": "run_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "token_usage_org_id_orgs_id_fk": { + "name": "token_usage_org_id_orgs_id_fk", + "tableFrom": "token_usage", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "token_usage_project_id_projects_id_fk": { + "name": "token_usage_project_id_projects_id_fk", + "tableFrom": "token_usage", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "token_usage_run_id_runs_id_fk": { + "name": "token_usage_run_id_runs_id_fk", + "tableFrom": "token_usage", + "tableTo": "runs", + "columnsFrom": ["run_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "token_usage_step_id_run_steps_id_fk": { + "name": "token_usage_step_id_run_steps_id_fk", + "tableFrom": "token_usage", + "tableTo": "run_steps", + "columnsFrom": ["step_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "token_usage_model_id_models_id_fk": { + "name": "token_usage_model_id_models_id_fk", + "tableFrom": "token_usage", + "tableTo": "models", + "columnsFrom": ["model_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + } + }, + "enums": {}, + "schemas": {}, + "sequences": {}, + "roles": {}, + "policies": {}, + "views": {}, + "_meta": { + "columns": {}, + "schemas": {}, + "tables": {} + } +} diff --git a/packages/db/drizzle/meta/_journal.json b/packages/db/drizzle/meta/_journal.json index bb8579f..a0d542a 100644 --- a/packages/db/drizzle/meta/_journal.json +++ b/packages/db/drizzle/meta/_journal.json @@ -15,6 +15,13 @@ "when": 1784261124504, "tag": "0001_run-step-output", "breakpoints": true + }, + { + "idx": 2, + "version": "7", + "when": 1784274345055, + "tag": "0002_run-resource-manifest", + "breakpoints": true } ] } diff --git a/packages/db/src/schema/runs.ts b/packages/db/src/schema/runs.ts index bf778c2..b915e76 100644 --- a/packages/db/src/schema/runs.ts +++ b/packages/db/src/schema/runs.ts @@ -58,6 +58,12 @@ export const runs = pgTable( executorId: text("executor_id").notNull(), paramsSnapshot: jsonb("params_snapshot").$type>().notNull(), modelResolution: jsonb("model_resolution").$type>().notNull(), + // authorized skills/MCP slugs, pinned at submit from project grants; the + // worker resolves resources only from this set (docs/design/04, ADR-0005) + resourceManifest: jsonb("resource_manifest") + .$type<{ mcpServers: string[]; skills: string[] }>() + .notNull() + .default({ mcpServers: [], skills: [] }), budget: jsonb("budget").$type>().notNull().default({}), usageTotals: jsonb("usage_totals").$type>().notNull().default({}), workspaceRef: text("workspace_ref"), diff --git a/packages/orchestration/src/engine/deps.ts b/packages/orchestration/src/engine/deps.ts index 52b9333..5067dee 100644 --- a/packages/orchestration/src/engine/deps.ts +++ b/packages/orchestration/src/engine/deps.ts @@ -13,6 +13,8 @@ export type WorkspaceSpec = { repo: unknown; ref?: string; access: "readOnly" | "readWrite"; + /** The run's project — repo connections are loaded scoped to it, never by raw id. */ + projectId: string; }; export interface WorkspaceManager { diff --git a/packages/orchestration/src/engine/engine.integration.test.ts b/packages/orchestration/src/engine/engine.integration.test.ts index c8c92f2..1018c6f 100644 --- a/packages/orchestration/src/engine/engine.integration.test.ts +++ b/packages/orchestration/src/engine/engine.integration.test.ts @@ -68,6 +68,8 @@ type DepsOptions = { mcpServers?: string[] }; type FixtureOptions = { params?: Record; quota?: { costLimitUsd?: number; tokenLimit?: number }; + /** Override the run's authorized-resource manifest (default: all template resources). */ + resourceManifest?: { mcpServers: string[]; skills: string[] }; }; async function setupFixture(options: FixtureOptions = {}): Promise { @@ -166,6 +168,10 @@ async function setupFixture(options: FixtureOptions = {}): Promise { executorId: "fake", paramsSnapshot: params, modelResolution, + resourceManifest: options.resourceManifest ?? { + mcpServers: template.spec.resources.mcpServers.map((m) => m.ref), + skills: template.spec.resources.skills.map((s) => s.ref.split("@")[0] as string), + }, budget: template.spec.budgets as unknown as Record, createdBy: user.id, }) @@ -505,6 +511,27 @@ describe.skipIf(!dbUp)("orchestration engine (FakeExecutor compliance suite)", ( expect(rows[0]?.status).toBe("skipped"); }); + it("skips open-pr when the optional MCP server is not authorized, even if it exists", async () => { + // manifest omits github: the project has no grant. The server is otherwise + // available (materializer has it), but an ungranted optional resource must + // never be resolved — else the run would receive the global GitHub token. + const { db, runId, makeDeps } = await setupFixture({ + params: { autoOpenPr: true }, + resourceManifest: { mcpServers: [], skills: [] }, + }); + await executeRun(makeDeps(HAPPY_SCRIPT, { mcpServers: ["github"] }), runId); + await approve(db, runId); + const deps = makeDeps(HAPPY_SCRIPT, { mcpServers: ["github"] }); + expect(await executeRun(deps, runId)).toBe("succeeded"); + const rows = await db + .select() + .from(runSteps) + .where(and(eq(runSteps.runId, runId), eq(runSteps.stepId, "open-pr"))); + expect(rows[0]?.status).toBe("skipped"); + // the executor never saw the github server + expect(deps.executor.requests.some((r) => r.stepId === "open-pr")).toBe(false); + }); + it("streams live events over the bus while executing", async () => { const { runId, makeDeps, bus } = await setupFixture(); const seen: string[] = []; diff --git a/packages/orchestration/src/engine/engine.ts b/packages/orchestration/src/engine/engine.ts index d35f71c..864b85a 100644 --- a/packages/orchestration/src/engine/engine.ts +++ b/packages/orchestration/src/engine/engine.ts @@ -368,9 +368,16 @@ class RunEngine { return; } if (step.requires) { - const { missing } = await this.deps.resources.mcpServers(step.requires.mcpServers); - if (missing.length > 0) { - await this.markSkipped(phase, step, `missing optional resources: ${missing.join(", ")}`); + const authorized = this.authorizedMcpRefs(step.requires.mcpServers); + const ungranted = step.requires.mcpServers.filter((ref) => !authorized.includes(ref)); + const { missing } = await this.deps.resources.mcpServers(authorized); + const unavailable = [...ungranted, ...missing]; + if (unavailable.length > 0) { + await this.markSkipped( + phase, + step, + `missing optional resources: ${unavailable.join(", ")}`, + ); return; } } @@ -424,7 +431,12 @@ class RunEngine { ? evaluateExpression(wrapped[1] as string, this.expressionContext()) : spec.repo; const ref = spec.ref ? interpolate(spec.ref, this.expressionContext()) : undefined; - await this.deps.workspace.checkout(this.run.id, { repo, ref, access: spec.access }); + await this.deps.workspace.checkout(this.run.id, { + repo, + ref, + access: spec.access, + projectId: this.run.projectId, + }); await this.emit("workspace.ready", { ref }); } } @@ -522,8 +534,15 @@ class RunEngine { model: modelFor(s.model.role), })); - const skills = await this.deps.resources.skills(step.skills, this.workspaceDir); - const { resolved: mcpServers, missing } = await this.deps.resources.mcpServers(step.mcpServers); + // resolve only what the run is authorized for — ungranted optional + // resources are dropped here, never resolved from the global registry + const skills = await this.deps.resources.skills( + this.authorizedSkillRefs(step.skills), + this.workspaceDir, + ); + const { resolved: mcpServers, missing } = await this.deps.resources.mcpServers( + this.authorizedMcpRefs(step.mcpServers), + ); const optionalRefs = new Set( this.template.spec.resources.mcpServers.filter((m) => m.optional).map((m) => m.ref), ); @@ -569,6 +588,18 @@ class RunEngine { }; } + /** MCP refs the run is authorized to use (pinned at submit; see resolve.authorizeResources). */ + private authorizedMcpRefs(refs: string[]): string[] { + const allowed = new Set(this.run.resourceManifest.mcpServers); + return refs.filter((ref) => allowed.has(ref)); + } + + /** Skill refs whose slug the run is authorized to use. */ + private authorizedSkillRefs(refs: string[]): string[] { + const allowed = new Set(this.run.resourceManifest.skills); + return refs.filter((ref) => allowed.has(ref.split("@")[0] as string)); + } + private expressionContext(): Record { return { inputs: this.run.paramsSnapshot, diff --git a/packages/orchestration/src/resolve.ts b/packages/orchestration/src/resolve.ts index 6fd3786..4e2687a 100644 --- a/packages/orchestration/src/resolve.ts +++ b/packages/orchestration/src/resolve.ts @@ -1,5 +1,5 @@ import type { ModelTier } from "@agrippa/core"; -import { type Db, models, projectResourceGrants } from "@agrippa/db"; +import { type Db, models, projectResourceGrants, repoConnections } from "@agrippa/db"; import { and, eq, inArray } from "drizzle-orm"; import { z } from "zod"; import type { TemplateDoc, TemplateInput } from "./template-schema"; @@ -132,8 +132,21 @@ export async function resolveModelRoles( return resolution; } -/** Required (non-optional) skills and MCP servers must be granted to the project. */ -export async function verifyResourceGrants( +/** The set of resource slugs a run is authorized to use, pinned at submit. */ +export type ResourceManifest = { mcpServers: string[]; skills: string[] }; + +/** + * Verify required skill/MCP grants and pin the authorized set. + * + * Required resources must be granted (else the submit fails). Optional + * resources are *included only when granted* — previously they skipped the + * grant check entirely and the worker then resolved them from the global + * registry with the platform's global credential, so a project with no grant + * still received (for example) the shared GitHub token. The returned manifest + * is frozen onto the run; the worker resolves resources only from it, never by + * re-reading the mutable global registry. + */ +export async function authorizeResources( db: Db, projectId: string, compiled: TemplateDoc, @@ -141,7 +154,7 @@ export async function verifyResourceGrants( skillIdBySlug: Map; mcpIdBySlug: Map; }, -): Promise { +): Promise { const grants = await db .select() .from(projectResourceGrants) @@ -153,25 +166,67 @@ export async function verifyResourceGrants( grantedByType.set(grant.resourceType, set); } + const manifest: ResourceManifest = { mcpServers: [], skills: [] }; + for (const skill of compiled.spec.resources.skills) { - if (skill.optional) continue; const slug = skill.ref.split("@")[0] as string; const id = registry.skillIdBySlug.get(slug); - if (!id) throw new SubmitError("skill_unregistered", `Skill '${slug}' is not registered`); - if (!grantedByType.get("skill")?.has(id)) { - throw new SubmitError("skill_not_granted", `Skill '${slug}' is not granted to this project`); + const granted = id !== undefined && (grantedByType.get("skill")?.has(id) ?? false); + if (!skill.optional) { + if (!id) throw new SubmitError("skill_unregistered", `Skill '${slug}' is not registered`); + if (!granted) { + throw new SubmitError( + "skill_not_granted", + `Skill '${slug}' is not granted to this project`, + ); + } } + if (granted) manifest.skills.push(slug); } for (const server of compiled.spec.resources.mcpServers) { - if (server.optional) continue; const id = registry.mcpIdBySlug.get(server.ref); - if (!id) { - throw new SubmitError("mcp_unregistered", `MCP server '${server.ref}' is not registered`); + const granted = id !== undefined && (grantedByType.get("mcp_server")?.has(id) ?? false); + if (!server.optional) { + if (!id) { + throw new SubmitError("mcp_unregistered", `MCP server '${server.ref}' is not registered`); + } + if (!granted) { + throw new SubmitError( + "mcp_not_granted", + `MCP server '${server.ref}' is not granted to this project`, + ); + } } - if (!grantedByType.get("mcp_server")?.has(id)) { + if (granted) manifest.mcpServers.push(server.ref); + } + return manifest; +} + +/** + * Validate that every repoRef param points at a repo connection owned by the + * submitting project. Without this a member could submit another project's (or + * tenant's) `repoConnectionId` and the worker would clone that repo with its + * stored credential — a cross-tenant authorization bypass. + */ +export async function verifyRepoRefs( + db: Db, + projectId: string, + inputs: TemplateInput[], + params: Record, +): Promise { + const repoInputs = inputs.filter((i) => i.type === "repoRef"); + for (const input of repoInputs) { + const value = params[input.key] as { repoConnectionId?: string } | undefined; + const id = value?.repoConnectionId; + if (!id) continue; // optional/unset repoRef — nothing to authorize + const [connection] = await db + .select({ id: repoConnections.id }) + .from(repoConnections) + .where(and(eq(repoConnections.id, id), eq(repoConnections.projectId, projectId))); + if (!connection) { throw new SubmitError( - "mcp_not_granted", - `MCP server '${server.ref}' is not granted to this project`, + "repo_not_in_project", + `Repo connection '${id}' does not belong to this project`, ); } } From 173d673410f81986711627d92c10790967795c9f Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 16:02:26 +0800 Subject: [PATCH 04/18] fix(engine): atomic run transitions, DB-allocated event seq, recoverable approvals Run status changes, event-seq allocation, and approval decisions were spread across the API, engine, and worker and none were atomic: - run status was written with an id-only WHERE, so a late worker finalize could overwrite a concurrent cancellation; - event seq was max(seq)+1 seeded into an in-memory counter in both the API and the engine, so two writers could collide on the unique (run_id, seq) index; - approval decisions were check-then-update with no status='pending' guard, so a user decision and the expiry worker could overwrite each other; and the status was committed before the resume enqueue, so an enqueue failure stranded the run in waiting_approval forever (the sweeper only re-enqueued 'queued' runs). Introduce a run-lifecycle module owning all three as atomic operations: transitionRun (compare-and-swap on the expected `from`), appendRunEvent (seq allocated inside the INSERT, retried on the rare unique-index race), and decideApproval (CAS on status='pending'). The engine finalize bails when its CAS loses, so it never clobbers another terminal outcome; the API and expiry worker route approvals through decideApproval; and the sweeper now re-enqueues waiting_approval runs whose approval is already decided, recovering any lost resume enqueue. Adds lifecycle CAS/seq integration tests. --- apps/api/src/routes/execution.ts | 46 ++++----- apps/worker/src/index.ts | 28 +++--- .../src/engine/engine.integration.test.ts | 48 +++++++++ packages/orchestration/src/engine/engine.ts | 57 ++++++----- .../orchestration/src/engine/run-lifecycle.ts | 99 +++++++++++++++++++ packages/orchestration/src/index.ts | 1 + 6 files changed, 218 insertions(+), 61 deletions(-) create mode 100644 packages/orchestration/src/engine/run-lifecycle.ts diff --git a/apps/api/src/routes/execution.ts b/apps/api/src/routes/execution.ts index cedfb43..c012a3f 100644 --- a/apps/api/src/routes/execution.ts +++ b/apps/api/src/routes/execution.ts @@ -18,14 +18,16 @@ import { templateVersions, } from "@agrippa/db"; import { + appendRunEvent as allocateRunEvent, authorizeResources, buildParamsValidator, + decideApproval, resolveModelRoles, SubmitError, type TemplateDoc, verifyRepoRefs, } from "@agrippa/orchestration"; -import { and, asc, desc, eq, gt, max } from "drizzle-orm"; +import { and, asc, desc, eq, gt } from "drizzle-orm"; import { Hono } from "hono"; import { streamSSE } from "hono/streaming"; import type { AppEnv } from "../context"; @@ -47,26 +49,15 @@ async function loadRunScoped( return run; } -/** Appends an API-originated event to the run log (seq = max + 1) and the bus. */ +/** Appends an API-originated event to the run log (DB-allocated seq) and the bus. */ async function appendRunEvent( c: { var: AppEnv["Variables"] }, runId: string, type: string, payload: Record, ): Promise { - const [maxSeq] = await c.var.db - .select({ v: max(runEvents.seq) }) - .from(runEvents) - .where(eq(runEvents.runId, runId)); - const seq = (maxSeq?.v ?? 0) + 1; - await c.var.db.insert(runEvents).values({ runId, seq, type, payload }); - await c.var.bus?.publish({ - runId, - seq, - type, - payload, - createdAt: new Date().toISOString(), - }); + const { seq, createdAt } = await allocateRunEvent(c.var.db, { runId, type, payload }); + await c.var.bus?.publish({ runId, seq, type, payload, createdAt: createdAt.toISOString() }); } export const executionRoutes = new Hono() @@ -310,19 +301,20 @@ export const executionRoutes = new Hono() .from(approvals) .where(and(eq(approvals.id, c.req.param("approvalId")), eq(approvals.runId, run.id))); if (!approval) throw AppError.notFound("Approval"); - if (approval.status !== "pending") { - throw AppError.conflict("already_decided", `Approval is ${approval.status}`); + // compare-and-swap on status='pending' so a user decision and the expiry + // worker cannot overwrite each other + const updated = await decideApproval(c.var.db, approval.id, { + status: input.decision, + decidedBy: c.var.user.id, + comment: input.comment, + }); + if (!updated) { + // already decided, OR a prior attempt decided then failed to enqueue: in + // both cases the durable state is correct, so re-enqueue to unstick the + // run (the sweeper also backstops this) and report the conflict + await c.var.queue?.enqueueRun(run.id); + throw AppError.conflict("already_decided", "Approval is already decided"); } - const [updated] = await c.var.db - .update(approvals) - .set({ - status: input.decision, - decidedBy: c.var.user.id, - decidedAt: new Date(), - comment: input.comment, - }) - .where(eq(approvals.id, approval.id)) - .returning(); await appendRunEvent(c, run.id, "approval.decided", { approvalId: approval.id, checkpointId: approval.checkpointId, diff --git a/apps/worker/src/index.ts b/apps/worker/src/index.ts index 6c8582f..9c50ba6 100644 --- a/apps/worker/src/index.ts +++ b/apps/worker/src/index.ts @@ -9,13 +9,14 @@ import { approvals, createDb, runs } from "@agrippa/db"; import { createClaudeExecutor } from "@agrippa/executor-claude"; import { createRunQueue, + decideApproval, durationToMinutes, type EngineDeps, executeRun, InProcessEventBus, RedisEventBus, } from "@agrippa/orchestration"; -import { and, eq, lt, sql } from "drizzle-orm"; +import { and, eq, lt, ne, sql } from "drizzle-orm"; import type { Job, JobWithMetadata } from "pg-boss"; import { DiskArtifactStore } from "./deps/artifacts"; import { DemoExecutor } from "./deps/demo-executor"; @@ -76,16 +77,11 @@ await queue.boss.work( await queue.boss.work(QUEUE_APPROVAL_EXPIRE, async (jobs: Job[]) => { for (const job of jobs) { - const [approval] = await db - .select() - .from(approvals) - .where(eq(approvals.id, job.data.approvalId)); - if (approval?.status !== "pending") continue; - await db - .update(approvals) - .set({ status: "expired", decidedAt: new Date() }) - .where(eq(approvals.id, approval.id)); - deps.logger.warn(`approval ${approval.id} expired — resuming run for onTimeout handling`); + // CAS pending → expired; null means a user already decided it (or a prior + // run of this job did) — nothing to do + const expired = await decideApproval(db, job.data.approvalId, { status: "expired" }); + if (!expired) continue; + deps.logger.warn(`approval ${expired.id} expired — resuming run for onTimeout handling`); await queue.enqueueRun(job.data.runId); // engine applies the template's onTimeout } }); @@ -129,6 +125,16 @@ setInterval(async () => { .from(runs) .where(and(eq(runs.status, "queued"), lt(runs.queuedAt, sql`now() - interval '30 seconds'`))); for (const run of stragglers) await queue.enqueueRun(run.id); + + // runs paused on an approval that has since been decided but whose resume + // enqueue was lost (e.g. the API/worker died between the decision and the + // send) — re-enqueue so the decision actually takes effect + const strandedApprovals = await db + .selectDistinct({ id: runs.id }) + .from(runs) + .innerJoin(approvals, eq(approvals.runId, runs.id)) + .where(and(eq(runs.status, "waiting_approval"), ne(approvals.status, "pending"))); + for (const run of strandedApprovals) await queue.enqueueRun(run.id); } catch (err) { deps.logger.warn("sweeper failed", { err: String(err) }); } diff --git a/packages/orchestration/src/engine/engine.integration.test.ts b/packages/orchestration/src/engine/engine.integration.test.ts index 1018c6f..5a7b857 100644 --- a/packages/orchestration/src/engine/engine.integration.test.ts +++ b/packages/orchestration/src/engine/engine.integration.test.ts @@ -35,6 +35,7 @@ import { InMemoryArtifactStore, silentLogger, } from "./fakes"; +import { appendRunEvent, decideApproval, transitionRun } from "./run-lifecycle"; const TEST_DATABASE_URL = process.env.TEST_DATABASE_URL ?? "postgres://localhost:5432/agrippa_test"; const TEMPLATES_DIR = path.resolve(import.meta.dirname, "../../../../templates"); @@ -544,3 +545,50 @@ describe.skipIf(!dbUp)("orchestration engine (FakeExecutor compliance suite)", ( expect(seen).toContain("approval.required"); }); }); + +describe.skipIf(!dbUp)("run-lifecycle module", () => { + it("transitionRun is a compare-and-swap on the expected status", async () => { + const { db, runId } = await setupFixture(); + // queued → running applies once; a second call with the stale `from` fails + expect(await transitionRun(db, runId, "queued", "running")).toBe(true); + expect(await transitionRun(db, runId, "queued", "running")).toBe(false); + // a late finalize cannot overwrite a status it no longer holds + expect(await transitionRun(db, runId, "running", "cancelled")).toBe(true); + expect(await transitionRun(db, runId, "running", "succeeded")).toBe(false); + const [row] = await db.select({ status: runs.status }).from(runs).where(eq(runs.id, runId)); + expect(row?.status).toBe("cancelled"); + }); + + it("appendRunEvent allocates a monotonic per-run seq from the database", async () => { + const { db, runId } = await setupFixture(); + const a = await appendRunEvent(db, { runId, type: "x.one", payload: {} }); + const b = await appendRunEvent(db, { runId, type: "x.two", payload: {} }); + // concurrent appends must not collide on the unique (run_id, seq) index + const [c, d] = await Promise.all([ + appendRunEvent(db, { runId, type: "x.three", payload: {} }), + appendRunEvent(db, { runId, type: "x.four", payload: {} }), + ]); + const seqs = [a.seq, b.seq, c.seq, d.seq].sort((m, n) => m - n); + expect(new Set(seqs).size).toBe(4); + expect(b.seq).toBeGreaterThan(a.seq); + }); + + it("decideApproval is a compare-and-swap on pending", async () => { + const { db, runId } = await setupFixture(); + const [approval] = await db + .insert(approvals) + .values({ runId, checkpointId: "cp-1", status: "pending" }) + .returning(); + const id = approval?.id as string; + const first = await decideApproval(db, id, { status: "approved" }); + expect(first?.status).toBe("approved"); + // a racing expiry cannot overwrite the user's decision + const second = await decideApproval(db, id, { status: "expired" }); + expect(second).toBeNull(); + const [row] = await db + .select({ status: approvals.status }) + .from(approvals) + .where(eq(approvals.id, id)); + expect(row?.status).toBe("approved"); + }); +}); diff --git a/packages/orchestration/src/engine/engine.ts b/packages/orchestration/src/engine/engine.ts index 864b85a..8923b01 100644 --- a/packages/orchestration/src/engine/engine.ts +++ b/packages/orchestration/src/engine/engine.ts @@ -1,9 +1,4 @@ -import { - canTransitionRun, - isTerminalRunStatus, - type RunStatus, - type StepStatus, -} from "@agrippa/core"; +import { isTerminalRunStatus, type RunStatus, type StepStatus } from "@agrippa/core"; import { approvals, artifacts, @@ -38,6 +33,7 @@ import { type TemplateStep, } from "../template-schema"; import type { EngineDeps, RunOutcome } from "./deps"; +import { appendRunEvent, transitionRun } from "./run-lifecycle"; type RunRow = typeof runs.$inferSelect; type StepRow = typeof runSteps.$inferSelect; @@ -92,7 +88,6 @@ export async function executeRun(deps: EngineDeps, runId: string): Promise 0; + const resuming = (maxSeq?.v ?? 0) > 0; - await this.transition(run.status, "running"); + if (!(await this.transition(run.status, "running"))) { + // another worker/path advanced the run between our read and here + const [current] = await db + .select({ status: runs.status }) + .from(runs) + .where(eq(runs.id, run.id)); + if (!current || isTerminalRunStatus(current.status)) { + throw new RunFailure( + current?.status === "cancelled" ? "cancelled" : "superseded", + "run already finalized by another worker", + current?.status === "cancelled" ? "cancelled" : "failed", + ); + } + this.run.status = current.status; + } if (run.startedAt === null) { await db.update(runs).set({ startedAt: new Date() }).where(eq(runs.id, run.id)); this.run.startedAt = new Date(); @@ -755,18 +763,17 @@ class RunEngine { payload: Record, stepRowId?: string, ): Promise { - this.seq += 1; - const createdAt = new Date(); - await this.db.insert(runEvents).values({ + // seq is allocated by the database (run-lifecycle.appendRunEvent), not from + // an in-memory counter that a concurrent writer could collide with + const { seq, createdAt } = await appendRunEvent(this.db, { runId: this.run.id, stepId: stepRowId ?? this.currentStepRowId, - seq: this.seq, type, payload, }); await this.deps.bus.publish({ runId: this.run.id, - seq: this.seq, + seq, type, payload, createdAt: createdAt.toISOString(), @@ -804,13 +811,15 @@ class RunEngine { if (this.abortReason) throw this.abortFailure(); } - private async transition(from: RunStatus, to: RunStatus): Promise { - if (from === to) return; - if (!canTransitionRun(from, to)) { - throw new Error(`illegal run transition ${from} → ${to}`); - } - await this.db.update(runs).set({ status: to }).where(eq(runs.id, this.run.id)); - this.run.status = to; + /** + * Compare-and-swap the run status. Returns false when the row had already + * moved off `from` (e.g. a concurrent cancel/finalize won the race); callers + * that must not clobber the other outcome check the result. + */ + private async transition(from: RunStatus, to: RunStatus): Promise { + const applied = await transitionRun(this.db, this.run.id, from, to); + if (applied) this.run.status = to; + return applied; } private async handleFailure(err: unknown): Promise { @@ -842,7 +851,9 @@ class RunEngine { error: { code: string; message: string } | null, ): Promise { const snapshot = this.meter?.snapshot() ?? { costUsd: 0, tokens: 0, perPhaseCostUsd: {} }; - await this.transition(this.run.status, status); + // CAS: if another path (e.g. a concurrent cancel) already finalized the run, + // do not overwrite its terminal status/error with ours + if (!(await this.transition(this.run.status, status))) return; await this.db .update(runs) .set({ diff --git a/packages/orchestration/src/engine/run-lifecycle.ts b/packages/orchestration/src/engine/run-lifecycle.ts new file mode 100644 index 0000000..7ff7f52 --- /dev/null +++ b/packages/orchestration/src/engine/run-lifecycle.ts @@ -0,0 +1,99 @@ +import { canTransitionRun, type RunStatus } from "@agrippa/core"; +import { approvals, type Db, runEvents, runs } from "@agrippa/db"; +import { and, eq, sql } from "drizzle-orm"; + +/** + * Run-lifecycle module (docs/design/04, ADR-0007): the one place that mutates + * run status and appends events, always atomically. + * + * Every status change is a compare-and-swap on the expected `from` status, so a + * late worker completion can never overwrite a concurrent cancellation, and two + * resumed jobs for the same run cannot both "win". Every event seq is allocated + * by the database inside the INSERT, so the per-run monotonic seq can't be + * seeded stale in memory and lost to a concurrent writer. + */ + +export type RunEventInput = { + runId: string; + stepId?: string | null; + type: string; + payload?: Record; +}; + +export type AppendedRunEvent = { seq: number; createdAt: Date }; + +function isUniqueViolation(err: unknown): boolean { + const e = err as { code?: string; message?: string }; + return e?.code === "23505" || /duplicate key|run_events_run_seq_uq/i.test(e?.message ?? ""); +} + +/** + * Move a run from `from` to `to` iff it is still in `from` (compare-and-swap). + * Returns true when this caller made the change, false when the row had already + * moved on (e.g. a cancel landed first). Rejects illegal transitions up front. + */ +export async function transitionRun( + db: Db, + runId: string, + from: RunStatus, + to: RunStatus, +): Promise { + if (from === to) return true; + if (!canTransitionRun(from, to)) { + throw new Error(`illegal run transition ${from} → ${to}`); + } + const updated = await db + .update(runs) + .set({ status: to }) + .where(and(eq(runs.id, runId), eq(runs.status, from))) + .returning({ id: runs.id }); + return updated.length > 0; +} + +/** + * Append a run event with a database-allocated per-run seq. The seq is computed + * inside the INSERT (max+1 over the run's events), so serial writers are always + * correct; the unique (run_id, seq) index backstops the rare concurrent race, + * on which we retry rather than fail the job. + */ +export async function appendRunEvent(db: Db, event: RunEventInput): Promise { + const nextSeq = sql`(select coalesce(max(${runEvents.seq}), 0) + 1 from ${runEvents} where ${runEvents.runId} = ${event.runId})`; + for (let attempt = 0; attempt < 5; attempt++) { + try { + const [row] = await db + .insert(runEvents) + .values({ + runId: event.runId, + stepId: event.stepId ?? null, + seq: nextSeq, + type: event.type, + payload: event.payload ?? {}, + }) + .returning({ seq: runEvents.seq, createdAt: runEvents.createdAt }); + if (!row) throw new Error("run_events insert returned no row"); + return row; + } catch (err) { + if (isUniqueViolation(err) && attempt < 4) continue; + throw err; + } + } + throw new Error("run_events seq allocation exhausted retries"); +} + +/** + * Decide a pending approval atomically. The `status = 'pending'` predicate makes + * this a compare-and-swap: a user decision and the expiry worker can't overwrite + * each other. Returns the updated row, or null if it was no longer pending. + */ +export async function decideApproval( + db: Db, + approvalId: string, + patch: { status: "approved" | "rejected" | "expired"; decidedBy?: string; comment?: string }, +): Promise { + const [updated] = await db + .update(approvals) + .set({ ...patch, decidedAt: new Date() }) + .where(and(eq(approvals.id, approvalId), eq(approvals.status, "pending"))) + .returning(); + return updated ?? null; +} diff --git a/packages/orchestration/src/index.ts b/packages/orchestration/src/index.ts index 4332463..a1c9b43 100644 --- a/packages/orchestration/src/index.ts +++ b/packages/orchestration/src/index.ts @@ -4,6 +4,7 @@ export * from "./engine/deps"; export * from "./engine/engine"; export * from "./engine/fakes"; export * from "./engine/redis-bus"; +export * from "./engine/run-lifecycle"; export * from "./expression"; export * from "./queue"; export * from "./resolve"; From 2facd4b6c71a3024d1c8f119db2286b4721a93a0 Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 16:06:47 +0800 Subject: [PATCH 05/18] fix(engine): recover crashed no-retry steps and resume their session MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A worker that died mid-step left the step row 'running'; on resume the engine marked it failed and started the next attempt at previous+1. For a step with no template-level retry that made maxAttempts=1 and startAttempt=2, so the retry loop ran zero iterations and the engine proceeded with the step neither executed nor failed — silently skipped when its output was optional, or surfacing later as a spurious contract_violation when its artifact was required. The new attempt row also never carried the prior session id, so resumeSessionId was unreachable. Treat a crash as an interrupted attempt rather than a consumed retry: count crashed attempts (error.code='crashed') during initialize and add one extra attempt each, so a crashed no-retry step re-executes. Carry the crashed row's executorSessionId onto the recovery attempt so the executor resumes its session. The existing crash test only exercised run-tests (retry: 2), which masked this; adds a no-retry crash test asserting re-execution and session resume. --- .../src/engine/engine.integration.test.ts | 24 +++++++++++++++++++ packages/orchestration/src/engine/engine.ts | 24 +++++++++++++++++-- 2 files changed, 46 insertions(+), 2 deletions(-) diff --git a/packages/orchestration/src/engine/engine.integration.test.ts b/packages/orchestration/src/engine/engine.integration.test.ts index 5a7b857..6f1f55c 100644 --- a/packages/orchestration/src/engine/engine.integration.test.ts +++ b/packages/orchestration/src/engine/engine.integration.test.ts @@ -422,6 +422,30 @@ describe.skipIf(!dbUp)("orchestration engine (FakeExecutor compliance suite)", ( expect(runTestRows).toHaveLength(2); }); + it("crash on a no-retry step re-executes it on resume and resumes the session", async () => { + const { db, runId, makeDeps } = await setupFixture(); + // find-root-cause carries no template retry; crash it mid-step + const crashing = makeDeps({ ...HAPPY_SCRIPT, "find-root-cause": { kind: "crash" } }); + await expect(executeRun(crashing, runId)).rejects.toThrow("simulated worker crash"); + + // resume with a healthy worker: without the crash-recovery fix a no-retry + // step's loop is `for (2; 2 <= 1)` and the step is silently skipped + const healthy = makeDeps(HAPPY_SCRIPT); + expect(await executeRun(healthy, runId)).toBe("waiting_approval"); + + const attempts = await db + .select() + .from(runSteps) + .where(and(eq(runSteps.runId, runId), eq(runSteps.stepId, "find-root-cause"))); + expect(attempts.map((a) => [a.attempt, a.status]).sort()).toEqual([ + [1, "failed"], + [2, "succeeded"], + ]); + // the recovery attempt resumed the crashed executor session + const request = healthy.executor.requests.find((r) => r.stepId === "find-root-cause"); + expect(request?.resumeSessionId).toBe("fake-find-root-cause-1"); + }); + it("cancellation mid-step aborts promptly via the control channel", async () => { const { db, runId, makeDeps, bus } = await setupFixture(); const deps = makeDeps({ ...HAPPY_SCRIPT, "find-root-cause": { kind: "hang" } }); diff --git a/packages/orchestration/src/engine/engine.ts b/packages/orchestration/src/engine/engine.ts index 8923b01..c3ffbfa 100644 --- a/packages/orchestration/src/engine/engine.ts +++ b/packages/orchestration/src/engine/engine.ts @@ -99,6 +99,8 @@ class RunEngine { private modelPrices = new Map(); private currentStepRowId: string | null = null; private producedArtifacts = new Set(); + // stepId → crashed-attempt count + last executor session, for crash resume + private crashRecovery = new Map(); constructor( private readonly deps: EngineDeps, @@ -187,6 +189,15 @@ class RunEngine { }) .where(eq(runSteps.id, row.id)); row.status = "failed"; + row.error = { code: "crashed", message: "worker died mid-step" }; + } + // a crash is an interrupted attempt, not a consumed retry: track it so the + // step gets an extra attempt and resumes the executor session + if (row.status === "failed" && (row.error as { code?: string } | null)?.code === "crashed") { + const rec = this.crashRecovery.get(row.stepId) ?? { crashed: 0, sessionId: null }; + rec.crashed += 1; + if (row.executorSessionId) rec.sessionId = row.executorSessionId; + this.crashRecovery.set(row.stepId, rec); } const current = this.stepRows.get(row.stepId); if (!current || row.attempt > current.attempt) this.stepRows.set(row.stepId, row); @@ -367,7 +378,11 @@ class RunEngine { // ── Steps ──────────────────────────────────────────────────────────────────── private async executeStepWithRetry(phase: TemplatePhase, step: TemplateStep): Promise { - const maxAttempts = (step.retry?.max ?? 0) + 1; + // crashes don't consume the retry budget — each adds one extra attempt so a + // no-retry step that died mid-run still re-executes instead of silently + // being skipped (its loop would otherwise be `for (2; 2 <= 1)`) + const recovery = this.crashRecovery.get(step.id); + const maxAttempts = (step.retry?.max ?? 0) + 1 + (recovery?.crashed ?? 0); const startAttempt = (this.stepRows.get(step.id)?.attempt ?? 0) + 1; // conditional / requires gating @@ -392,7 +407,9 @@ class RunEngine { for (let attempt = startAttempt; attempt <= maxAttempts; attempt++) { await this.checkInterrupts(); - const row = await this.insertStepRow(phase, step, attempt); + // resume the crashed executor session on the first recovery attempt only + const resumeSessionId = attempt === startAttempt ? (recovery?.sessionId ?? null) : null; + const row = await this.insertStepRow(phase, step, attempt, resumeSessionId); this.currentStepRowId = row.id; try { if (step.kind === "system") { @@ -689,6 +706,7 @@ class RunEngine { phase: TemplatePhase, step: TemplateStep, attempt: number, + resumeSessionId: string | null = null, ): Promise { const [row] = await this.db .insert(runSteps) @@ -700,6 +718,8 @@ class RunEngine { seq: this.stepSeq(step.id), status: "running", agentRef: step.kind === "agent" ? step.model.role : step.action, + // carry the crashed attempt's session so buildRequest can resume it + executorSessionId: resumeSessionId, startedAt: new Date(), }) .returning(); From b514faa89479eae3f2cd0bc56f1fc32613b2bf1d Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 16:10:49 +0800 Subject: [PATCH 06/18] fix(engine): align quota window and stop double-counting run spend on resume MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The submit gate counted the current month's project usage while the engine's mid-run quota check summed all-time usage — two different windows for the same limit. Worse, the engine seeded the budget meter with this run's persisted spend AND subtracted that same spend inside the headroom it checked against, so on resume the run's own usage was counted twice and could trip the quota early (spend > limit - spend). And the headroom was snapshotted once at start, so concurrent runs each measured only their own increment and could jointly overspend. Scope the engine's headroom query to the current month (matching the submit gate), exclude the run's own rows (the meter already carries them), and re-read it at every step boundary via a new BudgetMeter.refreshQuota so a run reacts to other runs' spend instead of a stale snapshot. Adds a resume regression test proving a run under quota is not failed by double-counting, and corrects the usage.ts comment that overstated the old enforcement. --- apps/api/src/lib/usage.ts | 6 ++-- packages/executor-core/src/budget.ts | 16 +++++++--- .../src/engine/engine.integration.test.ts | 31 +++++++++++++++++++ packages/orchestration/src/engine/engine.ts | 21 +++++++++++-- 4 files changed, 66 insertions(+), 8 deletions(-) diff --git a/apps/api/src/lib/usage.ts b/apps/api/src/lib/usage.ts index 5911f04..05afb5e 100644 --- a/apps/api/src/lib/usage.ts +++ b/apps/api/src/lib/usage.ts @@ -45,8 +45,10 @@ export async function projectUsage(db: Db, projectId: string): Promise { const [quota] = await db diff --git a/packages/executor-core/src/budget.ts b/packages/executor-core/src/budget.ts index b333e1a..1d298d5 100644 --- a/packages/executor-core/src/budget.ts +++ b/packages/executor-core/src/budget.ts @@ -35,13 +35,12 @@ export type BudgetSnapshot = { export class BudgetMeter { private costUsd: number; private tokens: number; + private limits: BudgetLimits; private readonly perPhase: Record; private currentPhase = ""; - constructor( - private readonly limits: BudgetLimits, - initial: Partial = {}, - ) { + constructor(limits: BudgetLimits, initial: Partial = {}) { + this.limits = limits; this.costUsd = initial.costUsd ?? 0; this.tokens = initial.tokens ?? 0; this.perPhase = { ...(initial.perPhaseCostUsd ?? {}) }; @@ -52,6 +51,15 @@ export class BudgetMeter { this.perPhase[phaseId] ??= 0; } + /** + * Refresh the project-quota headroom (leaving run/phase limits untouched). + * Called at each step boundary so a run reacts to quota consumed by other + * concurrent runs rather than a stale snapshot taken at start. + */ + refreshQuota(quotaCostUsd: number | undefined, quotaTokens: number | undefined): void { + this.limits = { ...this.limits, quotaCostUsd, quotaTokens }; + } + record(usage: UsageDelta & { costUsd: number }): void { this.costUsd += usage.costUsd; this.tokens += usage.inputTokens + usage.outputTokens; diff --git a/packages/orchestration/src/engine/engine.integration.test.ts b/packages/orchestration/src/engine/engine.integration.test.ts index 6f1f55c..1cdbc28 100644 --- a/packages/orchestration/src/engine/engine.integration.test.ts +++ b/packages/orchestration/src/engine/engine.integration.test.ts @@ -387,6 +387,37 @@ describe.skipIf(!dbUp)("orchestration engine (FakeExecutor compliance suite)", ( expect((run?.error as { code: string } | null)?.code).toBe("budget_exceeded"); }); + it("resume does not double-count the run's own spend against the quota", async () => { + const { runId, makeDeps } = await setupFixture({ quota: { tokenLimit: 3000 } }); + // cheap pre-approval steps: reproduce-bug spends 1600, find-root-cause 200 + const cheap: Record = { + ...HAPPY_SCRIPT, + "reproduce-bug": { + kind: "succeed", + usage: { inputTokens: 1000, outputTokens: 600 }, + events: [{ type: "artifact", key: "reproduction-report", kind: "markdown", inline: "# R" }], + output: "reproduced", + }, + "find-root-cause": { + kind: "succeed", + usage: { inputTokens: 100, outputTokens: 100 }, + events: [ + { type: "artifact", key: "localization-report", kind: "markdown", inline: "# RC" }, + ], + output: "rc", + }, + }; + // spend 1600, then crash before the approval gate + await expect( + executeRun(makeDeps({ ...cheap, "find-root-cause": { kind: "crash" } }), runId), + ).rejects.toThrow("simulated worker crash"); + + // 1600 is already persisted. The old code subtracted it from the headroom + // AND seeded the meter with it, double-counting on resume and tripping the + // 3000 quota (1600 > 3000 - 1600). It must instead reach the approval gate. + expect(await executeRun(makeDeps(cheap), runId)).toBe("waiting_approval"); + }); + it("crash mid-step → queue retry resumes, skips succeeded steps, never double-counts usage", async () => { const { db, runId, makeDeps } = await setupFixture(); await executeRun(makeDeps(HAPPY_SCRIPT), runId); diff --git a/packages/orchestration/src/engine/engine.ts b/packages/orchestration/src/engine/engine.ts index c3ffbfa..006ec59 100644 --- a/packages/orchestration/src/engine/engine.ts +++ b/packages/orchestration/src/engine/engine.ts @@ -23,7 +23,7 @@ import { type StepExecutionRequest, type UsageDelta, } from "@agrippa/executor-core"; -import { and, eq, max, sql } from "drizzle-orm"; +import { and, eq, gte, max, ne, sql } from "drizzle-orm"; import { evaluateCondition, evaluateExpression, interpolate } from "../expression"; import type { ModelResolution } from "../resolve"; import { @@ -250,6 +250,13 @@ class RunEngine { this.workspaceDir = await this.deps.workspace.ensureDir(run.id); } + /** + * Remaining project quota headroom for *this* run, refreshed at each step + * boundary so concurrent runs see each other's spend. Counts the current + * month (matching the submit-time gate in apps/api usage.ts) and excludes + * this run's own persisted usage — the meter already carries that, so + * including it here would double-count on resume. + */ private async quotaHeadroom(): Promise<{ costUsd?: number; tokens?: number }> { const [quota] = await this.db .select() @@ -262,7 +269,13 @@ class RunEngine { tokens: sql`coalesce(sum(${tokenUsage.inputTokens} + ${tokenUsage.outputTokens}), 0)`, }) .from(tokenUsage) - .where(eq(tokenUsage.projectId, this.run.projectId)); + .where( + and( + eq(tokenUsage.projectId, this.run.projectId), + ne(tokenUsage.runId, this.run.id), + gte(tokenUsage.occurredAt, sql`date_trunc('month', now())`), + ), + ); const headroom: { costUsd?: number; tokens?: number } = {}; if (quota.costLimitUsd !== null) { headroom.costUsd = Math.max(0, Number(quota.costLimitUsd) - Number(spent?.cost ?? 0)); @@ -300,6 +313,10 @@ class RunEngine { continue; } await this.checkInterrupts(); + // re-read project quota each step so concurrent runs can't collectively + // overspend by each checking only a stale start-of-run snapshot + const quota = await this.quotaHeadroom(); + this.meter.refreshQuota(quota.costUsd, quota.tokens); this.meter.check(); await this.executeStepWithRetry(phase, step); } From 9f001a1a3ebd86d8ca3125acc5f8943b96c40d4e Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 16:15:31 +0800 Subject: [PATCH 07/18] fix(artifacts): enforce the step output contract on ingestion The patch-exclusion filter in the artifact instructions matched a substring ("(patch)") the generated lines never contain, so patch steps were told to hand-write .agrippa/artifacts/patch. When the agent did, the executor collected it as kind 'file', the engine saw the key already produced and skipped its own git-diff generation, and the stored "patch" was the model's file rather than the real diff. The executor also collected every file in the directory with a kind guessed from the extension, so keys/kinds were never checked against the contract, files left by earlier steps were re-emitted on later ones, and missing files still created zero-byte artifact rows. Scope collection to the step's declared artifacts: emit only contracted keys, with the contracted kind, once each, skipping patch keys (the engine owns those) and files from other steps. The engine additionally drops any artifact event whose key is not in step.produces and refuses to store a source that produced no bytes (so a required-but-missing artifact still fails the contract instead of becoming an empty row). Adds an executor test for contract-scoped collection. --- packages/executor-claude/src/executor.test.ts | 37 +++++++++++++++ packages/executor-claude/src/executor.ts | 46 ++++++++++--------- packages/orchestration/src/engine/engine.ts | 19 +++++++- 3 files changed, 79 insertions(+), 23 deletions(-) diff --git a/packages/executor-claude/src/executor.test.ts b/packages/executor-claude/src/executor.test.ts index e07bfb1..705d74f 100644 --- a/packages/executor-claude/src/executor.test.ts +++ b/packages/executor-claude/src/executor.test.ts @@ -232,6 +232,43 @@ describe("claude executor event stream", () => { expect(events.at(-1)).toEqual({ type: "step.completed", output: "All done." }); }); + it("collects only contracted artifacts, skipping patch and uncontracted files", async () => { + const req = makeRequest({ + expectedArtifacts: [ + { key: "fix-report", kind: "markdown" }, + { key: "patch", kind: "patch" }, + ], + }); + const artifactDir = path.join(req.workspaceDir, ".agrippa/artifacts"); + mkdirSync(artifactDir, { recursive: true }); + writeFileSync(path.join(artifactDir, "fix-report.md"), "# fixed"); + // the agent also wrote the patch (engine generates it) and a stray file + writeFileSync(path.join(artifactDir, "patch"), "diff --git a/x b/x"); + writeFileSync(path.join(artifactDir, "scratch.txt"), "junk"); + + // patch instructions must not tell the agent to author the patch file + const { prompt } = buildQueryArgs(req, makeCtx(), new AbortController()); + expect(prompt).toContain(".agrippa/artifacts/fix-report.md"); + expect(prompt).not.toContain(".agrippa/artifacts/patch"); + + const executor = createClaudeExecutor( + scriptedQuery([ + { type: "system", subtype: "init", session_id: "s" }, + { type: "result", subtype: "success", result: "done", is_error: false }, + ]), + ); + const events = await collect(executor.executeStep(req, makeCtx())); + const artifacts = events.filter((e) => e.type === "artifact"); + expect(artifacts).toEqual([ + { + type: "artifact", + key: "fix-report", + kind: "markdown", + path: ".agrippa/artifacts/fix-report.md", + }, + ]); + }); + it("maps SDK errors and aborts to step.failed", async () => { const failing = createClaudeExecutor( scriptedQuery([ diff --git a/packages/executor-claude/src/executor.ts b/packages/executor-claude/src/executor.ts index cf27885..5003d0d 100644 --- a/packages/executor-claude/src/executor.ts +++ b/packages/executor-claude/src/executor.ts @@ -1,6 +1,5 @@ import { readdirSync } from "node:fs"; import path from "node:path"; -import type { ArtifactKind } from "@agrippa/core"; import { buildScrubbedEnv, type ExecutionContext, @@ -15,24 +14,18 @@ type QueryFn = (params: { prompt: string; options?: Options }) => AsyncIterable< const ARTIFACT_DIR = ".agrippa/artifacts"; -const KIND_BY_EXT: Record = { - ".md": "markdown", - ".txt": "markdown", - ".json": "json", - ".diff": "patch", - ".patch": "patch", - ".url": "link", -}; - /** Prompt preamble telling the agent where declared artifacts must land. */ function artifactInstructions(expected: StepExecutionRequest["expectedArtifacts"]): string { - if (expected.length === 0) return ""; - const list = expected + // patch artifacts are generated by the engine from the git diff — the agent + // must not hand-write them, or the engine skips its own diff (it sees the key + // already produced) and stores the model's file instead of the real patch + const authored = expected.filter((a) => a.kind !== "patch"); + if (authored.length === 0) return ""; + const list = authored .map( (a) => `- ${ARTIFACT_DIR}/${a.key}${a.kind === "json" ? ".json" : a.kind === "markdown" ? ".md" : ""}`, ) - .filter((line) => !line.includes("(patch)")) .join("\n"); return [ "", @@ -141,8 +134,18 @@ function textOf(content: Array<{ type: string; text?: string }>): string { .join(""); } -/** Scans the artifact convention directory and emits one event per file. */ -function* collectArtifacts(workspaceDir: string): Generator { +/** + * Scans the artifact convention directory and emits one event per file that the + * step actually contracted to produce. Scoping to `expected` (a) validates the + * key/kind against the contract, (b) uses the declared kind rather than guessing + * from the extension, (c) skips patch artifacts (the engine generates those), + * and (d) ignores files left by earlier steps so a rescan can't re-emit them. + */ +function* collectArtifacts( + workspaceDir: string, + expected: StepExecutionRequest["expectedArtifacts"], +): Generator { + const kindByKey = new Map(expected.map((a) => [a.key, a.kind] as const)); const dir = path.join(workspaceDir, ARTIFACT_DIR); let entries: string[] = []; try { @@ -150,15 +153,14 @@ function* collectArtifacts(workspaceDir: string): Generator { } catch { return; } + const emitted = new Set(); for (const entry of entries) { const ext = path.extname(entry); const key = ext ? entry.slice(0, -ext.length) : entry; - yield { - type: "artifact", - key, - kind: KIND_BY_EXT[ext] ?? "file", - path: path.join(ARTIFACT_DIR, entry), - }; + const kind = kindByKey.get(key); + if (kind === undefined || kind === "patch" || emitted.has(key)) continue; + emitted.add(key); + yield { type: "artifact", key, kind, path: path.join(ARTIFACT_DIR, entry) }; } } @@ -289,7 +291,7 @@ export function createClaudeExecutor(queryFn: QueryFn = sdkQuery as QueryFn): Ex return; } if (terminal?.type === "step.completed") { - yield* collectArtifacts(req.workspaceDir); + yield* collectArtifacts(req.workspaceDir, req.expectedArtifacts); } yield terminal ?? { type: "step.failed", diff --git a/packages/orchestration/src/engine/engine.ts b/packages/orchestration/src/engine/engine.ts index 006ec59..b4937bb 100644 --- a/packages/orchestration/src/engine/engine.ts +++ b/packages/orchestration/src/engine/engine.ts @@ -668,7 +668,21 @@ class RunEngine { .where(eq(runSteps.id, row.id)); } if (event.type === "artifact") { - await this.storeArtifact(row, event); + // only accept artifacts the step contracted to produce — a non-compliant + // executor cannot smuggle in uncontracted keys or a mismatched kind + const produces = "produces" in step ? step.produces : []; + if (!produces.includes(event.key)) { + this.deps.logger.warn("dropping uncontracted artifact", { + runId: this.run.id, + stepId: step.id, + key: event.key, + }); + return; + } + const contractKind = this.template.spec.outputs.artifacts.find( + (a) => a.key === event.key, + )?.kind; + await this.storeArtifact(row, { ...event, kind: contractKind ?? event.kind }); } } @@ -683,6 +697,9 @@ class RunEngine { { inline: event.inline, path: event.path }, this.workspaceDir, ); + // a missing/empty source produced no bytes — don't create a zero-byte row + // (and don't mark the key produced, so a required artifact still fails) + if (stored.inline === null && stored.storageRef === null) return; await this.db.insert(artifacts).values({ runId: this.run.id, stepId: row.id, From ad487004af213ba8be0fb4386a8064c94b986314 Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 16:17:49 +0800 Subject: [PATCH 08/18] fix(api): close the SSE replay/subscribe gap by subscribing first MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The events stream replayed Postgres history and only then subscribed to the bus. An event committed and published in the window between those two steps was in neither the replayed rows nor the (not-yet-active) subscription, so it surfaced live only at the terminal replay — a live viewer of a long-running run could miss an event for the run's entire duration, contradicting ADR-0007's gap-free guarantee. Subscribe before the initial replay so live events buffer while history is read, then flush the buffer deduped against the replay by seq. The no-bus DB-polling path is unchanged. Strengthens the SSE test to assert delivered seqs are unique and strictly increasing, guarding the dedup this reorder depends on. --- apps/api/src/routes/execution.ts | 14 ++++++++------ apps/api/src/test/execution.integration.test.ts | 4 ++++ 2 files changed, 12 insertions(+), 6 deletions(-) diff --git a/apps/api/src/routes/execution.ts b/apps/api/src/routes/execution.ts index c012a3f..a60db76 100644 --- a/apps/api/src/routes/execution.ts +++ b/apps/api/src/routes/execution.ts @@ -418,9 +418,6 @@ export const executionRoutes = new Hono() return rows; }; - // 1) replay history (gap-free by construction) - await replay(); - const isTerminal = async (): Promise => { const [row] = await db .select({ status: runs.status }) @@ -429,9 +426,11 @@ export const executionRoutes = new Hono() return row ? isTerminalRunStatus(row.status) : true; }; - if (await isTerminal()) return; - - // 2) live: bridge the bus when present, else poll the DB + // Live: bridge the bus when present, else poll the DB. With a bus we must + // subscribe BEFORE replaying Postgres — an event committed and published + // in the window between replay and subscribe would otherwise be delivered + // only at the terminal replay, contradicting ADR-0007's gap-free + // guarantee. Buffered live events dedupe against the replay by seq. if (bus) { const queue: Array<() => Promise> = []; let notify: (() => void) | null = null; @@ -444,6 +443,7 @@ export const executionRoutes = new Hono() notify?.(); }); try { + await replay(); // history first; the subscription is already buffering while (!closed) { while (queue.length > 0) { const job = queue.shift(); @@ -463,6 +463,8 @@ export const executionRoutes = new Hono() unsubscribe(); } } else { + await replay(); + if (await isTerminal()) return; while (!closed) { await new Promise((resolve) => setTimeout(resolve, 1000)); await replay(); diff --git a/apps/api/src/test/execution.integration.test.ts b/apps/api/src/test/execution.integration.test.ts index 9bccd05..adb57a7 100644 --- a/apps/api/src/test/execution.integration.test.ts +++ b/apps/api/src/test/execution.integration.test.ts @@ -250,6 +250,10 @@ describe.skipIf(!dbUp)("execution api (submit → engine → approve → artifac // resume from the middle: only later events are replayed const ids = [...text.matchAll(/^id: (\d+)$/gm)].map((m) => Number(m[1])); + // subscribe-before-replay must not deliver any event twice: seq strictly + // increasing, deduped by cursor (ADR-0007) + expect(new Set(ids).size).toBe(ids.length); + expect([...ids].sort((a, b) => a - b)).toEqual(ids); const middle = ids[Math.floor(ids.length / 2)] as number; const partial = await viewer.request(`/api/v1/runs/${runId}/events`, { headers: { "last-event-id": String(middle) }, From dbc379fe1499e67e43f6ee84a9ee447e094d6bcd Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 16:25:56 +0800 Subject: [PATCH 09/18] docs: record the security & correctness hardening Update the authoritative design docs, ADRs, manual, and changelog to match the P0/P1 fixes: - ADR-0009 records the three deep-module decisions (execution-isolation seam, authorized run-manifest, run-lifecycle module). - design/03 documents the isolation seam, env scrubbing, OS sandbox, repo-config stripping, and contract-scoped artifact collection; 04 documents repoRef ownership + the pinned manifest at submit, CAS transitions, DB-allocated event seq, crash-recovery of no-retry steps, the corrected quota window/double-count, approval CAS + recovery, and subscribe-before-replay SSE; 05 notes repoRef authorization and approval CAS; 09 replaces the false frontend-test claim with the tests that now exist (isolation, worker-adapter, lifecycle, quota-resume). - ARCHITECTURE.md gains invariants for the isolation seam, the trusted manifest, and atomic lifecycle mutations. - The bilingual manual notes that optional resources need a grant to be used and adds troubleshooting rows; CHANGELOG [Unreleased] summarizes the security and fixed entries. --- ARCHITECTURE.md | 9 ++++-- CHANGELOG.md | 14 +++++++++ .../0009-security-correctness-deep-modules.md | 30 +++++++++++++++++++ docs/design/03-executor-abstraction.md | 13 ++++---- docs/design/04-execution-runtime.md | 25 +++++++++------- docs/design/05-api-and-auth.md | 2 ++ docs/design/09-testing-and-ci.md | 17 +++++++---- docs/manual/en/04-administration.md | 2 +- docs/manual/en/06-operations.md | 2 ++ docs/manual/zh-CN/04-administration.md | 2 +- docs/manual/zh-CN/06-operations.md | 2 ++ 11 files changed, 92 insertions(+), 26 deletions(-) create mode 100644 docs/adr/0009-security-correctness-deep-modules.md diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index dc5327b..5741bed 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -1,6 +1,6 @@ # Architecture -This document is the orientation map for contributors. The authoritative design lives in `docs/design/` (10 documents) and `docs/adr/` (8 decision records); this file tells you what exists, where it lives, and which invariants hold it together. +This document is the orientation map for contributors. The authoritative design lives in `docs/design/` (10 documents) and `docs/adr/` (9 decision records); this file tells you what exists, where it lives, and which invariants hold it together. ## Bird's-eye view @@ -35,7 +35,7 @@ Dependency direction is enforced by `scripts/check-deps.ts` (runtime deps only). ## Key flows -**Submit → run.** `POST /projects/:id/tasks` validates params against the compiled input schema (the same schema the SPA rendered the form from), verifies skill/MCP grants, resolves model roles → concrete granted models (frozen into `runs.model_resolution`), checks hard-stop quota headroom, then inserts task+run in one transaction and sends a pg-boss job keyed by run id. A worker sweeper re-enqueues stragglers, so a run can never be stranded nor double-queued. +**Submit → run.** `POST /projects/:id/tasks` validates params against the compiled input schema (the same schema the SPA rendered the form from), verifies each `repoRef` is owned by the project, resolves model roles → concrete granted models (frozen into `runs.model_resolution`), pins the authorized skills/MCP into `runs.resource_manifest` (required grants enforced, optional included only when granted), checks hard-stop quota headroom, then inserts task+run in one transaction and sends a pg-boss job keyed by run id. A worker sweeper re-enqueues stragglers (and runs left paused by a lost approval-resume), so a run can never be stranded nor double-queued. **Engine loop.** The engine (worker-side) walks phases/steps: `when:` conditions, optional-resource skips, per-step retries, budget checks at every boundary. Agent steps stream normalized executor events which the engine persists to `run_events` (per-run monotonic `seq`), mirrors to the bus, and projects into `run_steps`/`token_usage`/`artifacts`. At the end it enforces the artifact contract — *succeeded* always means the contracted outputs exist. @@ -52,7 +52,10 @@ Dependency direction is enforced by `scripts/check-deps.ts` (runtime deps only). 3. **Steps are the idempotency unit** — restart-safe by template rule, resumable by session id where the executor supports it. 4. **Usage rows are keyed `(run, step, attempt)`** so retries re-incur cost without ever double-counting. 5. **Executors are stateless I/O**: all inputs in the request, all outputs as events; they never touch the database. -6. **Secrets never leave as plaintext**: encrypted at rest (AES-256-GCM), write-only in the API, scrubbed from git remotes before agent code runs. +6. **Secrets never leave as plaintext**: encrypted at rest (AES-256-GCM), write-only in the API, scrubbed from git remotes before agent code runs, and stripped from the agent subprocess environment (the master key and datastore URLs never reach a tool call). +7. **Containment goes through one seam**: every tool call and the subprocess env are decided by `packages/executor-core/isolation.ts`; the adapter never reimplements it (ADR-0009). +8. **The worker trusts only the pinned manifest**: repos are project-scoped and skills/MCP resolve solely from `runs.resource_manifest`, never the mutable global registry. +9. **Lifecycle mutations are atomic**: run status transitions are compare-and-swap on the expected status and event `seq` is allocated by the database, so concurrent writers can't clobber a status or collide on a seq (`run-lifecycle.ts`). ## Where to look diff --git a/CHANGELOG.md b/CHANGELOG.md index 2c73a45..0d8fd88 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,20 @@ All notable changes to Agrippa are documented here. The format follows ## [Unreleased] +### Security + +- **Executor isolation seam** — one enforceable place (`packages/executor-core/isolation.ts`) decides every tool call and scrubs the subprocess environment. Read-only workspaces now actually deny shell and confine writes to the artifact directory; read-write workspaces confine writes to the workspace with a boundary-safe check (the previous `startsWith` let a sibling `-evil` path through, and `Bash` bypassed the check entirely). The SDK subprocess runs with the platform secrets (`AGRIPPA_SECRET_KEY`, datastore URLs) stripped from its environment, the OS `sandbox` enabled where available, `strictMcpConfig`, and repo-supplied `.claude`/`.mcp.json` removed after checkout so a checked-out repository can't inject hooks or permission overrides. The worker image runs as a non-root user. +- **Artifact path containment** — artifact ingestion resolves sources through `realpath` and rejects any that escape the workspace, closing a symlink disclosure (e.g. `ln -s /proc/self/environ`) that could exfiltrate secrets or other runs' files through the download endpoint. +- **Cross-tenant resource authorization** — submission rejects a `repoConnectionId` that isn't owned by the project, and the worker loads repo connections scoped to the run's project. Optional skills/MCP servers are now grant-checked: an authorized-resource manifest is pinned onto the run at submit (required grants enforced, optional resources included only when granted) and the worker resolves resources only from it, so a project without a grant can no longer receive the platform's global credential (e.g. the shared GitHub token). + +### Fixed + +- **Crash recovery for no-retry steps** — a worker that died mid-step no longer silently skips (or spuriously fails) a step without template retries; the crashed attempt no longer consumes the retry budget, and the executor session is carried onto the recovery attempt so resume works. +- **Atomic run lifecycle** — run status transitions are compare-and-swap on the expected status (a late finalize can't overwrite a cancellation), event sequence numbers are allocated by the database (no `max+1` collisions), and approval decisions are CAS on `pending` with the sweeper re-enqueuing any run left paused by a lost resume enqueue. +- **Quota accounting** — the engine now counts the same monthly window as the submit gate, excludes the run's own spend from the headroom it checks (no double-count on resume), and re-reads project usage at each step boundary so concurrent runs can't jointly overspend. +- **Artifact output contract** — patch steps no longer hand-write the diff (the engine generates it from `git diff` as intended), collected artifacts are validated against the step's declared keys/kinds, files from earlier steps aren't re-emitted, and missing/empty sources don't create zero-byte artifact rows. +- **SSE live gap** — the events stream subscribes before replaying history, so an event committed in the replay/subscribe window is delivered live instead of only at the terminal replay (ADR-0007). + ## [0.1.0] — 2026-07-17 The M1 milestone: all three layers of the platform, working end to end. diff --git a/docs/adr/0009-security-correctness-deep-modules.md b/docs/adr/0009-security-correctness-deep-modules.md new file mode 100644 index 0000000..cc94062 --- /dev/null +++ b/docs/adr/0009-security-correctness-deep-modules.md @@ -0,0 +1,30 @@ +# ADR-0009: Deep Modules for Execution Isolation, Authorized Resources, and the Run Lifecycle + +- Status: accepted · Date: 2026-07-17 + +## Context + +An M1 code review found that several documented invariants were declared but not enforced, and that the enforcement logic that did exist was scattered across the API, engine, worker, and SDK adapter — so it was easy for one caller to skip a check. The critical cases: + +- **Execution isolation** leaked into the SDK adapter: only three write tools were path-checked (Bash and everything else bypassed containment), the check used a prefix test that admitted sibling paths, `workspace.access: readOnly` was inert, the checked-out repository's `.claude` settings/hooks were honored, and the agent subprocess inherited the worker's full environment including the master secret key. +- **Resource authorization** was split: submission validated required grants, but the worker independently re-resolved skills/MCP from the mutable global registry — and optional resources skipped the grant check entirely, so an ungranted project could still be handed the platform's global credential. `repoRef` was only shape-validated and the worker loaded the connection by raw id, a cross-tenant IDOR. +- **Run lifecycle** mutations (status transitions, event-seq allocation, approval decisions, queue handoff) were non-atomic and duplicated between the API, engine, and worker, so concurrent writers could clobber each other or strand a run. + +## Decision + +Concentrate each concern behind one deep module whose interface is the test surface, rather than re-litigating the accepted ADRs (Bun, Drizzle, pg-boss, SSE): + +1. **Execution-isolation seam** (`packages/executor-core/isolation.ts`) — `evaluateToolCall`, `isWithin`, and `buildScrubbedEnv` are pure and back both the SDK adapter and its tests. The adapter must route every tool decision and the subprocess environment through it, and layers OS-level controls on top (the SDK `sandbox`, a non-root worker, repo-config stripping at checkout). +2. **Authorized run-manifest** — `resolve.authorizeResources` pins the exact skills/MCP a run may use (required grants enforced, optional included only when granted) into `runs.resource_manifest` at submit; `verifyRepoRefs` enforces repo-connection ownership. The engine resolves resources only from the manifest, never the global registry, and the worker loads repo connections scoped to the run's project. +3. **Run-lifecycle module** (`packages/orchestration/src/engine/run-lifecycle.ts`) — `transitionRun` (compare-and-swap on the expected status), `appendRunEvent` (database-allocated per-run seq), and `decideApproval` (CAS on `pending`) own every lifecycle mutation for both the API and the worker. + +## Alternatives considered + +- **Minimal in-place patches** (add a WHERE clause here, an `if` there): faster, but leaves the cross-cutting sprawl the review flagged, so the next caller can skip the check again. Rejected in favor of one enforceable seam per concern. +- **Grant-aware resolution at the worker instead of a pinned manifest**: closes the token leak but still trusts the mutable global registry mid-run; the manifest pins the decision at submit, which is both safer and a smaller worker interface. + +## Consequences + +- The static isolation layer contains file writes and refuses shell in read-only workspaces, but it cannot bound arbitrary writes a shell command makes in a read-write workspace — that remains the OS sandbox / non-root worker / container's job, and full container-level isolation is still future work. +- Adds a `runs.resource_manifest` column (migration 0002); retries and resumes carry the pinned manifest, so authorization can't drift after submit. +- Grants now genuinely gate optional resources: an optional skill/MCP with no project grant is treated as unavailable, so its dependent step is skipped rather than silently privileged. diff --git a/docs/design/03-executor-abstraction.md b/docs/design/03-executor-abstraction.md index 72a071b..63f5209 100644 --- a/docs/design/03-executor-abstraction.md +++ b/docs/design/03-executor-abstraction.md @@ -91,20 +91,21 @@ The engine consumes this stream and: appends each event to `run_events` (assigni | Resume | capture `session_id` from init → `step.started.sessionId`; restore via `resume` option | | `limits.maxTurns` | `maxTurns`; duration via `AbortSignal.timeout` composed into `ctx.signal` | -**Artifacts convention**: the step prompt instructs the agent to write declared artifacts to `/.agrippa/artifacts/`. The executor watches that directory and emits `artifact` events. `patch`-kind artifacts are generated by the *engine* (not the executor) via `git diff` against the checkout base after the step completes — uniform across executors. +**Artifacts convention**: the step prompt instructs the agent to write declared artifacts to `/.agrippa/artifacts/`. The executor scans that directory after a successful step and emits an `artifact` event **only for the keys the step contracted to produce**, using the declared kind — so files left by earlier steps aren't re-emitted, and an uncontracted or mismatched artifact can't slip through. `patch`-kind artifacts are excluded from both the prompt and the scan: they are generated by the *engine* (not the executor) via `git diff` against the checkout base after the step completes — uniform across executors. The engine additionally refuses to store a source that produced no bytes, so a required-but-missing artifact fails the output contract rather than becoming an empty row. -**`priorContext`**: each completed step's final output (plus artifact keys) is summarized and prepended to subsequent steps' prompts by the engine. Steps are separate SDK sessions by default; `resumeSessionId` exists for retry-resume, not for cross-step continuity (a deliberate v1 simplification — revisit if prompt-cache economics or context continuity demand same-session phases). +**`priorContext`**: each completed step's final output (plus artifact keys) is summarized and prepended to subsequent steps' prompts by the engine. Steps are separate SDK sessions by default; `resumeSessionId` exists for crash-recovery resume (the engine carries a crashed attempt's session onto its recovery attempt), not for cross-step continuity (a deliberate v1 simplification — revisit if prompt-cache economics or context continuity demand same-session phases). ## Where Execution Runs & Sandboxing (M1 posture) -All executor work happens in the **worker container** (`apps/worker`), one run per worker slot: +All executor work happens in the **worker container** (`apps/worker`), one run per worker slot. Containment lives behind one **isolation seam** (`packages/executor-core/isolation.ts` — `evaluateToolCall`, `isWithin`, `buildScrubbedEnv`); the SDK adapter must route every tool decision and the subprocess environment through it rather than reimplementing checks inline (ADR-0009): - Each run gets a throwaway workspace `/work/runs/`, deleted after terminal state (configurable retention for debugging). -- Git credentials are injected per-run (credential helper scoped to the workspace) and scrubbed afterward; they never enter the agent's environment variables. -- Tool policy denies file writes outside the workspace and restricts Bash (no package-manager global installs; network egress list configurable). +- Git credentials are injected per-run into the clone URL and scrubbed from the remote immediately afterward; they never persist in `.git/config`. They are still passed as a clone argument today — moving to a workspace-scoped credential helper is follow-up work. +- Tool policy is enforced for **every** write-capable tool, Bash included, with a boundary-safe containment check (not a `startsWith` prefix): `readOnly` workspaces deny shell and confine writes to `.agrippa/artifacts`; `readWrite` workspaces confine writes to the workspace. The static layer cannot bound arbitrary writes a shell command makes in a read-write workspace — that is the OS sandbox's job (below). +- The agent subprocess runs with a **scrubbed environment** (`buildScrubbedEnv`): the master `AGRIPPA_SECRET_KEY` and datastore URLs are removed while the Anthropic auth vars the SDK needs are kept. The SDK `sandbox` (bubblewrap) is enabled where the host supports it, `strictMcpConfig` ignores repo `.mcp.json`, and the worker strips repo-supplied `.claude` settings/hooks at checkout so a checked-out repository can't inject hooks or permission overrides. The worker image runs as a non-root user. - MCP secrets resolve lazily at server spawn and are not logged; `run_events` payloads are scrubbed against known secret values before persistence. -**Explicitly deferred to M2**: per-run container/micro-VM isolation. The interface already localizes this change to the worker — the engine hands the executor a `workspaceDir` and a signal; whether that directory lives in the worker's filesystem or a jailed container is invisible above the interface. M1 is adequate for a trusted single org, not for hostile inputs; this is risk #2 in [00-overview](00-overview.md). +**Explicitly deferred**: per-run container/micro-VM isolation and a fully non-root, network-egress-restricted sandbox. The isolation seam localizes this — the engine hands the executor a `workspaceDir`, an `access` mode, and a signal; whether that directory lives in the worker's filesystem or a jailed container is invisible above the interface. The static containment plus env-scrub plus OS sandbox is adequate for a trusted org running semi-trusted repositories; hostile multi-tenant inputs need the container layer, which is risk #2 in [00-overview](00-overview.md). ## FakeExecutor — the Compliance Contract diff --git a/docs/design/04-execution-runtime.md b/docs/design/04-execution-runtime.md index 365c877..53ea9ec 100644 --- a/docs/design/04-execution-runtime.md +++ b/docs/design/04-execution-runtime.md @@ -6,17 +6,19 @@ How a submitted task becomes a finished run: queueing, the run state machine, re ## Submission (transactional) -`POST /projects/:id/tasks` validates params against the compiled template inputs, checks resource grants and quota headroom, then in **one Postgres transaction**: +`POST /projects/:id/tasks` validates params against the compiled template inputs, verifies each `repoRef` points at a repo connection **owned by the project**, checks resource grants and quota headroom, then in **one Postgres transaction**: 1. insert `tasks` row, -2. insert `runs` row (`status = queued`, pinned `template_version_id`, `params_snapshot`, frozen `model_resolution`, computed `budget`), +2. insert `runs` row (`status = queued`, pinned `template_version_id`, `params_snapshot`, frozen `model_resolution`, a pinned `resource_manifest` of the skills/MCP the run is authorized to use, computed `budget`), 3. enqueue pg-boss job `run.execute({runId})`. +The `resource_manifest` is the authorization boundary: required grants are enforced at submit and optional resources are included **only when granted**, so the worker resolves skills/MCP strictly from the manifest and never re-reads the mutable global registry — an ungranted optional resource is simply unavailable (see [ADR-0009](../adr/0009-security-correctness-deep-modules.md)). + Because pg-boss stores jobs in Postgres, there is no dual-write window: either the run and its job both exist, or neither does. This is the primary reason for pg-boss over a Redis-backed queue. ## Run State Machine -Pure function in `@agrippa/core` (`transition(state, event) → state | error`); every transition is persisted and audited. +Pure function in `@agrippa/core` (`transition(state, event) → state | error`); every transition is persisted and audited. The persist step is a **compare-and-swap** on the expected `from` status (`run-lifecycle.transitionRun`), so a late worker finalize can't overwrite a status another path (e.g. a concurrent cancel) already moved on from — the loser of the race simply doesn't write. ``` ┌────────────────────────────┐ @@ -69,7 +71,7 @@ finalize: usage_totals, workspace cleanup, terminal event Steps are the idempotency unit. On retry/resume, the engine loads `run_steps`, **skips succeeded steps**, and re-executes the first non-terminal step: -- If the executor supports resume and `executor_session_id` exists → resume that session. +- A step left `running` by a dead worker is marked `crashed`. A crash is an *interrupted* attempt, not a consumed retry: it adds one extra attempt (so even a no-retry step re-executes rather than being silently skipped), and the crashed attempt's `executor_session_id` is carried onto the recovery attempt so a resume-capable executor resumes that session. - Otherwise → restart the step as `attempt + 1` (templates must keep steps restart-safe; the workspace checkout is deterministic and `system` actions are idempotent). Budget correctness on resume: the `BudgetMeter` initializes from **persisted** `token_usage` totals, and usage rows are keyed by `(run_id, step_id, attempt)` — a partially-executed attempt's cost is counted, never double-counted. @@ -82,6 +84,8 @@ Approvals **do not hold a worker slot**. When a checkpoint is hit: `approvals` r - `rejected` → run → `failed` with `error.code = "approval_rejected"`. - expiry → per-template `onTimeout` (`cancel | reject | approve`). +Decisions are a compare-and-swap on `status = 'pending'` (`run-lifecycle.decideApproval`), so a user decision and the expiry worker can't overwrite each other. The decision is durable before the resume enqueue; if that enqueue is lost, the reconciliation sweeper re-enqueues any `waiting_approval` run whose approval is already decided, so a run can't be stranded. + ### Cancellation `POST /runs/:id/cancel` sets `runs.cancel_requested = true` and publishes on Redis channel `run:{id}:control`. The worker's control subscriber fires the run's `AbortController`; the executor aborts; the engine records `cancelled`. If no worker holds the run (queued / waiting_approval), the API transitions it directly and cancels the pending job. The DB flag backstops the pubsub message (worker checks it at step boundaries), so a lost message delays cancellation by at most one step. @@ -91,16 +95,17 @@ Approvals **do not hold a worker slot**. When a checkpoint is hit: `approvals` r Two independent layers, both enforced: - **Run budget** (template `budgets`): `BudgetMeter` accumulates `usage` events against run-level and per-phase `maxCostUsd`; breach → abort signal → `failed` with `budget_exceeded`. `maxDurationMinutes` → composed `AbortSignal.timeout` → `timed_out`. -- **Project quota** (`project_quotas`): checked at submit (reject with quota error) and re-checked at every step boundary; if `hard_stop` and exhausted mid-run → abort as `budget_exceeded` with quota provenance. Soft quotas surface warnings in the UI instead of aborting. +- **Project quota** (`project_quotas`): checked at submit (reject with quota error) and re-read from the database at every step boundary; if `hard_stop` and exhausted mid-run → abort as `budget_exceeded` with quota provenance. Submit and engine count the **same monthly window**, and the engine's headroom **excludes the run's own spend** (the meter already carries it, so including it would double-count on resume). Re-reading each step lets concurrent runs see each other's spend rather than each measuring only a stale start-of-run snapshot. Soft quotas surface warnings in the UI instead of aborting. ## Live Progress (SSE) -Ordering rule: the engine writes `run_events` **first** (assigning per-run monotonic `seq`), then publishes the same event to Redis `run:{id}:events`. +Ordering rule: the engine writes `run_events` **first** — the per-run monotonic `seq` is allocated by the database inside the INSERT (`run-lifecycle.appendRunEvent`), not from an in-memory `max+1` that a concurrent writer could collide with — then publishes the same event to Redis `run:{id}:events`. `GET /runs/:id/events` (SSE): -1. replay `run_events WHERE run_id = ? AND seq > :lastEventId ORDER BY seq` (from the `Last-Event-ID` header, or 0), -2. subscribe to the Redis channel, bridging live events (deduplicating by `seq` across the replay boundary), -3. emit each as `id: \nevent: \ndata: `. +1. **subscribe** to the Redis channel first, buffering live events, +2. replay `run_events WHERE run_id = ? AND seq > :lastEventId ORDER BY seq` (from the `Last-Event-ID` header, or 0), +3. flush the buffer, deduplicating by `seq` against the replay, +4. emit each as `id: \nevent: \ndata: `. -Client reconnects are therefore gap-free by construction; no polling anywhere. Redis here is a pure fan-out optimization — if Redis is briefly down, clients reconnect and replay from Postgres. +Subscribing **before** replaying is what makes reconnection gap-free by construction: an event committed and published in the window between replay and subscribe would otherwise be delivered only at the terminal replay. No polling anywhere (a bus-less deployment falls back to periodic DB replay). Redis here is a pure fan-out optimization — if Redis is briefly down, clients reconnect and replay from Postgres. diff --git a/docs/design/05-api-and-auth.md b/docs/design/05-api-and-auth.md index 8d5d313..d29e066 100644 --- a/docs/design/05-api-and-auth.md +++ b/docs/design/05-api-and-auth.md @@ -71,6 +71,8 @@ GET /runs/:id/approvals POST /runs/:id/approvals/:approvalId GET /runs/:id/artifacts GET /artifacts/:id/download ``` +Submission authorizes the resources a task references before persisting: a `repoRef` param must name a repo connection **owned by the project** (else `400 {code: "repo_not_in_project"}`), and the run's authorized skills/MCP are pinned into a resource manifest (see [04](04-execution-runtime.md) and [ADR-0009](../adr/0009-security-correctness-deep-modules.md)). Approval decisions are a compare-and-swap on the pending status: a decision that lost the race (already decided, or expired) returns `409 {code: "already_decided"}`. + ### Resource layer (org_admin writes; members read) ``` CRUD /fabri diff --git a/docs/design/09-testing-and-ci.md b/docs/design/09-testing-and-ci.md index 4fb072e..70144d6 100644 --- a/docs/design/09-testing-and-ci.md +++ b/docs/design/09-testing-and-ci.md @@ -21,7 +21,10 @@ Runs against `docker-compose.dev.yml` Postgres. Scenarios, each asserting both ` - happy path: all phases → `succeeded`, contract artifacts present, usage totals correct; - approval: pause (job completes, slot freed) → approve → resume; reject → `failed`; expire → per-template `onTimeout`; - budget: run-level and per-phase `maxCostUsd` abort; duration timeout → `timed_out`; project hard-stop quota mid-run; -- crash-resume: kill mid-step → retry skips succeeded steps → resumes/restarts correct attempt → **no double-counted usage**; +- crash-resume: kill mid-step → retry skips succeeded steps → resumes/restarts correct attempt → **no double-counted usage**; a **no-retry** step crash re-executes (not silently skipped) and resumes its session; +- run lifecycle: `transitionRun` compare-and-swap, database-allocated event seq (no collision under concurrent append), `decideApproval` CAS on `pending`; +- authorization: an ungranted optional MCP server is not resolved even when it exists in the registry; +- budget: run-level and per-phase `maxCostUsd` abort; duration timeout → `timed_out`; project hard-stop quota mid-run; resume does not double-count the run's own spend against the quota; - cancellation: mid-step abort latency, queued/waiting cancellation via API path; - `when:` false and unmet optional `requires:` → `skipped`; - required-artifact missing → `failed` with `contract_violation`. @@ -30,17 +33,21 @@ This suite doubles as the executor compliance spec (any future executor must pas ### API integration (Hono `app.request()` × real Postgres/Redis) -Auth flows, RBAC allow/deny matrix per role × endpoint class, transactional task submission (run + job atomicity), SSE replay from `Last-Event-ID`, grants gating submission, quota rejection at submit, audit rows on every mutation. +Auth flows, RBAC allow/deny matrix per role × endpoint class, transactional task submission (run + job atomicity), SSE replay from `Last-Event-ID` (deduped, strictly increasing seq), a cross-project `repoConnectionId` refused at submit, grants gating submission, quota rejection at submit, audit rows on every mutation. ### Claude executor -- Unit: mocked SDK `query()` asserting the full option-mapping table from [03](03-executor-abstraction.md) (subagents, skills materialization path, MCP config, tool policy, resume). +- Unit: mocked SDK `query()` asserting the full option-mapping table from [03](03-executor-abstraction.md) (subagents, skills materialization path, MCP config, resume); the isolation policy (`evaluateToolCall` — Bash and write containment, read-only enforcement, boundary-safe check) and env scrubbing; contract-scoped artifact collection (patch and uncontracted files skipped). - One live smoke test behind `ANTHROPIC_API_KEY`, excluded from CI, run manually before releases. +### Worker adapters + +- `DiskArtifactStore` path containment: normal file stored, escaping symlink rejected, missing source is not a zero-byte artifact. (The production workspace/resource adapters remain higher-risk untested surface — follow-up work.) + ### Frontend -- Component tests for `TaskParamsForm` (schema → widgets → zod validation, both locales) and the run timeline reducer (event stream → UI state). -- Full-browser E2E is deferred to M1.4 exit criteria as a manual scripted walkthrough (automating with Playwright is a stretch goal, not a gate). +- No automated frontend tests yet (the `TaskParamsForm` and run-timeline reducer component tests described in earlier plans are not implemented — a known gap). +- Full-browser E2E is deferred as a manual scripted walkthrough (automating with Playwright is a stretch goal, not a gate). ### Cross-cutting guards diff --git a/docs/manual/en/04-administration.md b/docs/manual/en/04-administration.md index e124ce2..a57ac50 100644 --- a/docs/manual/en/04-administration.md +++ b/docs/manual/en/04-administration.md @@ -21,7 +21,7 @@ The first account ever created is the **org admin**; everyone else signs up as * ## Project settings (Settings tab, project admins) - **Members** — add by email (the person must have an account), change roles, remove. A project always keeps at least one admin; the platform blocks demoting or removing the last one. -- **Resources** — the grant toggles per registry type. This is the gate: a template requirement that isn't granted here makes submission fail fast with a named error. +- **Resources** — the grant toggles per registry type. This is the gate: a template requirement that isn't granted here makes submission fail fast with a named error. **Optional** resources are also gated — an optional skill or MCP server that a template can use (for example the GitHub server behind an "open a PR" step) is only made available to the run when it's granted; without the grant that step is simply skipped rather than run with a shared credential. - **Repositories** — git remotes the project's runs may check out: URL, default branch, optional access token (write-only, encrypted; injected only during clone and scrubbed before agent code runs). - **Quota** — monthly cost (USD) and/or token ceilings with a **hard stop** switch. Hard-stop quotas reject new submissions once exhausted and abort in-flight runs at the next step boundary; soft quotas are informational. diff --git a/docs/manual/en/06-operations.md b/docs/manual/en/06-operations.md index 1b5e884..e75452b 100644 --- a/docs/manual/en/06-operations.md +++ b/docs/manual/en/06-operations.md @@ -53,6 +53,8 @@ Reverse proxy note: **disable response buffering** for `/api/v1/runs/*/events` ( | Live progress lags ~1 s, no push | `REDIS_URL` unset/unreachable — SSE falls back to DB polling. Harmless; restore Redis for instant updates. | | Submission rejected `skill_not_granted` / `mcp_not_granted` / `model_unresolvable` | Grant the resource under Project → Settings → Resources (models must cover the tiers the template requests). | | Submission rejected `quota_exhausted` | The project's hard-stop quota is spent this month — raise it, disable hard stop, or wait for the period. | +| Submission rejected `repo_not_in_project` | The `repoConnectionId` doesn't belong to this project — pick a repository registered under this project's Settings → Repositories. | +| An optional step (e.g. "open a PR") was skipped | Its optional resource isn't granted — grant the MCP server under Settings → Resources; ungranted optional resources are skipped, not run with a shared credential. | | Run failed `contract_violation` | The agent never produced a required artifact — inspect the step outputs; usually a prompt/instructions issue in the template. | | Checkout fails for a private repo | The repo connection's token is missing/expired — re-add it under Settings → Repositories (tokens are write-only; re-enter, don't "view"). | | Need to inspect what an agent actually did on disk | Set `AGRIPPA_KEEP_WORKSPACES=1` on the worker and re-run; workspaces persist under `WORKSPACE_ROOT/`. | diff --git a/docs/manual/zh-CN/04-administration.md b/docs/manual/zh-CN/04-administration.md index 9100810..8ad4c23 100644 --- a/docs/manual/zh-CN/04-administration.md +++ b/docs/manual/zh-CN/04-administration.md @@ -21,7 +21,7 @@ ## 项目设置(设置页签,项目管理员) - **成员** —— 按邮箱添加(对方需已注册)、调整角色、移除。项目必须至少保留一名管理员,平台会阻止降级或移除最后一名。 -- **资源授权** —— 按资源类型的授权开关。这里是闸门:模板需要而这里未授权的资源,会让提交快速失败并给出具名错误。 +- **资源授权** —— 按资源类型的授权开关。这里是闸门:模板需要而这里未授权的资源,会让提交快速失败并给出具名错误。**可选**资源同样受此限制——模板可用的可选技能或 MCP 服务(例如「提交 PR」步骤所依赖的 GitHub 服务),只有在授权后才会提供给执行使用;未授权时该步骤会被直接跳过,而不会用共享凭证去运行。 - **代码仓库** —— 项目执行可检出的 git 远端:地址、默认分支、可选访问令牌(只写、加密;仅在克隆时注入,智能体代码运行前即被清除)。 - **配额** —— 每月费用(美元)和/或 Token 上限,附**强制停止**开关。强制配额耗尽后拒绝新提交、并在下一个步骤边界中止进行中的执行;非强制配额仅作提示。 diff --git a/docs/manual/zh-CN/06-operations.md b/docs/manual/zh-CN/06-operations.md index c6284e2..88e3557 100644 --- a/docs/manual/zh-CN/06-operations.md +++ b/docs/manual/zh-CN/06-operations.md @@ -53,6 +53,8 @@ | 实时进度延迟约 1 秒、无推送 | `REDIS_URL` 未设置或不可达——SSE 退化为数据库轮询。无害;恢复 Redis 即恢复即时推送。 | | 提交被拒 `skill_not_granted` / `mcp_not_granted` / `model_unresolvable` | 到 项目 → 设置 → 资源授权 打开对应资源(模型需覆盖模板请求的档位)。 | | 提交被拒 `quota_exhausted` | 项目当月强制配额已用尽——上调、关闭强制停止或等待下一周期。 | +| 提交被拒 `repo_not_in_project` | 该 `repoConnectionId` 不属于本项目——请选择本项目「设置 → 代码仓库」中已注册的仓库。 | +| 某个可选步骤(如「提交 PR」)被跳过 | 其可选资源未授权——在「设置 → 资源授权」中授权对应 MCP 服务;未授权的可选资源会被跳过,而不会用共享凭证运行。 | | 执行失败 `contract_violation` | 智能体未产出某个必需产出物——查看各步骤输出,通常是模板指令的问题。 | | 私有仓库检出失败 | 仓库连接的令牌缺失或过期——到 设置 → 代码仓库 重新录入(令牌只写不读,重新填写即可)。 | | 想看智能体在磁盘上到底做了什么 | 给 worker 设置 `AGRIPPA_KEEP_WORKSPACES=1` 后重跑;工作区保留在 `WORKSPACE_ROOT/`。 | From 300df2cab4f135d6d4778a478fbbf122e8722a04 Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 17:06:43 +0800 Subject: [PATCH 10/18] =?UTF-8?q?fix:=20address=20review=20=E2=80=94=20ato?= =?UTF-8?q?mic=20approvals,=20safe=20sweeper,=20tighter=20lifecycle=20&=20?= =?UTF-8?q?artifact=20contract?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up to the code review on the hardening PR: - The stranded-approval sweeper matched *any* non-pending approval via an inner join, so a multi-checkpoint run with an earlier decided approval and a current pending one was re-enqueued every tick. Replace with a `not exists (… pending)` guard, extracted as run-lifecycle.findStrandedApprovalRuns with a regression test for the multi-approval case. - The approval decision, its `approval.decided` event, and the audit row now commit in one transaction (publish + enqueue after commit), so a partial write can't leave the timeline/audit missing a decision a retry would then skip. Adds a DbOrTx type so lifecycle/audit helpers can run inside a caller's transaction. - transitionRun no longer trusts a self-transition (`from === to`) blindly; it verifies the row is still in that status, and initialize now stops (rather than proceeding) when it loses the run-claim to another live worker, avoiding duplicate side effects. (A full lease for the running↔running overlap is noted as future work.) - Artifact collection matches the exact contracted filename (key + kind's extension), so a `json` artifact `report` is no longer satisfied by `report.md`. --- apps/api/src/lib/audit.ts | 4 +- apps/api/src/routes/execution.ts | 71 +++++++++++-------- apps/worker/src/index.ts | 10 +-- packages/db/src/client.ts | 10 +++ packages/executor-claude/src/executor.ts | 52 +++++++------- .../src/engine/engine.integration.test.ts | 31 +++++++- packages/orchestration/src/engine/engine.ts | 27 +++++-- .../orchestration/src/engine/run-lifecycle.ts | 35 +++++++-- 8 files changed, 161 insertions(+), 79 deletions(-) diff --git a/apps/api/src/lib/audit.ts b/apps/api/src/lib/audit.ts index 5adf8bf..ee731b3 100644 --- a/apps/api/src/lib/audit.ts +++ b/apps/api/src/lib/audit.ts @@ -1,4 +1,4 @@ -import { auditLogs, type Db } from "@agrippa/db"; +import { auditLogs, type DbOrTx } from "@agrippa/db"; import type { Context } from "hono"; import type { AppEnv } from "../context"; @@ -14,7 +14,7 @@ type AuditEntry = { * Every mutating handler records an audit row (docs/design/05-api-and-auth.md). * Accepts an explicit tx so creations can be atomic with their mutation. */ -export async function audit(c: Context, entry: AuditEntry, tx?: Db): Promise { +export async function audit(c: Context, entry: AuditEntry, tx?: DbOrTx): Promise { const db = tx ?? c.var.db; await db.insert(auditLogs).values({ orgId: c.var.user.orgId, diff --git a/apps/api/src/routes/execution.ts b/apps/api/src/routes/execution.ts index a60db76..db83909 100644 --- a/apps/api/src/routes/execution.ts +++ b/apps/api/src/routes/execution.ts @@ -49,17 +49,6 @@ async function loadRunScoped( return run; } -/** Appends an API-originated event to the run log (DB-allocated seq) and the bus. */ -async function appendRunEvent( - c: { var: AppEnv["Variables"] }, - runId: string, - type: string, - payload: Record, -): Promise { - const { seq, createdAt } = await allocateRunEvent(c.var.db, { runId, type, payload }); - await c.var.bus?.publish({ runId, seq, type, payload, createdAt: createdAt.toISOString() }); -} - export const executionRoutes = new Hono() // ── Submit ────────────────────────────────────────────────────────────────── .post( @@ -301,34 +290,56 @@ export const executionRoutes = new Hono() .from(approvals) .where(and(eq(approvals.id, c.req.param("approvalId")), eq(approvals.runId, run.id))); if (!approval) throw AppError.notFound("Approval"); - // compare-and-swap on status='pending' so a user decision and the expiry - // worker cannot overwrite each other - const updated = await decideApproval(c.var.db, approval.id, { - status: input.decision, - decidedBy: c.var.user.id, - comment: input.comment, + const eventPayload = { + approvalId: approval.id, + checkpointId: approval.checkpointId, + decision: input.decision, + }; + // decision (CAS on status='pending'), the approval.decided event, and the + // audit row commit together — a partial write can't leave the timeline or + // audit log missing the decision that a later retry would then skip + const result = await c.var.db.transaction(async (tx) => { + const updated = await decideApproval(tx, approval.id, { + status: input.decision, + decidedBy: c.var.user.id, + comment: input.comment, + }); + if (!updated) return { updated: null as null }; + const event = await allocateRunEvent(tx, { + runId: run.id, + type: "approval.decided", + payload: eventPayload, + }); + await audit( + c, + { + action: "run.approval.decide", + resourceType: "approval", + resourceId: approval.id, + projectId: run.projectId, + payload: { decision: input.decision }, + }, + tx, + ); + return { updated, event }; }); - if (!updated) { + if (!result.updated) { // already decided, OR a prior attempt decided then failed to enqueue: in // both cases the durable state is correct, so re-enqueue to unstick the // run (the sweeper also backstops this) and report the conflict await c.var.queue?.enqueueRun(run.id); throw AppError.conflict("already_decided", "Approval is already decided"); } - await appendRunEvent(c, run.id, "approval.decided", { - approvalId: approval.id, - checkpointId: approval.checkpointId, - decision: input.decision, - }); - await audit(c, { - action: "run.approval.decide", - resourceType: "approval", - resourceId: approval.id, - projectId: run.projectId, - payload: { decision: input.decision }, + // publish + enqueue only after the decision durably committed + await c.var.bus?.publish({ + runId: run.id, + seq: result.event.seq, + type: "approval.decided", + payload: eventPayload, + createdAt: result.event.createdAt.toISOString(), }); await c.var.queue?.enqueueRun(run.id); // resume at the gated phase - return c.json(updated); + return c.json(result.updated); }) // ── Artifacts ─────────────────────────────────────────────────────────────── diff --git a/apps/worker/src/index.ts b/apps/worker/src/index.ts index 9c50ba6..f290bff 100644 --- a/apps/worker/src/index.ts +++ b/apps/worker/src/index.ts @@ -13,10 +13,11 @@ import { durationToMinutes, type EngineDeps, executeRun, + findStrandedApprovalRuns, InProcessEventBus, RedisEventBus, } from "@agrippa/orchestration"; -import { and, eq, lt, ne, sql } from "drizzle-orm"; +import { and, eq, lt, sql } from "drizzle-orm"; import type { Job, JobWithMetadata } from "pg-boss"; import { DiskArtifactStore } from "./deps/artifacts"; import { DemoExecutor } from "./deps/demo-executor"; @@ -129,12 +130,7 @@ setInterval(async () => { // runs paused on an approval that has since been decided but whose resume // enqueue was lost (e.g. the API/worker died between the decision and the // send) — re-enqueue so the decision actually takes effect - const strandedApprovals = await db - .selectDistinct({ id: runs.id }) - .from(runs) - .innerJoin(approvals, eq(approvals.runId, runs.id)) - .where(and(eq(runs.status, "waiting_approval"), ne(approvals.status, "pending"))); - for (const run of strandedApprovals) await queue.enqueueRun(run.id); + for (const runId of await findStrandedApprovalRuns(db)) await queue.enqueueRun(runId); } catch (err) { deps.logger.warn("sweeper failed", { err: String(err) }); } diff --git a/packages/db/src/client.ts b/packages/db/src/client.ts index 8c10735..8ebdcd2 100644 --- a/packages/db/src/client.ts +++ b/packages/db/src/client.ts @@ -9,3 +9,13 @@ export function createDb(url: string | undefined = process.env.DATABASE_URL) { } export type Db = ReturnType; + +/** The transaction handle drizzle passes to a `db.transaction(async (tx) => …)` callback. */ +export type Transaction = Parameters[0]>[0]; + +/** + * Either a pooled connection or an open transaction — the query-capable surface + * shared by both. Helpers that must be able to run inside a caller's transaction + * accept this rather than `Db` (which additionally exposes `$client`). + */ +export type DbOrTx = Db | Transaction; diff --git a/packages/executor-claude/src/executor.ts b/packages/executor-claude/src/executor.ts index 5003d0d..d21ca73 100644 --- a/packages/executor-claude/src/executor.ts +++ b/packages/executor-claude/src/executor.ts @@ -1,4 +1,4 @@ -import { readdirSync } from "node:fs"; +import { existsSync } from "node:fs"; import path from "node:path"; import { buildScrubbedEnv, @@ -14,6 +14,14 @@ type QueryFn = (params: { prompt: string; options?: Options }) => AsyncIterable< const ARTIFACT_DIR = ".agrippa/artifacts"; +type ExpectedArtifact = StepExecutionRequest["expectedArtifacts"][number]; + +/** The exact filename a declared artifact must be written to (key + kind's extension). */ +function expectedFilename(a: ExpectedArtifact): string { + const ext = a.kind === "json" ? ".json" : a.kind === "markdown" ? ".md" : ""; + return `${a.key}${ext}`; +} + /** Prompt preamble telling the agent where declared artifacts must land. */ function artifactInstructions(expected: StepExecutionRequest["expectedArtifacts"]): string { // patch artifacts are generated by the engine from the git diff — the agent @@ -21,12 +29,7 @@ function artifactInstructions(expected: StepExecutionRequest["expectedArtifacts" // already produced) and stores the model's file instead of the real patch const authored = expected.filter((a) => a.kind !== "patch"); if (authored.length === 0) return ""; - const list = authored - .map( - (a) => - `- ${ARTIFACT_DIR}/${a.key}${a.kind === "json" ? ".json" : a.kind === "markdown" ? ".md" : ""}`, - ) - .join("\n"); + const list = authored.map((a) => `- ${ARTIFACT_DIR}/${expectedFilename(a)}`).join("\n"); return [ "", "---", @@ -135,32 +138,27 @@ function textOf(content: Array<{ type: string; text?: string }>): string { } /** - * Scans the artifact convention directory and emits one event per file that the - * step actually contracted to produce. Scoping to `expected` (a) validates the - * key/kind against the contract, (b) uses the declared kind rather than guessing - * from the extension, (c) skips patch artifacts (the engine generates those), - * and (d) ignores files left by earlier steps so a rescan can't re-emit them. + * Emits one event per declared artifact whose **exact** contracted file exists. + * Iterating the contract (not the directory) means a `json` artifact `report` + * is matched only by `report.json`, never `report.md`; patch artifacts are + * skipped (the engine generates those from the git diff); and files left by + * earlier steps or written under other names are ignored. */ function* collectArtifacts( workspaceDir: string, expected: StepExecutionRequest["expectedArtifacts"], ): Generator { - const kindByKey = new Map(expected.map((a) => [a.key, a.kind] as const)); const dir = path.join(workspaceDir, ARTIFACT_DIR); - let entries: string[] = []; - try { - entries = readdirSync(dir); - } catch { - return; - } - const emitted = new Set(); - for (const entry of entries) { - const ext = path.extname(entry); - const key = ext ? entry.slice(0, -ext.length) : entry; - const kind = kindByKey.get(key); - if (kind === undefined || kind === "patch" || emitted.has(key)) continue; - emitted.add(key); - yield { type: "artifact", key, kind, path: path.join(ARTIFACT_DIR, entry) }; + for (const a of expected) { + if (a.kind === "patch") continue; + const filename = expectedFilename(a); + if (!existsSync(path.join(dir, filename))) continue; + yield { + type: "artifact", + key: a.key, + kind: a.kind, + path: path.join(ARTIFACT_DIR, filename), + }; } } diff --git a/packages/orchestration/src/engine/engine.integration.test.ts b/packages/orchestration/src/engine/engine.integration.test.ts index 1cdbc28..dd6a821 100644 --- a/packages/orchestration/src/engine/engine.integration.test.ts +++ b/packages/orchestration/src/engine/engine.integration.test.ts @@ -35,7 +35,12 @@ import { InMemoryArtifactStore, silentLogger, } from "./fakes"; -import { appendRunEvent, decideApproval, transitionRun } from "./run-lifecycle"; +import { + appendRunEvent, + decideApproval, + findStrandedApprovalRuns, + transitionRun, +} from "./run-lifecycle"; const TEST_DATABASE_URL = process.env.TEST_DATABASE_URL ?? "postgres://localhost:5432/agrippa_test"; const TEMPLATES_DIR = path.resolve(import.meta.dirname, "../../../../templates"); @@ -628,6 +633,30 @@ describe.skipIf(!dbUp)("run-lifecycle module", () => { expect(b.seq).toBeGreaterThan(a.seq); }); + it("findStrandedApprovalRuns selects only runs with no pending approval", async () => { + const { db, runId } = await setupFixture(); + await db.update(runs).set({ status: "waiting_approval" }).where(eq(runs.id, runId)); + + // one pending approval → not stranded + const [a1] = await db + .insert(approvals) + .values({ runId, checkpointId: "cp-1", status: "pending" }) + .returning(); + expect(await findStrandedApprovalRuns(db)).not.toContain(runId); + + // a second, earlier checkpoint gets approved while cp-1 is still pending → + // still not stranded (the multi-approval trap the old innerJoin fell into) + await db + .insert(approvals) + .values({ runId, checkpointId: "cp-0", status: "approved" }) + .returning(); + expect(await findStrandedApprovalRuns(db)).not.toContain(runId); + + // cp-1 decided too → now every approval is decided → stranded, re-enqueue + await decideApproval(db, a1?.id as string, { status: "approved" }); + expect(await findStrandedApprovalRuns(db)).toContain(runId); + }); + it("decideApproval is a compare-and-swap on pending", async () => { const { db, runId } = await setupFixture(); const [approval] = await db diff --git a/packages/orchestration/src/engine/engine.ts b/packages/orchestration/src/engine/engine.ts index b4937bb..7824028 100644 --- a/packages/orchestration/src/engine/engine.ts +++ b/packages/orchestration/src/engine/engine.ts @@ -51,6 +51,18 @@ class RunFailure extends Error { } } +/** + * Raised when this worker loses the run-claim to another still-active worker. + * We must not finalize or execute — the owner is running the run — so execute() + * catches it and exits without touching the run. + */ +class RunClaimLost extends Error { + constructor() { + super("run is owned by another worker"); + this.name = "RunClaimLost"; + } +} + /** * Executes (or resumes) one run to its next stopping point: a terminal state * or a waiting_approval pause. Steps are the idempotency unit — on resume, @@ -133,6 +145,8 @@ class RunEngine { const outcome = await this.runPhases(); return outcome; } catch (err) { + // another worker owns the run — leave it entirely to them, finalize nothing + if (err instanceof RunClaimLost) return "already_terminal"; return await this.handleFailure(err); } finally { if (this.timeoutTimer) clearTimeout(this.timeoutTimer); @@ -157,14 +171,13 @@ class RunEngine { .select({ status: runs.status }) .from(runs) .where(eq(runs.id, run.id)); - if (!current || isTerminalRunStatus(current.status)) { - throw new RunFailure( - current?.status === "cancelled" ? "cancelled" : "superseded", - "run already finalized by another worker", - current?.status === "cancelled" ? "cancelled" : "failed", - ); + if (current && current.status === "cancelled") { + throw new RunFailure("cancelled", "run cancelled", "cancelled"); } - this.run.status = current.status; + // it's terminal (another worker finished it) or already `running` under + // another live worker — either way this worker must not proceed and + // duplicate side effects; the owner (or a later re-delivery) drives it + throw new RunClaimLost(); } if (run.startedAt === null) { await db.update(runs).set({ startedAt: new Date() }).where(eq(runs.id, run.id)); diff --git a/packages/orchestration/src/engine/run-lifecycle.ts b/packages/orchestration/src/engine/run-lifecycle.ts index 7ff7f52..9dec179 100644 --- a/packages/orchestration/src/engine/run-lifecycle.ts +++ b/packages/orchestration/src/engine/run-lifecycle.ts @@ -1,5 +1,5 @@ import { canTransitionRun, type RunStatus } from "@agrippa/core"; -import { approvals, type Db, runEvents, runs } from "@agrippa/db"; +import { approvals, type DbOrTx, runEvents, runs } from "@agrippa/db"; import { and, eq, sql } from "drizzle-orm"; /** @@ -33,12 +33,17 @@ function isUniqueViolation(err: unknown): boolean { * moved on (e.g. a cancel landed first). Rejects illegal transitions up front. */ export async function transitionRun( - db: Db, + db: DbOrTx, runId: string, from: RunStatus, to: RunStatus, ): Promise { - if (from === to) return true; + if (from === to) { + // a self-transition is an assertion "the run is still in `from`" — verify + // against the database rather than trusting a possibly-stale caller value + const [row] = await db.select({ status: runs.status }).from(runs).where(eq(runs.id, runId)); + return row?.status === to; + } if (!canTransitionRun(from, to)) { throw new Error(`illegal run transition ${from} → ${to}`); } @@ -56,7 +61,7 @@ export async function transitionRun( * correct; the unique (run_id, seq) index backstops the rare concurrent race, * on which we retry rather than fail the job. */ -export async function appendRunEvent(db: Db, event: RunEventInput): Promise { +export async function appendRunEvent(db: DbOrTx, event: RunEventInput): Promise { const nextSeq = sql`(select coalesce(max(${runEvents.seq}), 0) + 1 from ${runEvents} where ${runEvents.runId} = ${event.runId})`; for (let attempt = 0; attempt < 5; attempt++) { try { @@ -80,13 +85,33 @@ export async function appendRunEvent(db: Db, event: RunEventInput): Promise { + const rows = await db + .select({ id: runs.id }) + .from(runs) + .where( + and( + eq(runs.status, "waiting_approval"), + sql`not exists (select 1 from ${approvals} where ${approvals.runId} = ${runs.id} and ${approvals.status} = 'pending')`, + ), + ); + return rows.map((r) => r.id); +} + /** * Decide a pending approval atomically. The `status = 'pending'` predicate makes * this a compare-and-swap: a user decision and the expiry worker can't overwrite * each other. Returns the updated row, or null if it was no longer pending. */ export async function decideApproval( - db: Db, + db: DbOrTx, approvalId: string, patch: { status: "approved" | "rejected" | "expired"; decidedBy?: string; comment?: string }, ): Promise { From 28c319e0f1b9f69e29efc1802257e88337bd9e96 Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 17:11:30 +0800 Subject: [PATCH 11/18] =?UTF-8?q?fix(security):=20address=20review=20?= =?UTF-8?q?=E2=80=94=20symlink=20writes,=20env=20allowlist,=20/app=20owner?= =?UTF-8?q?ship,=20artifact=20events?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Second round of review follow-ups, all security-tagged: - Write containment was purely lexical, so a checked-out symlink component (workspace/link -> /app) let a Write escape. The executor now canonicalizes the real write target (realWriteContained) and denies a write that resolves outside the workspace; fail-closed on resolution errors. - The env scrubber exempted the whole ANTHROPIC_/CLAUDE_ namespace, so a stray ANTHROPIC_PRIVATE_KEY or CLAUDE_ADMIN_TOKEN would ride along. Replace the namespace exemption with an explicit allowlist of the SDK auth variables and apply the secret heuristic to everything else. - The worker Dockerfile chowned /app to bun, letting agent commands modify worker code/deps/templates and persist into later runs. Keep /app root-owned; only /work (runtime storage) is writable. Also install bubblewrap so the SDK command sandbox actually engages in the container instead of silently degrading. - Uncontracted artifact events were emitted to run_events/SSE before the contract check dropped the store, leaking their inline contents. Validate the contract before emit, and emit the normalized contract kind without the inline body (the content lives in the artifacts table). --- infra/Dockerfile.worker | 14 ++-- packages/executor-claude/src/executor.ts | 23 ++++-- packages/executor-core/src/isolation.test.ts | 36 +++++++++- packages/executor-core/src/isolation.ts | 73 +++++++++++++++++--- packages/orchestration/src/engine/engine.ts | 41 +++++++---- 5 files changed, 151 insertions(+), 36 deletions(-) diff --git a/infra/Dockerfile.worker b/infra/Dockerfile.worker index fc90d08..9b9c85e 100644 --- a/infra/Dockerfile.worker +++ b/infra/Dockerfile.worker @@ -2,9 +2,10 @@ FROM oven/bun:1.3 WORKDIR /app -# git for run workspaces (clone/diff); ripgrep helps agent search +# git for run workspaces (clone/diff); ripgrep helps agent search; bubblewrap +# backs the SDK command sandbox (without it, sandboxing silently degrades) RUN apt-get update \ - && apt-get install -y --no-install-recommends git ripgrep ca-certificates \ + && apt-get install -y --no-install-recommends git ripgrep ca-certificates bubblewrap \ && rm -rf /var/lib/apt/lists/* COPY package.json bun.lock bunfig.toml tsconfig.base.json tsconfig.json ./ @@ -20,8 +21,11 @@ ENV WORKSPACE_ROOT=/work/runs ENV ARTIFACT_STORAGE_ROOT=/work/artifacts # Run agent code as a non-root user so a shell tool call cannot reach the -# container filesystem or other roots (docs/design/03 §Sandboxing). Chowning -# /work here makes a freshly-created named volume inherit `bun` ownership. -RUN mkdir -p /work/runs /work/artifacts && chown -R bun:bun /app /work +# container filesystem or other roots (docs/design/03 §Sandboxing). Only /work +# (runtime storage) is made writable — /app stays root-owned so agent commands +# can't modify worker code, dependencies, or templates and persist into later +# runs. Chowning /work here makes a freshly-created named volume inherit `bun` +# ownership. +RUN mkdir -p /work/runs /work/artifacts && chown -R bun:bun /work USER bun CMD ["bun", "apps/worker/src/index.ts"] diff --git a/packages/executor-claude/src/executor.ts b/packages/executor-claude/src/executor.ts index d21ca73..a74831d 100644 --- a/packages/executor-claude/src/executor.ts +++ b/packages/executor-claude/src/executor.ts @@ -6,7 +6,10 @@ import { type Executor, type ExecutorEvent, evaluateToolCall, + isWriteTool, + realWriteContained, type StepExecutionRequest, + writeTargetOf, } from "@agrippa/executor-core"; import { type Options, type SDKMessage, query as sdkQuery } from "@anthropic-ai/claude-agent-sdk"; @@ -110,13 +113,21 @@ export function buildQueryArgs( abortController, permissionMode: "acceptEdits", canUseTool: async (toolName, input) => { - const decision = evaluateToolCall( - req.toolPolicy, - req.workspaceDir, - toolName, - input as Record, - ); + const record = input as Record; + const decision = evaluateToolCall(req.toolPolicy, req.workspaceDir, toolName, record); if (decision.behavior === "deny") return decision; + // the lexical check above can't see through symlinks — verify the real + // write target stays inside the workspace before allowing a write tool + const target = writeTargetOf(record); + if (isWriteTool(toolName) && target !== undefined) { + const resolved = path.resolve(req.workspaceDir, target); + if (!(await realWriteContained(req.toolPolicy.writeRoot, resolved))) { + return { + behavior: "deny", + message: `write resolves (via a symlink) outside the run workspace (${target})`, + }; + } + } return { behavior: "allow", updatedInput: input }; }, }; diff --git a/packages/executor-core/src/isolation.test.ts b/packages/executor-core/src/isolation.test.ts index e199348..8324ea6 100644 --- a/packages/executor-core/src/isolation.test.ts +++ b/packages/executor-core/src/isolation.test.ts @@ -1,6 +1,9 @@ import { describe, expect, it } from "bun:test"; +import { mkdtempSync, rmSync, symlinkSync } from "node:fs"; +import { mkdir } from "node:fs/promises"; +import { tmpdir } from "node:os"; import path from "node:path"; -import { buildScrubbedEnv, evaluateToolCall, isWithin } from "./isolation"; +import { buildScrubbedEnv, evaluateToolCall, isWithin, realWriteContained } from "./isolation"; const ROOT = "/work/runs/run-1"; const rw = { access: "readWrite" as const, writeRoot: ROOT }; @@ -50,26 +53,55 @@ describe("evaluateToolCall — read-only workspace", () => { }); describe("buildScrubbedEnv", () => { - it("drops platform secrets but keeps Anthropic auth and system vars", () => { + it("drops platform secrets but keeps allow-listed SDK auth and system vars", () => { const env = buildScrubbedEnv({ PATH: "/usr/bin", HOME: "/home/bun", ANTHROPIC_API_KEY: "sk-ant-xxx", + ANTHROPIC_BASE_URL: "https://api.anthropic.com", AGRIPPA_SECRET_KEY: "master", DATABASE_URL: "postgres://secret", BETTER_AUTH_SECRET: "s", REDIS_URL: "redis://x", GITHUB_TOKEN: "ghp_x", SOME_PASSWORD: "p", + // namespaced secrets must NOT ride along just because of their prefix + ANTHROPIC_PRIVATE_KEY: "leak", + CLAUDE_ADMIN_TOKEN: "leak", }); expect(env.PATH).toBe("/usr/bin"); expect(env.HOME).toBe("/home/bun"); expect(env.ANTHROPIC_API_KEY).toBe("sk-ant-xxx"); + expect(env.ANTHROPIC_BASE_URL).toBe("https://api.anthropic.com"); expect(env.AGRIPPA_SECRET_KEY).toBeUndefined(); expect(env.DATABASE_URL).toBeUndefined(); expect(env.BETTER_AUTH_SECRET).toBeUndefined(); expect(env.REDIS_URL).toBeUndefined(); expect(env.GITHUB_TOKEN).toBeUndefined(); expect(env.SOME_PASSWORD).toBeUndefined(); + expect(env.ANTHROPIC_PRIVATE_KEY).toBeUndefined(); + expect(env.CLAUDE_ADMIN_TOKEN).toBeUndefined(); + }); +}); + +describe("realWriteContained", () => { + const dirs: string[] = []; + const ws = () => { + const d = mkdtempSync(path.join(tmpdir(), "iso-ws-")); + dirs.push(d); + return d; + }; + + it("allows a new file in the workspace and rejects a symlink escape", async () => { + const root = ws(); + await mkdir(path.join(root, "src"), { recursive: true }); + // a not-yet-existing file under a real dir is contained + expect(await realWriteContained(root, path.join(root, "src/new.ts"))).toBe(true); + // a symlinked directory pointing outside defeats the lexical check + const outside = ws(); + symlinkSync(outside, path.join(root, "escape")); + expect(await realWriteContained(root, path.join(root, "escape/x.ts"))).toBe(false); + + for (const d of dirs.splice(0)) rmSync(d, { recursive: true, force: true }); }); }); diff --git a/packages/executor-core/src/isolation.ts b/packages/executor-core/src/isolation.ts index 9017aa9..3f50534 100644 --- a/packages/executor-core/src/isolation.ts +++ b/packages/executor-core/src/isolation.ts @@ -1,3 +1,4 @@ +import { realpath } from "node:fs/promises"; import path from "node:path"; /** @@ -25,6 +26,16 @@ const WRITE_TOOLS = new Set(["Write", "Edit", "MultiEdit", "NotebookEdit"]); /** Tools that execute arbitrary commands (uncontainable by static rules). */ const EXEC_TOOLS = new Set(["Bash", "BashOutput", "KillShell", "KillBash"]); +/** Whether a tool writes to the filesystem through a path argument. */ +export function isWriteTool(toolName: string): boolean { + return WRITE_TOOLS.has(toolName); +} + +/** The path argument a write tool targets, if any. */ +export function writeTargetOf(input: Record): string | undefined { + return (input.file_path ?? input.path ?? input.notebook_path) as string | undefined; +} + export type ToolDecision = { behavior: "allow" } | { behavior: "deny"; message: string }; /** @@ -65,7 +76,7 @@ export function evaluateToolCall( } if (WRITE_TOOLS.has(toolName)) { - const target = (input.file_path ?? input.path ?? input.notebook_path) as string | undefined; + const target = writeTargetOf(input); if (target === undefined) return { behavior: "allow" }; const resolved = path.resolve(workspaceDir, target); if (!isWithin(writeRoot, resolved)) { @@ -89,10 +100,49 @@ export function evaluateToolCall( } /** - * Environment variables that must never reach the agent subprocess: leaking - * `AGRIPPA_SECRET_KEY` decrypts every stored credential, and the datastore - * URLs grant direct access to run/tenant data. The Anthropic/Claude vars the - * SDK needs to authenticate are preserved. + * Symlink-safe containment for a write target. `evaluateToolCall` is purely + * lexical, so a symlink component (e.g. `workspace/link -> /app`) would slip a + * write past it; this canonicalizes the nearest existing ancestor of the target + * (the file itself may not exist yet) and confirms it stays inside `writeRoot`. + * Fail-closed on any resolution error. + */ +export async function realWriteContained(writeRoot: string, target: string): Promise { + let root: string; + try { + root = await realpath(path.resolve(writeRoot)); + } catch { + return false; + } + let dir = path.resolve(target); + for (;;) { + try { + const real = await realpath(dir); + return real === root || real.startsWith(root + path.sep); + } catch { + const parent = path.dirname(dir); + if (parent === dir) return false; // reached filesystem root without a hit + dir = parent; + } + } +} + +/** + * SDK/CLI authentication variables the agent subprocess legitimately needs. + * Everything else that looks like a secret is dropped, even in the Anthropic / + * Claude namespaces — so an admin/private variable can't ride along. + */ +const SDK_AUTH_ALLOW = new Set([ + "ANTHROPIC_API_KEY", + "ANTHROPIC_AUTH_TOKEN", + "ANTHROPIC_BASE_URL", + "ANTHROPIC_MODEL", + "CLAUDE_CODE_OAUTH_TOKEN", +]); + +/** + * Platform secrets that must never reach the agent subprocess: leaking + * `AGRIPPA_SECRET_KEY` decrypts every stored credential, and the datastore URLs + * grant direct access to run/tenant data. */ const SECRET_ENV_KEYS = new Set([ "AGRIPPA_SECRET_KEY", @@ -102,17 +152,18 @@ const SECRET_ENV_KEYS = new Set([ "REDIS_URL", ]); -/** Heuristic secret-name match, minus the Anthropic/Claude auth vars we keep. */ +/** Heuristic secret-name match (applied to every non-allowlisted variable). */ function looksSecret(key: string): boolean { - if (/^(ANTHROPIC_|CLAUDE_)/.test(key)) return false; - return /(SECRET|PASSWORD|PRIVATE_KEY|_TOKEN$|_KEY$)/i.test(key); + return /(SECRET|PASSWORD|PRIVATE_KEY|CREDENTIAL|_TOKEN$|_KEY$)/i.test(key); } /** * Build the subprocess environment for the agent, dropping platform secrets. * The SDK's `env` option REPLACES the child environment wholesale, so we start * from the worker env and remove what the agent must not see, rather than - * allow-listing (which would starve the CLI of PATH/HOME/locale it needs). + * allow-listing (which would starve the CLI of PATH/HOME/locale it needs). The + * SDK auth variables are kept via an explicit allowlist so the secret heuristic + * doesn't drop them. */ export function buildScrubbedEnv( source: Record = process.env, @@ -120,6 +171,10 @@ export function buildScrubbedEnv( const out: Record = {}; for (const [key, value] of Object.entries(source)) { if (value === undefined) continue; + if (SDK_AUTH_ALLOW.has(key)) { + out[key] = value; + continue; + } if (SECRET_ENV_KEYS.has(key) || looksSecret(key)) continue; out[key] = value; } diff --git a/packages/orchestration/src/engine/engine.ts b/packages/orchestration/src/engine/engine.ts index 7824028..d9da070 100644 --- a/packages/orchestration/src/engine/engine.ts +++ b/packages/orchestration/src/engine/engine.ts @@ -672,17 +672,10 @@ class RunEngine { row: StepRow, event: ExecutorEvent, ): Promise { - const { type, ...payload } = event as { type: string } & Record; - await this.emit(type, { phaseId: phase.id, stepId: step.id, ...payload }, row.id); - if (event.type === "step.started" && event.sessionId) { - await this.db - .update(runSteps) - .set({ executorSessionId: event.sessionId }) - .where(eq(runSteps.id, row.id)); - } if (event.type === "artifact") { - // only accept artifacts the step contracted to produce — a non-compliant - // executor cannot smuggle in uncontracted keys or a mismatched kind + // validate the contract BEFORE emitting: an uncontracted artifact's inline + // contents/path/key must not leak into run_events or the SSE stream. Emit + // the normalized contract kind so downstream sees the declared type. const produces = "produces" in step ? step.produces : []; if (!produces.includes(event.key)) { this.deps.logger.warn("dropping uncontracted artifact", { @@ -692,10 +685,30 @@ class RunEngine { }); return; } - const contractKind = this.template.spec.outputs.artifacts.find( - (a) => a.key === event.key, - )?.kind; - await this.storeArtifact(row, { ...event, kind: contractKind ?? event.kind }); + const contractKind = + this.template.spec.outputs.artifacts.find((a) => a.key === event.key)?.kind ?? event.kind; + await this.emit( + "artifact", + { + phaseId: phase.id, + stepId: step.id, + key: event.key, + kind: contractKind, + path: event.path, + }, + row.id, + ); + await this.storeArtifact(row, { ...event, kind: contractKind }); + return; + } + + const { type, ...payload } = event as { type: string } & Record; + await this.emit(type, { phaseId: phase.id, stepId: step.id, ...payload }, row.id); + if (event.type === "step.started" && event.sessionId) { + await this.db + .update(runSteps) + .set({ executorSessionId: event.sessionId }) + .where(eq(runSteps.id, row.id)); } } From 775db9d4a6d593dd72655665b8eb27c6ed468739 Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 17:13:03 +0800 Subject: [PATCH 12/18] docs: correct the enqueue window, SSE polling, and optional-resource wording MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review follow-ups on the docs: - design/04: the pg-boss job is enqueued after the task+run transaction commits, not inside it — describe the mitigated post-commit dual-write window (repaired by the sweeper) instead of claiming there is none. - design/04: drop the "No polling anywhere" line that contradicted the bus-less DB-polling fallback; state that Redis is optional and polling preserves correctness. - manual (both locales): an ungranted optional resource is withheld, not resolved with a shared credential; a step that requires it is skipped, one that merely could use it runs without it. --- docs/design/04-execution-runtime.md | 9 ++++----- docs/manual/en/04-administration.md | 2 +- docs/manual/zh-CN/04-administration.md | 2 +- 3 files changed, 6 insertions(+), 7 deletions(-) diff --git a/docs/design/04-execution-runtime.md b/docs/design/04-execution-runtime.md index 53ea9ec..3527275 100644 --- a/docs/design/04-execution-runtime.md +++ b/docs/design/04-execution-runtime.md @@ -9,12 +9,11 @@ How a submitted task becomes a finished run: queueing, the run state machine, re `POST /projects/:id/tasks` validates params against the compiled template inputs, verifies each `repoRef` points at a repo connection **owned by the project**, checks resource grants and quota headroom, then in **one Postgres transaction**: 1. insert `tasks` row, -2. insert `runs` row (`status = queued`, pinned `template_version_id`, `params_snapshot`, frozen `model_resolution`, a pinned `resource_manifest` of the skills/MCP the run is authorized to use, computed `budget`), -3. enqueue pg-boss job `run.execute({runId})`. +2. insert `runs` row (`status = queued`, pinned `template_version_id`, `params_snapshot`, frozen `model_resolution`, a pinned `resource_manifest` of the skills/MCP the run is authorized to use, computed `budget`). -The `resource_manifest` is the authorization boundary: required grants are enforced at submit and optional resources are included **only when granted**, so the worker resolves skills/MCP strictly from the manifest and never re-reads the mutable global registry — an ungranted optional resource is simply unavailable (see [ADR-0009](../adr/0009-security-correctness-deep-modules.md)). +After the transaction commits, the handler enqueues the pg-boss job `run.execute({runId})`. The `resource_manifest` is the authorization boundary: required grants are enforced at submit and optional resources are included **only when granted**, so the worker resolves skills/MCP strictly from the manifest and never re-reads the mutable global registry — an ungranted optional resource is simply unavailable (see [ADR-0009](../adr/0009-security-correctness-deep-modules.md)). -Because pg-boss stores jobs in Postgres, there is no dual-write window: either the run and its job both exist, or neither does. This is the primary reason for pg-boss over a Redis-backed queue. +The enqueue is a post-commit send, so a narrow dual-write window exists (a crash between commit and send would leave a `queued` run with no job). It is mitigated, not eliminated: the worker's reconciliation sweeper re-enqueues `queued` runs older than 30 s. pg-boss stores jobs in Postgres, so once the send lands the job is durable — the primary reason for pg-boss over a Redis-backed queue. ## Run State Machine @@ -108,4 +107,4 @@ Ordering rule: the engine writes `run_events` **first** — the per-run monotoni 3. flush the buffer, deduplicating by `seq` against the replay, 4. emit each as `id: \nevent: \ndata: `. -Subscribing **before** replaying is what makes reconnection gap-free by construction: an event committed and published in the window between replay and subscribe would otherwise be delivered only at the terminal replay. No polling anywhere (a bus-less deployment falls back to periodic DB replay). Redis here is a pure fan-out optimization — if Redis is briefly down, clients reconnect and replay from Postgres. +Subscribing **before** replaying is what makes reconnection gap-free by construction: an event committed and published in the window between replay and subscribe would otherwise be delivered only at the terminal replay. Redis is optional: with a bus, live events push instantly; without one, the stream falls back to periodic DB replay, which preserves correctness (just with a small latency). Either way, if Redis is briefly down, clients reconnect and replay from Postgres. diff --git a/docs/manual/en/04-administration.md b/docs/manual/en/04-administration.md index a57ac50..f656825 100644 --- a/docs/manual/en/04-administration.md +++ b/docs/manual/en/04-administration.md @@ -21,7 +21,7 @@ The first account ever created is the **org admin**; everyone else signs up as * ## Project settings (Settings tab, project admins) - **Members** — add by email (the person must have an account), change roles, remove. A project always keeps at least one admin; the platform blocks demoting or removing the last one. -- **Resources** — the grant toggles per registry type. This is the gate: a template requirement that isn't granted here makes submission fail fast with a named error. **Optional** resources are also gated — an optional skill or MCP server that a template can use (for example the GitHub server behind an "open a PR" step) is only made available to the run when it's granted; without the grant that step is simply skipped rather than run with a shared credential. +- **Resources** — the grant toggles per registry type. This is the gate: a template requirement that isn't granted here makes submission fail fast with a named error. **Optional** resources are also gated — an optional skill or MCP server (for example the GitHub server behind an "open a PR" step) is withheld from the run unless it's granted, never resolved with a shared credential. A step that explicitly requires that resource is then skipped; a step that merely could use it runs without it. - **Repositories** — git remotes the project's runs may check out: URL, default branch, optional access token (write-only, encrypted; injected only during clone and scrubbed before agent code runs). - **Quota** — monthly cost (USD) and/or token ceilings with a **hard stop** switch. Hard-stop quotas reject new submissions once exhausted and abort in-flight runs at the next step boundary; soft quotas are informational. diff --git a/docs/manual/zh-CN/04-administration.md b/docs/manual/zh-CN/04-administration.md index 8ad4c23..855df2a 100644 --- a/docs/manual/zh-CN/04-administration.md +++ b/docs/manual/zh-CN/04-administration.md @@ -21,7 +21,7 @@ ## 项目设置(设置页签,项目管理员) - **成员** —— 按邮箱添加(对方需已注册)、调整角色、移除。项目必须至少保留一名管理员,平台会阻止降级或移除最后一名。 -- **资源授权** —— 按资源类型的授权开关。这里是闸门:模板需要而这里未授权的资源,会让提交快速失败并给出具名错误。**可选**资源同样受此限制——模板可用的可选技能或 MCP 服务(例如「提交 PR」步骤所依赖的 GitHub 服务),只有在授权后才会提供给执行使用;未授权时该步骤会被直接跳过,而不会用共享凭证去运行。 +- **资源授权** —— 按资源类型的授权开关。这里是闸门:模板需要而这里未授权的资源,会让提交快速失败并给出具名错误。**可选**资源同样受此限制——可选的技能或 MCP 服务(例如「提交 PR」步骤所依赖的 GitHub 服务),未授权时不会提供给执行,也绝不会用共享凭证去解析。此时:明确要求该资源的步骤会被跳过,而仅仅「可以用到」该资源的步骤则会在没有它的情况下继续运行。 - **代码仓库** —— 项目执行可检出的 git 远端:地址、默认分支、可选访问令牌(只写、加密;仅在克隆时注入,智能体代码运行前即被清除)。 - **配额** —— 每月费用(美元)和/或 Token 上限,附**强制停止**开关。强制配额耗尽后拒绝新提交、并在下一个步骤边界中止进行中的执行;非强制配额仅作提示。 From f8985a22aabe1cfe2d13532a170c69bc99992379 Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 18:06:10 +0800 Subject: [PATCH 13/18] fix(security): confine reads + redact event secrets; make seq/finalize atomic MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up review (round 3) — the security & lifecycle enforcement half. Isolation seam (S1): - The tool policy only gated writes and shell, so Read/Grep/Glob could open any absolute path — /proc/self/environ (the kept ANTHROPIC_API_KEY), another run's /work/runs/, the shared artifact store. evaluateToolCall now confines the read tools to the workspace too (with the same symlink-real check as writes); a read with no path still defaults to the workspace cwd. - Event payloads were persisted and streamed verbatim, so a secret the agent echoed reached run_events/SSE. A SecretRedactor (built from the env secret values + the run's resolved MCP tokens) now scrubs every event in emit. This makes the redaction the design doc already claimed real. Lifecycle seam (S2): - Event seq was max(seq)+1 with retry; inside the approval transaction the first unique violation aborted the tx, so the retry could never recover. Replace it with an atomic per-run counter (runs.next_event_seq, migration 0003 backfilled to max(seq)) allocated via UPDATE … RETURNING — collision-free and tx-safe. - finalize now commits the status CAS + finishedAt/totals + terminal event in one transaction and publishes to the bus post-commit, so a crash can't leave an unrepairable half-finalized run; it also re-checks cancelRequested so a late cancel wins over a success. - markRunFailed (retry-exhaustion) routes through transitionRun instead of an id-only update, so it can't clobber a concurrent transition. --- apps/worker/src/index.ts | 23 +- .../db/drizzle/0003_run-next-event-seq.sql | 6 + packages/db/drizzle/meta/0003_snapshot.json | 3075 +++++++++++++++++ packages/db/drizzle/meta/_journal.json | 7 + packages/db/src/schema/runs.ts | 2 + packages/executor-claude/src/executor.test.ts | 8 + packages/executor-claude/src/executor.ts | 15 +- packages/executor-core/src/isolation.test.ts | 46 +- packages/executor-core/src/isolation.ts | 124 +- .../src/engine/engine.integration.test.ts | 40 + packages/orchestration/src/engine/engine.ts | 90 +- .../orchestration/src/engine/run-lifecycle.ts | 53 +- 12 files changed, 3404 insertions(+), 85 deletions(-) create mode 100644 packages/db/drizzle/0003_run-next-event-seq.sql create mode 100644 packages/db/drizzle/meta/0003_snapshot.json diff --git a/apps/worker/src/index.ts b/apps/worker/src/index.ts index f290bff..096efd4 100644 --- a/apps/worker/src/index.ts +++ b/apps/worker/src/index.ts @@ -16,6 +16,7 @@ import { findStrandedApprovalRuns, InProcessEventBus, RedisEventBus, + transitionRun, } from "@agrippa/orchestration"; import { and, eq, lt, sql } from "drizzle-orm"; import type { Job, JobWithMetadata } from "pg-boss"; @@ -103,16 +104,20 @@ async function scheduleApprovalExpiry(runId: string): Promise { } async function markRunFailed(runId: string, err: unknown): Promise { - const [run] = await db.select().from(runs).where(eq(runs.id, runId)); + const [run] = await db.select({ status: runs.status }).from(runs).where(eq(runs.id, runId)); if (!run || isTerminalRunStatus(run.status)) return; - await db - .update(runs) - .set({ - status: "failed", - finishedAt: new Date(), - error: { code: "internal", message: `retries exhausted: ${String(err).slice(0, 500)}` }, - }) - .where(eq(runs.id, runId)); + // route through the lifecycle CAS so a concurrent transition (e.g. a cancel or + // a resumed worker finalizing) can't be clobbered by this retry-exhaustion path + await db.transaction(async (tx) => { + if (!(await transitionRun(tx, runId, run.status, "failed"))) return; + await tx + .update(runs) + .set({ + finishedAt: new Date(), + error: { code: "internal", message: `retries exhausted: ${String(err).slice(0, 500)}` }, + }) + .where(eq(runs.id, runId)); + }); } /** diff --git a/packages/db/drizzle/0003_run-next-event-seq.sql b/packages/db/drizzle/0003_run-next-event-seq.sql new file mode 100644 index 0000000..604ce40 --- /dev/null +++ b/packages/db/drizzle/0003_run-next-event-seq.sql @@ -0,0 +1,6 @@ +ALTER TABLE "runs" ADD COLUMN "next_event_seq" integer DEFAULT 0 NOT NULL; +--> statement-breakpoint +UPDATE "runs" SET "next_event_seq" = COALESCE( + (SELECT MAX("seq") FROM "run_events" WHERE "run_events"."run_id" = "runs"."id"), + 0 +); diff --git a/packages/db/drizzle/meta/0003_snapshot.json b/packages/db/drizzle/meta/0003_snapshot.json new file mode 100644 index 0000000..8b6eadc --- /dev/null +++ b/packages/db/drizzle/meta/0003_snapshot.json @@ -0,0 +1,3075 @@ +{ + "id": "46b81d2c-7b0f-4ab8-aae2-4735097ce331", + "prevId": "00a6f090-e151-4e07-a855-f65a5107ff60", + "version": "7", + "dialect": "postgresql", + "tables": { + "public.api_keys": { + "name": "api_keys", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "name": { + "name": "name", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "key_hash": { + "name": "key_hash", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "prefix": { + "name": "prefix", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "scopes": { + "name": "scopes", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'[]'::jsonb" + }, + "created_by": { + "name": "created_by", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "expires_at": { + "name": "expires_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "revoked_at": { + "name": "revoked_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "last_used_at": { + "name": "last_used_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + } + }, + "indexes": {}, + "foreignKeys": { + "api_keys_org_id_orgs_id_fk": { + "name": "api_keys_org_id_orgs_id_fk", + "tableFrom": "api_keys", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "api_keys_project_id_projects_id_fk": { + "name": "api_keys_project_id_projects_id_fk", + "tableFrom": "api_keys", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "api_keys_created_by_users_id_fk": { + "name": "api_keys_created_by_users_id_fk", + "tableFrom": "api_keys", + "tableTo": "users", + "columnsFrom": ["created_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.audit_logs": { + "name": "audit_logs", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "actor_user_id": { + "name": "actor_user_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "actor_api_key_id": { + "name": "actor_api_key_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "action": { + "name": "action", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "resource_type": { + "name": "resource_type", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "resource_id": { + "name": "resource_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "payload": { + "name": "payload", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "ip": { + "name": "ip", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "audit_logs_org_time_idx": { + "name": "audit_logs_org_time_idx", + "columns": [ + { + "expression": "org_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "created_at", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "audit_logs_org_id_orgs_id_fk": { + "name": "audit_logs_org_id_orgs_id_fk", + "tableFrom": "audit_logs", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "audit_logs_project_id_projects_id_fk": { + "name": "audit_logs_project_id_projects_id_fk", + "tableFrom": "audit_logs", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "audit_logs_actor_user_id_users_id_fk": { + "name": "audit_logs_actor_user_id_users_id_fk", + "tableFrom": "audit_logs", + "tableTo": "users", + "columnsFrom": ["actor_user_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "audit_logs_actor_api_key_id_api_keys_id_fk": { + "name": "audit_logs_actor_api_key_id_api_keys_id_fk", + "tableFrom": "audit_logs", + "tableTo": "api_keys", + "columnsFrom": ["actor_api_key_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.accounts": { + "name": "accounts", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "account_id": { + "name": "account_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "provider_id": { + "name": "provider_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "access_token": { + "name": "access_token", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "refresh_token": { + "name": "refresh_token", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "id_token": { + "name": "id_token", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "access_token_expires_at": { + "name": "access_token_expires_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "refresh_token_expires_at": { + "name": "refresh_token_expires_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "scope": { + "name": "scope", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "password": { + "name": "password", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "accounts_user_id_users_id_fk": { + "name": "accounts_user_id_users_id_fk", + "tableFrom": "accounts", + "tableTo": "users", + "columnsFrom": ["user_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.sessions": { + "name": "sessions", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "token": { + "name": "token", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "expires_at": { + "name": "expires_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true + }, + "ip_address": { + "name": "ip_address", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "user_agent": { + "name": "user_agent", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "sessions_user_id_users_id_fk": { + "name": "sessions_user_id_users_id_fk", + "tableFrom": "sessions", + "tableTo": "users", + "columnsFrom": ["user_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "sessions_token_unique": { + "name": "sessions_token_unique", + "nullsNotDistinct": false, + "columns": ["token"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.users": { + "name": "users", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "name": { + "name": "name", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "email": { + "name": "email", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "email_verified": { + "name": "email_verified", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": false + }, + "image": { + "name": "image", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "locale": { + "name": "locale", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'en'" + }, + "org_role": { + "name": "org_role", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'org_member'" + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "users_org_id_orgs_id_fk": { + "name": "users_org_id_orgs_id_fk", + "tableFrom": "users", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "users_email_unique": { + "name": "users_email_unique", + "nullsNotDistinct": false, + "columns": ["email"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.verifications": { + "name": "verifications", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "identifier": { + "name": "identifier", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "value": { + "name": "value", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "expires_at": { + "name": "expires_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.orgs": { + "name": "orgs", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "slug": { + "name": "slug", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "name": { + "name": "name", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "orgs_slug_unique": { + "name": "orgs_slug_unique", + "nullsNotDistinct": false, + "columns": ["slug"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.project_members": { + "name": "project_members", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "role": { + "name": "role", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "project_members_uq": { + "name": "project_members_uq", + "columns": [ + { + "expression": "project_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "user_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "project_members_project_id_projects_id_fk": { + "name": "project_members_project_id_projects_id_fk", + "tableFrom": "project_members", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "project_members_user_id_users_id_fk": { + "name": "project_members_user_id_users_id_fk", + "tableFrom": "project_members", + "tableTo": "users", + "columnsFrom": ["user_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.project_quotas": { + "name": "project_quotas", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "period": { + "name": "period", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'monthly'" + }, + "token_limit": { + "name": "token_limit", + "type": "bigint", + "primaryKey": false, + "notNull": false + }, + "cost_limit_usd": { + "name": "cost_limit_usd", + "type": "numeric(12, 2)", + "primaryKey": false, + "notNull": false + }, + "hard_stop": { + "name": "hard_stop", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": true + }, + "current_period_start": { + "name": "current_period_start", + "type": "date", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "project_quotas_project_id_projects_id_fk": { + "name": "project_quotas_project_id_projects_id_fk", + "tableFrom": "project_quotas", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "project_quotas_project_id_unique": { + "name": "project_quotas_project_id_unique", + "nullsNotDistinct": false, + "columns": ["project_id"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.project_resource_grants": { + "name": "project_resource_grants", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "resource_type": { + "name": "resource_type", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "resource_id": { + "name": "resource_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "config_override": { + "name": "config_override", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "granted_by": { + "name": "granted_by", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "project_grants_uq": { + "name": "project_grants_uq", + "columns": [ + { + "expression": "project_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "resource_type", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "resource_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "project_resource_grants_project_id_projects_id_fk": { + "name": "project_resource_grants_project_id_projects_id_fk", + "tableFrom": "project_resource_grants", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "project_resource_grants_granted_by_users_id_fk": { + "name": "project_resource_grants_granted_by_users_id_fk", + "tableFrom": "project_resource_grants", + "tableTo": "users", + "columnsFrom": ["granted_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.projects": { + "name": "projects", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "slug": { + "name": "slug", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "name": { + "name": "name", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "description": { + "name": "description", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'active'" + }, + "settings": { + "name": "settings", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "created_by": { + "name": "created_by", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "archived_at": { + "name": "archived_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + } + }, + "indexes": { + "projects_org_slug_uq": { + "name": "projects_org_slug_uq", + "columns": [ + { + "expression": "org_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "slug", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "projects_org_id_orgs_id_fk": { + "name": "projects_org_id_orgs_id_fk", + "tableFrom": "projects", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "projects_created_by_users_id_fk": { + "name": "projects_created_by_users_id_fk", + "tableFrom": "projects", + "tableTo": "users", + "columnsFrom": ["created_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.repo_connections": { + "name": "repo_connections", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "provider": { + "name": "provider", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "url": { + "name": "url", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "default_branch": { + "name": "default_branch", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'main'" + }, + "credential_secret_ref": { + "name": "credential_secret_ref", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'active'" + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "repo_connections_project_id_projects_id_fk": { + "name": "repo_connections_project_id_projects_id_fk", + "tableFrom": "repo_connections", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "repo_connections_credential_secret_ref_secrets_id_fk": { + "name": "repo_connections_credential_secret_ref_secrets_id_fk", + "tableFrom": "repo_connections", + "tableTo": "secrets", + "columnsFrom": ["credential_secret_ref"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.fabri": { + "name": "fabri", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "slug": { + "name": "slug", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "name_i18n": { + "name": "name_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "persona_i18n": { + "name": "persona_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "system_prompt": { + "name": "system_prompt", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "avatar": { + "name": "avatar", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "default_model_role_policy": { + "name": "default_model_role_policy", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'active'" + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "fabri_org_id_orgs_id_fk": { + "name": "fabri_org_id_orgs_id_fk", + "tableFrom": "fabri", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "fabri_slug_unique": { + "name": "fabri_slug_unique", + "nullsNotDistinct": false, + "columns": ["slug"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.mcp_servers": { + "name": "mcp_servers", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "slug": { + "name": "slug", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "name_i18n": { + "name": "name_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "transport": { + "name": "transport", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "config": { + "name": "config", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "auth_secret_ref": { + "name": "auth_secret_ref", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "config_revision": { + "name": "config_revision", + "type": "integer", + "primaryKey": false, + "notNull": true, + "default": 1 + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'active'" + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "mcp_servers_org_id_orgs_id_fk": { + "name": "mcp_servers_org_id_orgs_id_fk", + "tableFrom": "mcp_servers", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "mcp_servers_auth_secret_ref_secrets_id_fk": { + "name": "mcp_servers_auth_secret_ref_secrets_id_fk", + "tableFrom": "mcp_servers", + "tableTo": "secrets", + "columnsFrom": ["auth_secret_ref"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "mcp_servers_slug_unique": { + "name": "mcp_servers_slug_unique", + "nullsNotDistinct": false, + "columns": ["slug"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.models": { + "name": "models", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "provider": { + "name": "provider", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "provider_model_id": { + "name": "provider_model_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "display_name": { + "name": "display_name", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "tier": { + "name": "tier", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "capabilities": { + "name": "capabilities", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "context_window": { + "name": "context_window", + "type": "integer", + "primaryKey": false, + "notNull": false + }, + "input_cost_per_mtok": { + "name": "input_cost_per_mtok", + "type": "numeric(12, 4)", + "primaryKey": false, + "notNull": false + }, + "output_cost_per_mtok": { + "name": "output_cost_per_mtok", + "type": "numeric(12, 4)", + "primaryKey": false, + "notNull": false + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'active'" + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "models_org_id_orgs_id_fk": { + "name": "models_org_id_orgs_id_fk", + "tableFrom": "models", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "models_provider_model_id_unique": { + "name": "models_provider_model_id_unique", + "nullsNotDistinct": false, + "columns": ["provider_model_id"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.orchestration_templates": { + "name": "orchestration_templates", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "slug": { + "name": "slug", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "scenario_id": { + "name": "scenario_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "name_i18n": { + "name": "name_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "latest_published_version_id": { + "name": "latest_published_version_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "created_by": { + "name": "created_by", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "orchestration_templates_org_id_orgs_id_fk": { + "name": "orchestration_templates_org_id_orgs_id_fk", + "tableFrom": "orchestration_templates", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "orchestration_templates_scenario_id_scenarios_id_fk": { + "name": "orchestration_templates_scenario_id_scenarios_id_fk", + "tableFrom": "orchestration_templates", + "tableTo": "scenarios", + "columnsFrom": ["scenario_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "orchestration_templates_latest_published_version_id_template_versions_id_fk": { + "name": "orchestration_templates_latest_published_version_id_template_versions_id_fk", + "tableFrom": "orchestration_templates", + "tableTo": "template_versions", + "columnsFrom": ["latest_published_version_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "orchestration_templates_created_by_users_id_fk": { + "name": "orchestration_templates_created_by_users_id_fk", + "tableFrom": "orchestration_templates", + "tableTo": "users", + "columnsFrom": ["created_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "orchestration_templates_slug_unique": { + "name": "orchestration_templates_slug_unique", + "nullsNotDistinct": false, + "columns": ["slug"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.scenarios": { + "name": "scenarios", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "slug": { + "name": "slug", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "name_i18n": { + "name": "name_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "description_i18n": { + "name": "description_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "icon": { + "name": "icon", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "sort_order": { + "name": "sort_order", + "type": "integer", + "primaryKey": false, + "notNull": true, + "default": 0 + }, + "enabled": { + "name": "enabled", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "scenarios_org_id_orgs_id_fk": { + "name": "scenarios_org_id_orgs_id_fk", + "tableFrom": "scenarios", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "scenarios_slug_unique": { + "name": "scenarios_slug_unique", + "nullsNotDistinct": false, + "columns": ["slug"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.skill_versions": { + "name": "skill_versions", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "skill_id": { + "name": "skill_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "version": { + "name": "version", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "content_ref": { + "name": "content_ref", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "manifest": { + "name": "manifest", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'active'" + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "skill_versions_uq": { + "name": "skill_versions_uq", + "columns": [ + { + "expression": "skill_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "version", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "skill_versions_skill_id_skills_id_fk": { + "name": "skill_versions_skill_id_skills_id_fk", + "tableFrom": "skill_versions", + "tableTo": "skills", + "columnsFrom": ["skill_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.skills": { + "name": "skills", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "slug": { + "name": "slug", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "name_i18n": { + "name": "name_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "description_i18n": { + "name": "description_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "source": { + "name": "source", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "latest_version_id": { + "name": "latest_version_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "skills_org_id_orgs_id_fk": { + "name": "skills_org_id_orgs_id_fk", + "tableFrom": "skills", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "skills_latest_version_id_skill_versions_id_fk": { + "name": "skills_latest_version_id_skill_versions_id_fk", + "tableFrom": "skills", + "tableTo": "skill_versions", + "columnsFrom": ["latest_version_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "skills_slug_unique": { + "name": "skills_slug_unique", + "nullsNotDistinct": false, + "columns": ["slug"] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.task_types": { + "name": "task_types", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "scenario_id": { + "name": "scenario_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "slug": { + "name": "slug", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "name_i18n": { + "name": "name_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "description_i18n": { + "name": "description_i18n", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "template_id": { + "name": "template_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "default_faber_id": { + "name": "default_faber_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "enabled": { + "name": "enabled", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": true + }, + "sort_order": { + "name": "sort_order", + "type": "integer", + "primaryKey": false, + "notNull": true, + "default": 0 + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "task_types_uq": { + "name": "task_types_uq", + "columns": [ + { + "expression": "scenario_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "slug", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "task_types_scenario_id_scenarios_id_fk": { + "name": "task_types_scenario_id_scenarios_id_fk", + "tableFrom": "task_types", + "tableTo": "scenarios", + "columnsFrom": ["scenario_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "task_types_template_id_orchestration_templates_id_fk": { + "name": "task_types_template_id_orchestration_templates_id_fk", + "tableFrom": "task_types", + "tableTo": "orchestration_templates", + "columnsFrom": ["template_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "task_types_default_faber_id_fabri_id_fk": { + "name": "task_types_default_faber_id_fabri_id_fk", + "tableFrom": "task_types", + "tableTo": "fabri", + "columnsFrom": ["default_faber_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.template_versions": { + "name": "template_versions", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "template_id": { + "name": "template_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "version": { + "name": "version", + "type": "integer", + "primaryKey": false, + "notNull": true + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'draft'" + }, + "source_yaml": { + "name": "source_yaml", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "compiled": { + "name": "compiled", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "checksum": { + "name": "checksum", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "created_by": { + "name": "created_by", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "published_at": { + "name": "published_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + } + }, + "indexes": { + "template_versions_uq": { + "name": "template_versions_uq", + "columns": [ + { + "expression": "template_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "version", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "template_versions_template_id_orchestration_templates_id_fk": { + "name": "template_versions_template_id_orchestration_templates_id_fk", + "tableFrom": "template_versions", + "tableTo": "orchestration_templates", + "columnsFrom": ["template_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "template_versions_created_by_users_id_fk": { + "name": "template_versions_created_by_users_id_fk", + "tableFrom": "template_versions", + "tableTo": "users", + "columnsFrom": ["created_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.approvals": { + "name": "approvals", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "run_id": { + "name": "run_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "step_id": { + "name": "step_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "checkpoint_id": { + "name": "checkpoint_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'pending'" + }, + "payload": { + "name": "payload", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "requested_at": { + "name": "requested_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "decided_by": { + "name": "decided_by", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "decided_at": { + "name": "decided_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "comment": { + "name": "comment", + "type": "text", + "primaryKey": false, + "notNull": false + } + }, + "indexes": {}, + "foreignKeys": { + "approvals_run_id_runs_id_fk": { + "name": "approvals_run_id_runs_id_fk", + "tableFrom": "approvals", + "tableTo": "runs", + "columnsFrom": ["run_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "approvals_step_id_run_steps_id_fk": { + "name": "approvals_step_id_run_steps_id_fk", + "tableFrom": "approvals", + "tableTo": "run_steps", + "columnsFrom": ["step_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "approvals_decided_by_users_id_fk": { + "name": "approvals_decided_by_users_id_fk", + "tableFrom": "approvals", + "tableTo": "users", + "columnsFrom": ["decided_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.artifacts": { + "name": "artifacts", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "run_id": { + "name": "run_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "step_id": { + "name": "step_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "artifact_key": { + "name": "artifact_key", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "kind": { + "name": "kind", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "name": { + "name": "name", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "mime": { + "name": "mime", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "size": { + "name": "size", + "type": "integer", + "primaryKey": false, + "notNull": false + }, + "storage_ref": { + "name": "storage_ref", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "inline": { + "name": "inline", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "artifacts_run_id_runs_id_fk": { + "name": "artifacts_run_id_runs_id_fk", + "tableFrom": "artifacts", + "tableTo": "runs", + "columnsFrom": ["run_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "artifacts_step_id_run_steps_id_fk": { + "name": "artifacts_step_id_run_steps_id_fk", + "tableFrom": "artifacts", + "tableTo": "run_steps", + "columnsFrom": ["step_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.run_events": { + "name": "run_events", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "bigserial", + "primaryKey": true, + "notNull": true + }, + "run_id": { + "name": "run_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "step_id": { + "name": "step_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "seq": { + "name": "seq", + "type": "integer", + "primaryKey": false, + "notNull": true + }, + "type": { + "name": "type", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "payload": { + "name": "payload", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "run_events_run_seq_uq": { + "name": "run_events_run_seq_uq", + "columns": [ + { + "expression": "run_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "seq", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "run_events_run_id_runs_id_fk": { + "name": "run_events_run_id_runs_id_fk", + "tableFrom": "run_events", + "tableTo": "runs", + "columnsFrom": ["run_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "run_events_step_id_run_steps_id_fk": { + "name": "run_events_step_id_run_steps_id_fk", + "tableFrom": "run_events", + "tableTo": "run_steps", + "columnsFrom": ["step_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.run_steps": { + "name": "run_steps", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "run_id": { + "name": "run_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "phase_id": { + "name": "phase_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "step_id": { + "name": "step_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "attempt": { + "name": "attempt", + "type": "integer", + "primaryKey": false, + "notNull": true, + "default": 1 + }, + "seq": { + "name": "seq", + "type": "integer", + "primaryKey": false, + "notNull": true + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'pending'" + }, + "agent_ref": { + "name": "agent_ref", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "model_id": { + "name": "model_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "executor_session_id": { + "name": "executor_session_id", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "output": { + "name": "output", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "usage": { + "name": "usage", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "error": { + "name": "error", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "started_at": { + "name": "started_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "finished_at": { + "name": "finished_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + } + }, + "indexes": { + "run_steps_uq": { + "name": "run_steps_uq", + "columns": [ + { + "expression": "run_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "phase_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "step_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "attempt", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "run_steps_run_id_runs_id_fk": { + "name": "run_steps_run_id_runs_id_fk", + "tableFrom": "run_steps", + "tableTo": "runs", + "columnsFrom": ["run_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.runs": { + "name": "runs", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "task_id": { + "name": "task_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "number": { + "name": "number", + "type": "integer", + "primaryKey": false, + "notNull": true + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'queued'" + }, + "template_version_id": { + "name": "template_version_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "faber_id": { + "name": "faber_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "executor_id": { + "name": "executor_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "params_snapshot": { + "name": "params_snapshot", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "model_resolution": { + "name": "model_resolution", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "resource_manifest": { + "name": "resource_manifest", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{\"mcpServers\":[],\"skills\":[]}'::jsonb" + }, + "budget": { + "name": "budget", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "usage_totals": { + "name": "usage_totals", + "type": "jsonb", + "primaryKey": false, + "notNull": true, + "default": "'{}'::jsonb" + }, + "next_event_seq": { + "name": "next_event_seq", + "type": "integer", + "primaryKey": false, + "notNull": true, + "default": 0 + }, + "workspace_ref": { + "name": "workspace_ref", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "error": { + "name": "error", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "cancel_requested": { + "name": "cancel_requested", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": false + }, + "queued_at": { + "name": "queued_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "started_at": { + "name": "started_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "finished_at": { + "name": "finished_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + }, + "created_by": { + "name": "created_by", + "type": "uuid", + "primaryKey": false, + "notNull": true + } + }, + "indexes": { + "runs_task_number_uq": { + "name": "runs_task_number_uq", + "columns": [ + { + "expression": "task_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "number", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + }, + "runs_project_idx": { + "name": "runs_project_idx", + "columns": [ + { + "expression": "project_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "status", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "runs_task_id_tasks_id_fk": { + "name": "runs_task_id_tasks_id_fk", + "tableFrom": "runs", + "tableTo": "tasks", + "columnsFrom": ["task_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "runs_project_id_projects_id_fk": { + "name": "runs_project_id_projects_id_fk", + "tableFrom": "runs", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "runs_template_version_id_template_versions_id_fk": { + "name": "runs_template_version_id_template_versions_id_fk", + "tableFrom": "runs", + "tableTo": "template_versions", + "columnsFrom": ["template_version_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "runs_faber_id_fabri_id_fk": { + "name": "runs_faber_id_fabri_id_fk", + "tableFrom": "runs", + "tableTo": "fabri", + "columnsFrom": ["faber_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "runs_created_by_users_id_fk": { + "name": "runs_created_by_users_id_fk", + "tableFrom": "runs", + "tableTo": "users", + "columnsFrom": ["created_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.tasks": { + "name": "tasks", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "task_type_id": { + "name": "task_type_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "title": { + "name": "title", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "params": { + "name": "params", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "latest_run_id": { + "name": "latest_run_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "created_by": { + "name": "created_by", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "tasks_org_id_orgs_id_fk": { + "name": "tasks_org_id_orgs_id_fk", + "tableFrom": "tasks", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "tasks_project_id_projects_id_fk": { + "name": "tasks_project_id_projects_id_fk", + "tableFrom": "tasks", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "tasks_task_type_id_task_types_id_fk": { + "name": "tasks_task_type_id_task_types_id_fk", + "tableFrom": "tasks", + "tableTo": "task_types", + "columnsFrom": ["task_type_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "tasks_latest_run_id_runs_id_fk": { + "name": "tasks_latest_run_id_runs_id_fk", + "tableFrom": "tasks", + "tableTo": "runs", + "columnsFrom": ["latest_run_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "tasks_created_by_users_id_fk": { + "name": "tasks_created_by_users_id_fk", + "tableFrom": "tasks", + "tableTo": "users", + "columnsFrom": ["created_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.secrets": { + "name": "secrets", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "kind": { + "name": "kind", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "ciphertext": { + "name": "ciphertext", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "created_by": { + "name": "created_by", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "rotated_at": { + "name": "rotated_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": false + } + }, + "indexes": {}, + "foreignKeys": { + "secrets_org_id_orgs_id_fk": { + "name": "secrets_org_id_orgs_id_fk", + "tableFrom": "secrets", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "secrets_created_by_users_id_fk": { + "name": "secrets_created_by_users_id_fk", + "tableFrom": "secrets", + "tableTo": "users", + "columnsFrom": ["created_by"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.token_usage": { + "name": "token_usage", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "org_id": { + "name": "org_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "project_id": { + "name": "project_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "run_id": { + "name": "run_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "step_id": { + "name": "step_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "attempt": { + "name": "attempt", + "type": "integer", + "primaryKey": false, + "notNull": true, + "default": 1 + }, + "model_id": { + "name": "model_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "input_tokens": { + "name": "input_tokens", + "type": "bigint", + "primaryKey": false, + "notNull": true, + "default": 0 + }, + "output_tokens": { + "name": "output_tokens", + "type": "bigint", + "primaryKey": false, + "notNull": true, + "default": 0 + }, + "cache_read_tokens": { + "name": "cache_read_tokens", + "type": "bigint", + "primaryKey": false, + "notNull": true, + "default": 0 + }, + "cache_write_tokens": { + "name": "cache_write_tokens", + "type": "bigint", + "primaryKey": false, + "notNull": true, + "default": 0 + }, + "cost_usd": { + "name": "cost_usd", + "type": "numeric(12, 6)", + "primaryKey": false, + "notNull": true, + "default": "'0'" + }, + "occurred_at": { + "name": "occurred_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "token_usage_project_time_idx": { + "name": "token_usage_project_time_idx", + "columns": [ + { + "expression": "project_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "occurred_at", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "token_usage_run_idx": { + "name": "token_usage_run_idx", + "columns": [ + { + "expression": "run_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "token_usage_org_id_orgs_id_fk": { + "name": "token_usage_org_id_orgs_id_fk", + "tableFrom": "token_usage", + "tableTo": "orgs", + "columnsFrom": ["org_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "token_usage_project_id_projects_id_fk": { + "name": "token_usage_project_id_projects_id_fk", + "tableFrom": "token_usage", + "tableTo": "projects", + "columnsFrom": ["project_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "token_usage_run_id_runs_id_fk": { + "name": "token_usage_run_id_runs_id_fk", + "tableFrom": "token_usage", + "tableTo": "runs", + "columnsFrom": ["run_id"], + "columnsTo": ["id"], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "token_usage_step_id_run_steps_id_fk": { + "name": "token_usage_step_id_run_steps_id_fk", + "tableFrom": "token_usage", + "tableTo": "run_steps", + "columnsFrom": ["step_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + }, + "token_usage_model_id_models_id_fk": { + "name": "token_usage_model_id_models_id_fk", + "tableFrom": "token_usage", + "tableTo": "models", + "columnsFrom": ["model_id"], + "columnsTo": ["id"], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + } + }, + "enums": {}, + "schemas": {}, + "sequences": {}, + "roles": {}, + "policies": {}, + "views": {}, + "_meta": { + "columns": {}, + "schemas": {}, + "tables": {} + } +} diff --git a/packages/db/drizzle/meta/_journal.json b/packages/db/drizzle/meta/_journal.json index a0d542a..52e9e7e 100644 --- a/packages/db/drizzle/meta/_journal.json +++ b/packages/db/drizzle/meta/_journal.json @@ -22,6 +22,13 @@ "when": 1784274345055, "tag": "0002_run-resource-manifest", "breakpoints": true + }, + { + "idx": 3, + "version": "7", + "when": 1784282490871, + "tag": "0003_run-next-event-seq", + "breakpoints": true } ] } diff --git a/packages/db/src/schema/runs.ts b/packages/db/src/schema/runs.ts index b915e76..94d2c52 100644 --- a/packages/db/src/schema/runs.ts +++ b/packages/db/src/schema/runs.ts @@ -66,6 +66,8 @@ export const runs = pgTable( .default({ mcpServers: [], skills: [] }), budget: jsonb("budget").$type>().notNull().default({}), usageTotals: jsonb("usage_totals").$type>().notNull().default({}), + // atomic per-run event-seq allocator (UPDATE … RETURNING); avoids max(seq)+1 races + nextEventSeq: integer("next_event_seq").notNull().default(0), workspaceRef: text("workspace_ref"), error: jsonb("error").$type>(), cancelRequested: boolean("cancel_requested").notNull().default(false), diff --git a/packages/executor-claude/src/executor.test.ts b/packages/executor-claude/src/executor.test.ts index 705d74f..7ea1f98 100644 --- a/packages/executor-claude/src/executor.test.ts +++ b/packages/executor-claude/src/executor.test.ts @@ -116,6 +116,14 @@ describe("claude executor option mapping (docs/design/03)", () => { ); // read-write: shell is permitted (OS-sandboxed when available) expect((await rw("Bash", { command: "ls" }, ctx))?.behavior).toBe("allow"); + // reads are confined to the workspace: /proc and other runs are denied + expect((await rw("Read", { file_path: "/proc/self/environ" }, ctx))?.behavior).toBe("deny"); + expect((await rw("Read", { file_path: "/work/runs/other/secret" }, ctx))?.behavior).toBe( + "deny", + ); + expect( + (await rw("Read", { file_path: path.join(rwReq.workspaceDir, "src/a.ts") }, ctx))?.behavior, + ).toBe("allow"); // read-only: shell denied, repo writes denied, artifact writes allowed const roReq = makeRequest(); diff --git a/packages/executor-claude/src/executor.ts b/packages/executor-claude/src/executor.ts index a74831d..b5d2230 100644 --- a/packages/executor-claude/src/executor.ts +++ b/packages/executor-claude/src/executor.ts @@ -6,10 +6,11 @@ import { type Executor, type ExecutorEvent, evaluateToolCall, + isReadTool, isWriteTool, - realWriteContained, + pathArgOf, + realContained, type StepExecutionRequest, - writeTargetOf, } from "@agrippa/executor-core"; import { type Options, type SDKMessage, query as sdkQuery } from "@anthropic-ai/claude-agent-sdk"; @@ -117,14 +118,14 @@ export function buildQueryArgs( const decision = evaluateToolCall(req.toolPolicy, req.workspaceDir, toolName, record); if (decision.behavior === "deny") return decision; // the lexical check above can't see through symlinks — verify the real - // write target stays inside the workspace before allowing a write tool - const target = writeTargetOf(record); - if (isWriteTool(toolName) && target !== undefined) { + // target stays inside the workspace before allowing a read or write tool + const target = pathArgOf(record); + if ((isWriteTool(toolName) || isReadTool(toolName)) && target !== undefined) { const resolved = path.resolve(req.workspaceDir, target); - if (!(await realWriteContained(req.toolPolicy.writeRoot, resolved))) { + if (!(await realContained(req.toolPolicy.writeRoot, resolved))) { return { behavior: "deny", - message: `write resolves (via a symlink) outside the run workspace (${target})`, + message: `path resolves (via a symlink) outside the run workspace (${target})`, }; } } diff --git a/packages/executor-core/src/isolation.test.ts b/packages/executor-core/src/isolation.test.ts index 8324ea6..822e5e5 100644 --- a/packages/executor-core/src/isolation.test.ts +++ b/packages/executor-core/src/isolation.test.ts @@ -3,7 +3,13 @@ import { mkdtempSync, rmSync, symlinkSync } from "node:fs"; import { mkdir } from "node:fs/promises"; import { tmpdir } from "node:os"; import path from "node:path"; -import { buildScrubbedEnv, evaluateToolCall, isWithin, realWriteContained } from "./isolation"; +import { + buildScrubbedEnv, + createSecretRedactor, + evaluateToolCall, + isWithin, + realContained, +} from "./isolation"; const ROOT = "/work/runs/run-1"; const rw = { access: "readWrite" as const, writeRoot: ROOT }; @@ -32,6 +38,22 @@ describe("evaluateToolCall — read-write workspace", () => { // relative path resolves against the workspace, escaping is denied expect(evaluateToolCall(rw, ROOT, "Edit", { file_path: "../run-2/a" }).behavior).toBe("deny"); }); + + it("confines reads to the workspace (blocks /proc, other runs, artifact store)", () => { + // in-workspace reads and no-path reads (default cwd) are fine + expect(evaluateToolCall(rw, ROOT, "Read", { file_path: `${ROOT}/src/a.ts` }).behavior).toBe( + "allow", + ); + expect(evaluateToolCall(rw, ROOT, "Grep", { pattern: "TODO" }).behavior).toBe("allow"); + // escaping reads are denied + expect(evaluateToolCall(rw, ROOT, "Read", { file_path: "/proc/self/environ" }).behavior).toBe( + "deny", + ); + expect( + evaluateToolCall(rw, ROOT, "Read", { file_path: "/work/runs/run-2/secret" }).behavior, + ).toBe("deny"); + expect(evaluateToolCall(rw, ROOT, "Glob", { path: "/work/artifacts" }).behavior).toBe("deny"); + }); }); describe("evaluateToolCall — read-only workspace", () => { @@ -84,7 +106,7 @@ describe("buildScrubbedEnv", () => { }); }); -describe("realWriteContained", () => { +describe("realContained", () => { const dirs: string[] = []; const ws = () => { const d = mkdtempSync(path.join(tmpdir(), "iso-ws-")); @@ -96,12 +118,28 @@ describe("realWriteContained", () => { const root = ws(); await mkdir(path.join(root, "src"), { recursive: true }); // a not-yet-existing file under a real dir is contained - expect(await realWriteContained(root, path.join(root, "src/new.ts"))).toBe(true); + expect(await realContained(root, path.join(root, "src/new.ts"))).toBe(true); // a symlinked directory pointing outside defeats the lexical check const outside = ws(); symlinkSync(outside, path.join(root, "escape")); - expect(await realWriteContained(root, path.join(root, "escape/x.ts"))).toBe(false); + expect(await realContained(root, path.join(root, "escape/x.ts"))).toBe(false); for (const d of dirs.splice(0)) rmSync(d, { recursive: true, force: true }); }); }); + +describe("createSecretRedactor", () => { + it("replaces known secret values anywhere in a payload, ignoring short values", () => { + const r = createSecretRedactor(["sk-ant-supersecretvalue"]); + r.add(["ghp_anotherlongtoken12345", "b"]); // "b" is too short → ignored + const out = r.redact({ + text: "leaked sk-ant-supersecretvalue here", + nested: ["ghp_anotherlongtoken12345", { k: "safe b value" }], + num: 7, + }) as { text: string; nested: [string, { k: string }]; num: number }; + expect(out.text).toBe("leaked [REDACTED] here"); + expect(out.nested[0]).toBe("[REDACTED]"); + expect(out.nested[1].k).toBe("safe b value"); // short "b" not redacted + expect(out.num).toBe(7); + }); +}); diff --git a/packages/executor-core/src/isolation.ts b/packages/executor-core/src/isolation.ts index 3f50534..41528f4 100644 --- a/packages/executor-core/src/isolation.ts +++ b/packages/executor-core/src/isolation.ts @@ -10,10 +10,12 @@ import path from "node:path"; * adapter and its tests — the adapter must not re-implement any of it. * * What this layer can and cannot do: it statically contains the file-writing - * tools (Write/Edit/NotebookEdit) and refuses shell in read-only workspaces. - * It does **not** contain arbitrary writes a shell command makes in a - * read-write workspace — that requires OS-level isolation (the SDK `sandbox` - * option / a non-root worker / a container), layered on top by the adapter. + * and file-reading tools (Write/Edit/Read/Grep/Glob) to the workspace and + * refuses shell in read-only workspaces, and it redacts known secret values + * from event payloads. It does **not** contain what a shell command reads or + * writes in a read-write workspace — that requires OS-level isolation (the SDK + * `sandbox` option / a non-root worker / a container), layered on top by the + * adapter — and it cannot isolate one run from another at the OS level. */ export type WorkspaceAccess = "readOnly" | "readWrite"; @@ -23,6 +25,8 @@ export const ARTIFACT_SUBDIR = ".agrippa/artifacts"; /** Tools that write to the filesystem through a file_path / path arg. */ const WRITE_TOOLS = new Set(["Write", "Edit", "MultiEdit", "NotebookEdit"]); +/** Tools that read the filesystem through a file_path / path arg. */ +const READ_TOOLS = new Set(["Read", "Grep", "Glob", "NotebookRead"]); /** Tools that execute arbitrary commands (uncontainable by static rules). */ const EXEC_TOOLS = new Set(["Bash", "BashOutput", "KillShell", "KillBash"]); @@ -31,11 +35,19 @@ export function isWriteTool(toolName: string): boolean { return WRITE_TOOLS.has(toolName); } -/** The path argument a write tool targets, if any. */ -export function writeTargetOf(input: Record): string | undefined { +/** Whether a tool reads the filesystem through a path argument. */ +export function isReadTool(toolName: string): boolean { + return READ_TOOLS.has(toolName); +} + +/** The filesystem path argument a tool targets, if any. */ +export function pathArgOf(input: Record): string | undefined { return (input.file_path ?? input.path ?? input.notebook_path) as string | undefined; } +/** @deprecated use {@link pathArgOf}; kept for the write-tool call site. */ +export const writeTargetOf = pathArgOf; + export type ToolDecision = { behavior: "allow" } | { behavior: "deny"; message: string }; /** @@ -55,7 +67,11 @@ export function isWithin(parent: string, child: string): boolean { * - read-only workspace: file writes are confined to the artifact directory * (the agent still has to emit its declared artifacts) and shell is denied; * - read-write workspace: file writes must stay within the workspace root; - * shell is allowed (contained by the OS sandbox when available). + * shell is allowed (contained by the OS sandbox when available); + * - reads (Read/Grep/Glob) with an explicit path must stay within the workspace, + * so the agent can't read `/proc/self/environ`, another run's directory, or the + * shared artifact store. A read with no path argument defaults to the cwd + * (the workspace) and is allowed. */ export function evaluateToolCall( policy: { access: WorkspaceAccess; writeRoot: string }, @@ -76,7 +92,7 @@ export function evaluateToolCall( } if (WRITE_TOOLS.has(toolName)) { - const target = writeTargetOf(input); + const target = pathArgOf(input); if (target === undefined) return { behavior: "allow" }; const resolved = path.resolve(workspaceDir, target); if (!isWithin(writeRoot, resolved)) { @@ -96,20 +112,32 @@ export function evaluateToolCall( } } + if (READ_TOOLS.has(toolName)) { + const target = pathArgOf(input); + if (target === undefined) return { behavior: "allow" }; // defaults to the workspace cwd + const resolved = path.resolve(workspaceDir, target); + if (!isWithin(writeRoot, resolved)) { + return { + behavior: "deny", + message: `reads outside the run workspace are not permitted (${target})`, + }; + } + } + return { behavior: "allow" }; } /** - * Symlink-safe containment for a write target. `evaluateToolCall` is purely - * lexical, so a symlink component (e.g. `workspace/link -> /app`) would slip a - * write past it; this canonicalizes the nearest existing ancestor of the target - * (the file itself may not exist yet) and confirms it stays inside `writeRoot`. - * Fail-closed on any resolution error. + * Symlink-safe containment for a read or write target. `evaluateToolCall` is + * purely lexical, so a symlink component (e.g. `workspace/link -> /app`) would + * slip a target past it; this canonicalizes the nearest existing ancestor of the + * target (the file itself may not exist yet, e.g. a fresh write) and confirms it + * stays inside `root`. Fail-closed on any resolution error. */ -export async function realWriteContained(writeRoot: string, target: string): Promise { - let root: string; +export async function realContained(root: string, target: string): Promise { + let realRoot: string; try { - root = await realpath(path.resolve(writeRoot)); + realRoot = await realpath(path.resolve(root)); } catch { return false; } @@ -117,7 +145,7 @@ export async function realWriteContained(writeRoot: string, target: string): Pro for (;;) { try { const real = await realpath(dir); - return real === root || real.startsWith(root + path.sep); + return real === realRoot || real.startsWith(realRoot + path.sep); } catch { const parent = path.dirname(dir); if (parent === dir) return false; // reached filesystem root without a hit @@ -180,3 +208,65 @@ export function buildScrubbedEnv( } return out; } + +/** + * Secret VALUES worth redacting from anything the agent can surface (event + * payloads, tool output). Covers the platform secrets plus the provider auth + * variables that the subprocess legitimately keeps but must never be echoed + * back through SSE/the timeline. + */ +export function collectEnvSecretValues( + source: Record = process.env, +): string[] { + const keys = [ + ...SECRET_ENV_KEYS, + "ANTHROPIC_API_KEY", + "ANTHROPIC_AUTH_TOKEN", + "CLAUDE_CODE_OAUTH_TOKEN", + ]; + return keys + .map((k) => source[k]) + .filter((v): v is string => typeof v === "string" && v.length > 0); +} + +export type SecretRedactor = { + /** Add more secret values to redact (e.g. per-step resolved MCP tokens). */ + add(values: Array): void; + /** Deep-replace every known secret value with a placeholder. */ + redact(value: T): T; +}; + +const REDACTION_PLACEHOLDER = "[REDACTED]"; +/** Below this length a "secret" would match innocuous substrings and corrupt output. */ +const MIN_SECRET_LEN = 8; + +/** + * Redacts known secret values from event payloads before they are persisted or + * streamed. Values shorter than {@link MIN_SECRET_LEN} are ignored so a short or + * empty token can't blank out unrelated text. + */ +export function createSecretRedactor(initial: Array = []): SecretRedactor { + const secrets = new Set(); + const add = (values: Array) => { + for (const v of values) if (v && v.length >= MIN_SECRET_LEN) secrets.add(v); + }; + add(initial); + const redactString = (s: string): string => { + let out = s; + for (const secret of secrets) { + if (out.includes(secret)) out = out.split(secret).join(REDACTION_PLACEHOLDER); + } + return out; + }; + const walk = (value: unknown): unknown => { + if (typeof value === "string") return redactString(value); + if (Array.isArray(value)) return value.map(walk); + if (value && typeof value === "object") { + const out: Record = {}; + for (const [k, v] of Object.entries(value)) out[k] = walk(v); + return out; + } + return value; + }; + return { add, redact: (value) => walk(value) as typeof value }; +} diff --git a/packages/orchestration/src/engine/engine.integration.test.ts b/packages/orchestration/src/engine/engine.integration.test.ts index dd6a821..815ed0f 100644 --- a/packages/orchestration/src/engine/engine.integration.test.ts +++ b/packages/orchestration/src/engine/engine.integration.test.ts @@ -604,6 +604,39 @@ describe.skipIf(!dbUp)("orchestration engine (FakeExecutor compliance suite)", ( expect(seen).toContain("usage"); expect(seen).toContain("approval.required"); }); + + it("redacts known secret values from persisted events", async () => { + const secret = "sk-ant-supersecretvalue-1234567890"; + const prev = process.env.ANTHROPIC_API_KEY; + process.env.ANTHROPIC_API_KEY = secret; // the engine seeds its redactor from env + let db: Db; + let runId: string; + try { + const fx = await setupFixture(); + db = fx.db; + runId = fx.runId; + const script: Record = { + ...HAPPY_SCRIPT, + "reproduce-bug": { + kind: "succeed", + events: [ + { type: "message.completed", role: "assistant", text: `the key is ${secret} oops` }, + { type: "artifact", key: "reproduction-report", kind: "markdown", inline: "# R" }, + ], + output: "done", + }, + }; + await executeRun(fx.makeDeps(script), runId); + } finally { + if (prev === undefined) delete process.env.ANTHROPIC_API_KEY; + else process.env.ANTHROPIC_API_KEY = prev; + } + const events = await db.select().from(runEvents).where(eq(runEvents.runId, runId)); + const msg = events.find((e) => e.type === "message.completed"); + const serialized = JSON.stringify(msg?.payload); + expect(serialized).toContain("[REDACTED]"); + expect(serialized).not.toContain(secret); + }); }); describe.skipIf(!dbUp)("run-lifecycle module", () => { @@ -631,6 +664,13 @@ describe.skipIf(!dbUp)("run-lifecycle module", () => { const seqs = [a.seq, b.seq, c.seq, d.seq].sort((m, n) => m - n); expect(new Set(seqs).size).toBe(4); expect(b.seq).toBeGreaterThan(a.seq); + + // must also work INSIDE a transaction — the old max(seq)+1-with-retry aborted + // the whole tx on the first unique violation (the approval-flow regression) + const e = await db.transaction((tx) => + appendRunEvent(tx, { runId, type: "x.five", payload: {} }), + ); + expect(e.seq).toBeGreaterThan(d.seq); }); it("findStrandedApprovalRuns selects only runs with no pending approval", async () => { diff --git a/packages/orchestration/src/engine/engine.ts b/packages/orchestration/src/engine/engine.ts index d9da070..2b55cf3 100644 --- a/packages/orchestration/src/engine/engine.ts +++ b/packages/orchestration/src/engine/engine.ts @@ -15,11 +15,15 @@ import { import { BudgetExceededError, BudgetMeter, + collectEnvSecretValues, + createSecretRedactor, type ExecutionContext, type Executor, type ExecutorEvent, type PriorStepSummary, + type ResolvedMcpServer, type ResolvedModel, + type SecretRedactor, type StepExecutionRequest, type UsageDelta, } from "@agrippa/executor-core"; @@ -63,6 +67,18 @@ class RunClaimLost extends Error { } } +/** The credential values a resolved MCP server injects, for secret redaction. */ +function mcpSecretValues(server: ResolvedMcpServer): string[] { + if (server.transport === "stdio") return Object.values(server.env); + const values: string[] = []; + for (const header of Object.values(server.headers)) { + values.push(header); + const bearer = /^Bearer\s+(.+)$/i.exec(header); + if (bearer) values.push(bearer[1] as string); + } + return values; +} + /** * Executes (or resumes) one run to its next stopping point: a terminal state * or a waiting_approval pause. Steps are the idempotency unit — on resume, @@ -113,6 +129,8 @@ class RunEngine { private producedArtifacts = new Set(); // stepId → crashed-attempt count + last executor session, for crash resume private crashRecovery = new Map(); + // scrubs known secret values from event payloads before persist/publish + private readonly redactor: SecretRedactor = createSecretRedactor(collectEnvSecretValues()); constructor( private readonly deps: EngineDeps, @@ -598,6 +616,8 @@ class RunEngine { const { resolved: mcpServers, missing } = await this.deps.resources.mcpServers( this.authorizedMcpRefs(step.mcpServers), ); + // register the resolved MCP credentials so they're redacted from any event + this.redactor.add(mcpServers.flatMap(mcpSecretValues)); const optionalRefs = new Set( this.template.spec.resources.mcpServers.filter((m) => m.optional).map((m) => m.ref), ); @@ -843,19 +863,22 @@ class RunEngine { payload: Record, stepRowId?: string, ): Promise { + // redact known secret values (provider key, MCP tokens) an agent may have + // echoed into message/tool output, so they never reach run_events or SSE + const safePayload = this.redactor.redact(payload); // seq is allocated by the database (run-lifecycle.appendRunEvent), not from // an in-memory counter that a concurrent writer could collide with const { seq, createdAt } = await appendRunEvent(this.db, { runId: this.run.id, stepId: stepRowId ?? this.currentStepRowId, type, - payload, + payload: safePayload, }); await this.deps.bus.publish({ runId: this.run.id, seq, type, - payload, + payload: safePayload, createdAt: createdAt.toISOString(), }); } @@ -931,22 +954,53 @@ class RunEngine { error: { code: string; message: string } | null, ): Promise { const snapshot = this.meter?.snapshot() ?? { costUsd: 0, tokens: 0, perPhaseCostUsd: {} }; - // CAS: if another path (e.g. a concurrent cancel) already finalized the run, - // do not overwrite its terminal status/error with ours - if (!(await this.transition(this.run.status, status))) return; - await this.db - .update(runs) - .set({ - finishedAt: new Date(), - error: error ?? null, - usageTotals: { - costUsd: snapshot.costUsd, - tokens: snapshot.tokens, - perPhaseCostUsd: snapshot.perPhaseCostUsd, - }, - }) - .where(eq(runs.id, this.run.id)); - await this.emit(`run.${status}`, error ? { error } : {}); + + // a cancel that landed after the last interrupt check still wins over a + // success — otherwise the API returns "cancel requested" but the run succeeds + let finalStatus = status; + let finalError = error; + if (status === "succeeded") { + const [row] = await this.db + .select({ cancelRequested: runs.cancelRequested }) + .from(runs) + .where(eq(runs.id, this.run.id)); + if (row?.cancelRequested) { + finalStatus = "cancelled"; + finalError = { code: "cancelled", message: "run cancelled" }; + } + } + const type = `run.${finalStatus}`; + const eventPayload = this.redactor.redact(finalError ? { error: finalError } : {}); + + // status flip + finishedAt/totals + terminal event commit together — a crash + // can no longer leave a terminal run with no finishedAt/totals/event that + // executeRun would then never repair. Publish to the bus only after commit. + const committed = await this.db.transaction(async (tx) => { + // CAS: if another path (e.g. a concurrent cancel) already finalized, bail + if (!(await transitionRun(tx, this.run.id, this.run.status, finalStatus))) return null; + this.run.status = finalStatus; + await tx + .update(runs) + .set({ + finishedAt: new Date(), + error: finalError ?? null, + usageTotals: { + costUsd: snapshot.costUsd, + tokens: snapshot.tokens, + perPhaseCostUsd: snapshot.perPhaseCostUsd, + }, + }) + .where(eq(runs.id, this.run.id)); + return await appendRunEvent(tx, { runId: this.run.id, type, payload: eventPayload }); + }); + if (!committed) return; // another path finalized the run + await this.deps.bus.publish({ + runId: this.run.id, + seq: committed.seq, + type, + payload: eventPayload, + createdAt: committed.createdAt.toISOString(), + }); try { await this.deps.workspace.cleanup(this.run.id); } catch (err) { diff --git a/packages/orchestration/src/engine/run-lifecycle.ts b/packages/orchestration/src/engine/run-lifecycle.ts index 9dec179..db71166 100644 --- a/packages/orchestration/src/engine/run-lifecycle.ts +++ b/packages/orchestration/src/engine/run-lifecycle.ts @@ -22,11 +22,6 @@ export type RunEventInput = { export type AppendedRunEvent = { seq: number; createdAt: Date }; -function isUniqueViolation(err: unknown): boolean { - const e = err as { code?: string; message?: string }; - return e?.code === "23505" || /duplicate key|run_events_run_seq_uq/i.test(e?.message ?? ""); -} - /** * Move a run from `from` to `to` iff it is still in `from` (compare-and-swap). * Returns true when this caller made the change, false when the row had already @@ -56,33 +51,31 @@ export async function transitionRun( } /** - * Append a run event with a database-allocated per-run seq. The seq is computed - * inside the INSERT (max+1 over the run's events), so serial writers are always - * correct; the unique (run_id, seq) index backstops the rare concurrent race, - * on which we retry rather than fail the job. + * Append a run event with a database-allocated per-run seq. The seq comes from an + * atomic `UPDATE runs SET next_event_seq = next_event_seq + 1 … RETURNING`: the + * row lock serializes concurrent allocations, so this is collision-free and — key + * for the approval flow — works inside a caller's transaction (the old + * max(seq)+1-with-retry aborted on the first unique violation inside a tx). */ export async function appendRunEvent(db: DbOrTx, event: RunEventInput): Promise { - const nextSeq = sql`(select coalesce(max(${runEvents.seq}), 0) + 1 from ${runEvents} where ${runEvents.runId} = ${event.runId})`; - for (let attempt = 0; attempt < 5; attempt++) { - try { - const [row] = await db - .insert(runEvents) - .values({ - runId: event.runId, - stepId: event.stepId ?? null, - seq: nextSeq, - type: event.type, - payload: event.payload ?? {}, - }) - .returning({ seq: runEvents.seq, createdAt: runEvents.createdAt }); - if (!row) throw new Error("run_events insert returned no row"); - return row; - } catch (err) { - if (isUniqueViolation(err) && attempt < 4) continue; - throw err; - } - } - throw new Error("run_events seq allocation exhausted retries"); + const [seqRow] = await db + .update(runs) + .set({ nextEventSeq: sql`${runs.nextEventSeq} + 1` }) + .where(eq(runs.id, event.runId)) + .returning({ seq: runs.nextEventSeq }); + if (!seqRow) throw new Error(`appendRunEvent: run ${event.runId} not found`); + const [row] = await db + .insert(runEvents) + .values({ + runId: event.runId, + stepId: event.stepId ?? null, + seq: seqRow.seq, + type: event.type, + payload: event.payload ?? {}, + }) + .returning({ seq: runEvents.seq, createdAt: runEvents.createdAt }); + if (!row) throw new Error("run_events insert returned no row"); + return row; } /** From 8eeed203f8104b25cc5a64e722bdd3f4f573bde1 Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 18:27:24 +0800 Subject: [PATCH 14/18] fix: attempt-safe/binary artifacts, requires.skills, per-phase budget, SSE, compose MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up review (round 3) — the correctness & wiring half. Artifacts (S3): - A zero-byte source now fails the contract instead of passing it (the skip keyed on inline===null, so an empty file still produced a row). - The artifact dir is cleared before each agent-step attempt (new WorkspaceManager.clearArtifacts), so a failed attempt's stale file can't be re-collected as a later attempt's result. - `file`-kind artifacts are read as bytes and stored byte-exact on disk, not decoded as UTF-8 (which corrupted binary content). Resources & budget (S4): - `requires.skills` is now validated by the compiler and enforced by the engine: a step requiring an ungranted/unavailable skill is skipped, not run without it. - Resume rebuilds per-phase spend (token_usage joined to run_steps by phase), so per-phase budgets survive a crash instead of resetting. - scripts/backfill-manifest.ts backfills the resource manifest for non-terminal runs that predate migration 0002 (documented upgrade step). SSE (S5): - The stream awaits the subscription being live (RedisEventBus.subscribe now exposes a ready promise for the SUBSCRIBE ack) before replaying, closing the residual gap; and re-replays Postgres periodically to recover a dropped pub/sub message mid-run rather than only at terminal. Compose (S6): - The api service mounts /work (large-artifact downloads) and receives AGRIPPA_EXECUTOR (the API picks the executor at submit). Docs (ADR-0009, design 03/04, ARCHITECTURE, CHANGELOG) updated; the deferred architectural items (per-run container isolation, provider-key proxy, execution lease, quota reservations) are recorded as known limitations. --- ARCHITECTURE.md | 4 +- CHANGELOG.md | 13 ++-- apps/api/src/routes/execution.ts | 16 ++++- apps/worker/src/deps/artifacts.test.ts | 36 +++++++++++ apps/worker/src/deps/artifacts.ts | 60 +++++++++++++------ apps/worker/src/deps/workspace.ts | 7 +++ .../0009-security-correctness-deep-modules.md | 9 +-- docs/design/03-executor-abstraction.md | 4 +- docs/design/04-execution-runtime.md | 6 +- infra/docker-compose.yml | 5 ++ packages/orchestration/src/compile.test.ts | 7 +++ packages/orchestration/src/compile.ts | 4 ++ packages/orchestration/src/engine/bus.ts | 14 ++++- packages/orchestration/src/engine/deps.ts | 3 + .../src/engine/engine.integration.test.ts | 12 +++- packages/orchestration/src/engine/engine.ts | 40 ++++++++++--- packages/orchestration/src/engine/fakes.ts | 9 ++- .../orchestration/src/engine/redis-bus.ts | 26 +++++--- scripts/backfill-manifest.ts | 56 +++++++++++++++++ 19 files changed, 270 insertions(+), 61 deletions(-) create mode 100644 scripts/backfill-manifest.ts diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 5741bed..cac93d0 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -52,8 +52,8 @@ Dependency direction is enforced by `scripts/check-deps.ts` (runtime deps only). 3. **Steps are the idempotency unit** — restart-safe by template rule, resumable by session id where the executor supports it. 4. **Usage rows are keyed `(run, step, attempt)`** so retries re-incur cost without ever double-counting. 5. **Executors are stateless I/O**: all inputs in the request, all outputs as events; they never touch the database. -6. **Secrets never leave as plaintext**: encrypted at rest (AES-256-GCM), write-only in the API, scrubbed from git remotes before agent code runs, and stripped from the agent subprocess environment (the master key and datastore URLs never reach a tool call). -7. **Containment goes through one seam**: every tool call and the subprocess env are decided by `packages/executor-core/isolation.ts`; the adapter never reimplements it (ADR-0009). +6. **Secrets never leave as plaintext**: encrypted at rest (AES-256-GCM), write-only in the API, scrubbed from git remotes before agent code runs, stripped from the agent subprocess environment (the master key and datastore URLs never reach a tool call), and redacted from event payloads before they persist or stream. +7. **Containment goes through one seam**: every tool call (reads and writes confined to the workspace) and the subprocess env are decided by `packages/executor-core/isolation.ts`; the adapter never reimplements it (ADR-0009). OS-level isolation between runs and keeping the provider key out of the subprocess are deferred to the container layer. 8. **The worker trusts only the pinned manifest**: repos are project-scoped and skills/MCP resolve solely from `runs.resource_manifest`, never the mutable global registry. 9. **Lifecycle mutations are atomic**: run status transitions are compare-and-swap on the expected status and event `seq` is allocated by the database, so concurrent writers can't clobber a status or collide on a seq (`run-lifecycle.ts`). diff --git a/CHANGELOG.md b/CHANGELOG.md index 0d8fd88..2f1916c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,17 +8,20 @@ All notable changes to Agrippa are documented here. The format follows ### Security -- **Executor isolation seam** — one enforceable place (`packages/executor-core/isolation.ts`) decides every tool call and scrubs the subprocess environment. Read-only workspaces now actually deny shell and confine writes to the artifact directory; read-write workspaces confine writes to the workspace with a boundary-safe check (the previous `startsWith` let a sibling `-evil` path through, and `Bash` bypassed the check entirely). The SDK subprocess runs with the platform secrets (`AGRIPPA_SECRET_KEY`, datastore URLs) stripped from its environment, the OS `sandbox` enabled where available, `strictMcpConfig`, and repo-supplied `.claude`/`.mcp.json` removed after checkout so a checked-out repository can't inject hooks or permission overrides. The worker image runs as a non-root user. +- **Executor isolation seam** — one enforceable place (`packages/executor-core/isolation.ts`) decides every tool call and scrubs the subprocess environment. Read-only workspaces now actually deny shell and confine writes to the artifact directory; read-write workspaces confine writes to the workspace with a boundary-safe check (the previous `startsWith` let a sibling `-evil` path through, and `Bash` bypassed the check entirely). **Reads (Read/Grep/Glob) are confined to the workspace too**, so the agent can't read `/proc/self/environ`, another run's directory, or the shared artifact store. The SDK subprocess runs with the platform secrets (`AGRIPPA_SECRET_KEY`, datastore URLs) stripped from its environment (via an explicit allowlist so a namespaced `*_TOKEN`/`*_KEY` can't ride along), the OS `sandbox` enabled where available (bubblewrap installed in the worker image), `strictMcpConfig`, and repo-supplied `.claude`/`.mcp.json` removed after checkout so a checked-out repository can't inject hooks or permission overrides. The worker image runs as a non-root user with `/app` kept root-owned. +- **Event-payload secret redaction** — known secret values (the provider key, resolved MCP tokens) are redacted from every event before it is persisted or streamed over SSE, so a secret the agent echoes into output can't leak through the timeline. - **Artifact path containment** — artifact ingestion resolves sources through `realpath` and rejects any that escape the workspace, closing a symlink disclosure (e.g. `ln -s /proc/self/environ`) that could exfiltrate secrets or other runs' files through the download endpoint. - **Cross-tenant resource authorization** — submission rejects a `repoConnectionId` that isn't owned by the project, and the worker loads repo connections scoped to the run's project. Optional skills/MCP servers are now grant-checked: an authorized-resource manifest is pinned onto the run at submit (required grants enforced, optional resources included only when granted) and the worker resolves resources only from it, so a project without a grant can no longer receive the platform's global credential (e.g. the shared GitHub token). ### Fixed - **Crash recovery for no-retry steps** — a worker that died mid-step no longer silently skips (or spuriously fails) a step without template retries; the crashed attempt no longer consumes the retry budget, and the executor session is carried onto the recovery attempt so resume works. -- **Atomic run lifecycle** — run status transitions are compare-and-swap on the expected status (a late finalize can't overwrite a cancellation), event sequence numbers are allocated by the database (no `max+1` collisions), and approval decisions are CAS on `pending` with the sweeper re-enqueuing any run left paused by a lost resume enqueue. -- **Quota accounting** — the engine now counts the same monthly window as the submit gate, excludes the run's own spend from the headroom it checks (no double-count on resume), and re-reads project usage at each step boundary so concurrent runs can't jointly overspend. -- **Artifact output contract** — patch steps no longer hand-write the diff (the engine generates it from `git diff` as intended), collected artifacts are validated against the step's declared keys/kinds, files from earlier steps aren't re-emitted, and missing/empty sources don't create zero-byte artifact rows. -- **SSE live gap** — the events stream subscribes before replaying history, so an event committed in the replay/subscribe window is delivered live instead of only at the terminal replay (ADR-0007). +- **Atomic run lifecycle** — run status transitions are compare-and-swap on the expected status (a late finalize can't overwrite a cancellation), event sequence numbers come from an atomic per-run counter (`runs.next_event_seq`) so they never collide and allocate correctly inside a transaction, and approval decisions are CAS on `pending` with the sweeper re-enqueuing any run left paused by a lost resume enqueue. Finalization commits the status, totals, and terminal event in one transaction (no half-finalized runs), re-checks a late cancel so it wins over a success, and the retry-exhaustion path routes through the CAS. +- **Quota accounting** — the engine now counts the same monthly window as the submit gate, excludes the run's own spend from the headroom it checks (no double-count on resume), re-reads project usage at each step boundary so concurrent runs can't jointly overspend, and restores per-phase spend on resume so per-phase budgets aren't reset by a crash. +- **Artifact output contract** — patch steps no longer hand-write the diff (the engine generates it from `git diff` as intended), collected artifacts are validated against the step's declared keys/kinds and matched by exact filename, files from earlier steps aren't re-emitted, the artifact dir is cleared before each attempt so a failed attempt's stale file isn't emitted, missing/empty sources don't create zero-byte rows, and binary (`file`-kind) artifacts are stored byte-exact instead of decoded as text. +- **Authorized resources** — `requires.skills` is now validated by the compiler and enforced by the engine (a step requiring an ungranted skill is skipped, not run without it); `scripts/backfill-manifest.ts` backfills the manifest for runs that predate the manifest migration. +- **SSE live gap** — the events stream subscribes and *awaits* the subscription being live before replaying history (so an event in the replay/subscribe window is delivered live), and re-replays Postgres periodically to recover a dropped pub/sub message mid-run rather than only at terminal (ADR-0007). +- **Production Compose** — the `api` service now mounts the `/work` volume (so it can serve large artifact downloads) and receives `AGRIPPA_EXECUTOR` (so `fake` actually selects the fake executor, since the API chooses the executor at submit). ## [0.1.0] — 2026-07-17 diff --git a/apps/api/src/routes/execution.ts b/apps/api/src/routes/execution.ts index db83909..1cab692 100644 --- a/apps/api/src/routes/execution.ts +++ b/apps/api/src/routes/execution.ts @@ -445,7 +445,7 @@ export const executionRoutes = new Hono() if (bus) { const queue: Array<() => Promise> = []; let notify: (() => void) | null = null; - const unsubscribe = bus.subscribe(run.id, (event) => { + const subscription = bus.subscribe(run.id, (event) => { queue.push(async () => { if (event.seq > cursor) { await sendRow({ ...event, createdAt: event.createdAt }); @@ -454,7 +454,13 @@ export const executionRoutes = new Hono() notify?.(); }); try { - await replay(); // history first; the subscription is already buffering + // wait until the subscription is actually live, THEN replay history, + // so nothing published in between is dropped (ADR-0007) + await subscription.ready; + await replay(); + // periodically re-replay from Postgres so a dropped pub/sub message is + // recovered mid-run, not only when the run becomes terminal + let sinceReplay = 0; while (!closed) { while (queue.length > 0) { const job = queue.shift(); @@ -464,6 +470,10 @@ export const executionRoutes = new Hono() await replay(); // drain anything raced between bus and DB break; } + if (++sinceReplay >= 5) { + sinceReplay = 0; + await replay(); + } await new Promise((resolve) => { notify = resolve; setTimeout(resolve, 2000); @@ -471,7 +481,7 @@ export const executionRoutes = new Hono() notify = null; } } finally { - unsubscribe(); + subscription.unsubscribe(); } } else { await replay(); diff --git a/apps/worker/src/deps/artifacts.test.ts b/apps/worker/src/deps/artifacts.test.ts index e069071..c5b3bcb 100644 --- a/apps/worker/src/deps/artifacts.test.ts +++ b/apps/worker/src/deps/artifacts.test.ts @@ -67,4 +67,40 @@ describe("DiskArtifactStore path containment", () => { expect(stored.storageRef).toBeNull(); expect(stored.size).toBe(0); }); + + it("treats an existing but empty file as no content", async () => { + const ws = freshWorkspace(); + await mkdir(path.join(ws, ".agrippa/artifacts"), { recursive: true }); + writeFileSync(path.join(ws, ".agrippa/artifacts/empty.md"), ""); + const stored = await store.store( + "run-1", + "empty", + "markdown", + { path: ".agrippa/artifacts/empty.md" }, + ws, + ); + expect(stored.inline).toBeNull(); + expect(stored.storageRef).toBeNull(); + expect(stored.size).toBe(0); + }); + + it("stores a binary file-kind artifact byte-exact on disk, not decoded as text", async () => { + const ws = freshWorkspace(); + await mkdir(path.join(ws, ".agrippa/artifacts"), { recursive: true }); + const bytes = Uint8Array.from([0x89, 0x50, 0x4e, 0x47, 0x00, 0xff, 0xfe, 0x01]); // PNG-ish + nulls + writeFileSync(path.join(ws, ".agrippa/artifacts/blob"), bytes); + + const stored = await store.store( + "run-1", + "blob", + "file", + { path: ".agrippa/artifacts/blob" }, + ws, + ); + expect(stored.inline).toBeNull(); + expect(stored.storageRef).not.toBeNull(); + expect(stored.size).toBe(8); + const round = new Uint8Array(await Bun.file(stored.storageRef as string).arrayBuffer()); + expect([...round]).toEqual([...bytes]); // byte-exact, no UTF-8 corruption + }); }); diff --git a/apps/worker/src/deps/artifacts.ts b/apps/worker/src/deps/artifacts.ts index 151642b..7fe10a8 100644 --- a/apps/worker/src/deps/artifacts.ts +++ b/apps/worker/src/deps/artifacts.ts @@ -40,31 +40,53 @@ export class DiskArtifactStore implements ArtifactStore { source: { inline?: unknown; path?: string }, workspaceDir: string, ): Promise { - let content: string | null = null; - let mime: string | null = null; - + // engine-provided inline content (patch diffs, links) is always text if (source.inline !== undefined) { - content = typeof source.inline === "string" ? source.inline : JSON.stringify(source.inline); - mime = kind === "json" ? "application/json" : "text/markdown"; - } else if (source.path) { - const real = await resolveContainedPath(workspaceDir, source.path); - if (real === null) return EMPTY; - const file = Bun.file(real); - if (await file.exists()) { - content = await file.text(); - mime = file.type || null; - } + const content = + typeof source.inline === "string" ? source.inline : JSON.stringify(source.inline); + const mime = kind === "json" ? "application/json" : "text/markdown"; + return this.storeText(runId, key, content, mime); } - if (content === null) return EMPTY; + if (!source.path) return EMPTY; - const size = Buffer.byteLength(content); - if (size <= INLINE_LIMIT) { - return { inline: content, storageRef: null, size, mime }; + const real = await resolveContainedPath(workspaceDir, source.path); + if (real === null) return EMPTY; + const file = Bun.file(real); + if (!(await file.exists())) return EMPTY; + + // `file`-kind artifacts may be binary — read raw bytes and stream them on + // download rather than decoding to UTF-8 (which corrupts non-text content) + if (kind === "file") { + const bytes = new Uint8Array(await file.arrayBuffer()); + if (bytes.byteLength === 0) return EMPTY; + const storageRef = await this.writeToDisk(runId, key, bytes); + return { inline: null, storageRef, size: bytes.byteLength, mime: file.type || null }; } + return this.storeText(runId, key, await file.text(), file.type || null); + } + + private async storeText( + runId: string, + key: string, + content: string, + mime: string | null, + ): Promise { + const size = Buffer.byteLength(content); + if (size === 0) return EMPTY; + if (size <= INLINE_LIMIT) return { inline: content, storageRef: null, size, mime }; + const storageRef = await this.writeToDisk(runId, key, content); + return { inline: null, storageRef, size, mime }; + } + + private async writeToDisk( + runId: string, + key: string, + data: string | Uint8Array, + ): Promise { const dir = path.join(STORAGE_ROOT, runId); await mkdir(dir, { recursive: true }); const storageRef = path.join(dir, key); - await Bun.write(storageRef, content); - return { inline: null, storageRef, size, mime }; + await Bun.write(storageRef, data); + return storageRef; } } diff --git a/apps/worker/src/deps/workspace.ts b/apps/worker/src/deps/workspace.ts index 548287c..2604c0d 100644 --- a/apps/worker/src/deps/workspace.ts +++ b/apps/worker/src/deps/workspace.ts @@ -110,6 +110,13 @@ export class GitWorkspaceManager implements WorkspaceManager { } } + async clearArtifacts(runId: string): Promise { + await rm(path.join(this.dirFor(runId), ".agrippa", "artifacts"), { + recursive: true, + force: true, + }); + } + async cleanup(runId: string): Promise { if (process.env.AGRIPPA_KEEP_WORKSPACES === "1") return; await rm(this.dirFor(runId), { recursive: true, force: true }); diff --git a/docs/adr/0009-security-correctness-deep-modules.md b/docs/adr/0009-security-correctness-deep-modules.md index cc94062..1a8c707 100644 --- a/docs/adr/0009-security-correctness-deep-modules.md +++ b/docs/adr/0009-security-correctness-deep-modules.md @@ -15,7 +15,7 @@ An M1 code review found that several documented invariants were declared but not Concentrate each concern behind one deep module whose interface is the test surface, rather than re-litigating the accepted ADRs (Bun, Drizzle, pg-boss, SSE): 1. **Execution-isolation seam** (`packages/executor-core/isolation.ts`) — `evaluateToolCall`, `isWithin`, and `buildScrubbedEnv` are pure and back both the SDK adapter and its tests. The adapter must route every tool decision and the subprocess environment through it, and layers OS-level controls on top (the SDK `sandbox`, a non-root worker, repo-config stripping at checkout). -2. **Authorized run-manifest** — `resolve.authorizeResources` pins the exact skills/MCP a run may use (required grants enforced, optional included only when granted) into `runs.resource_manifest` at submit; `verifyRepoRefs` enforces repo-connection ownership. The engine resolves resources only from the manifest, never the global registry, and the worker loads repo connections scoped to the run's project. +2. **Authorized run-manifest** — `resolve.authorizeResources` pins which skills/MCP a run may use (required grants enforced, optional included only when granted) into `runs.resource_manifest` at submit; `verifyRepoRefs` enforces repo-connection ownership. The engine resolves resources only from the manifest, never the global registry, and the worker loads repo connections scoped to the run's project. The manifest pins authorized *slugs*, not exact skill versions or MCP config revisions — version/revision pinning (so a registry edit can't change what an in-flight run resolves) is future work. 3. **Run-lifecycle module** (`packages/orchestration/src/engine/run-lifecycle.ts`) — `transitionRun` (compare-and-swap on the expected status), `appendRunEvent` (database-allocated per-run seq), and `decideApproval` (CAS on `pending`) own every lifecycle mutation for both the API and the worker. ## Alternatives considered @@ -25,6 +25,7 @@ Concentrate each concern behind one deep module whose interface is the test surf ## Consequences -- The static isolation layer contains file writes and refuses shell in read-only workspaces, but it cannot bound arbitrary writes a shell command makes in a read-write workspace — that remains the OS sandbox / non-root worker / container's job, and full container-level isolation is still future work. -- Adds a `runs.resource_manifest` column (migration 0002); retries and resumes carry the pinned manifest, so authorization can't drift after submit. -- Grants now genuinely gate optional resources: an optional skill/MCP with no project grant is treated as unavailable, so its dependent step is skipped rather than silently privileged. +- The static isolation layer contains file writes **and reads** to the workspace, refuses shell in read-only workspaces, and redacts known secret values from event payloads. It still cannot bound what a shell command reads or writes in a read-write workspace, nor isolate one run from another at the OS level (all runs share one worker UID and `/work`), nor keep the provider key out of the agent subprocess — those remain the OS sandbox / non-root worker / per-run container + token-proxy's job, which is **explicitly deferred** as a follow-up epic. +- The run-lifecycle module allocates event seq from an atomic per-run counter (`runs.next_event_seq`, migration 0003) so it is collision-free and works inside a caller's transaction; `finalize` commits status + finishedAt/totals + terminal event together. A true execution *lease* (so two at-least-once deliveries can't both resume a `running` run) is still future work. +- Adds a `runs.resource_manifest` column (migration 0002); retries and resumes carry the pinned manifest, so authorization can't drift after submit. Existing runs from before the migration must be backfilled (`scripts/backfill-manifest.ts`). +- Grants now genuinely gate optional resources: an optional skill/MCP with no project grant is treated as unavailable, so its dependent step (whether via `requires.mcpServers` or `requires.skills`) is skipped rather than silently privileged. diff --git a/docs/design/03-executor-abstraction.md b/docs/design/03-executor-abstraction.md index 63f5209..6faaffe 100644 --- a/docs/design/03-executor-abstraction.md +++ b/docs/design/03-executor-abstraction.md @@ -101,9 +101,9 @@ All executor work happens in the **worker container** (`apps/worker`), one run p - Each run gets a throwaway workspace `/work/runs/`, deleted after terminal state (configurable retention for debugging). - Git credentials are injected per-run into the clone URL and scrubbed from the remote immediately afterward; they never persist in `.git/config`. They are still passed as a clone argument today — moving to a workspace-scoped credential helper is follow-up work. -- Tool policy is enforced for **every** write-capable tool, Bash included, with a boundary-safe containment check (not a `startsWith` prefix): `readOnly` workspaces deny shell and confine writes to `.agrippa/artifacts`; `readWrite` workspaces confine writes to the workspace. The static layer cannot bound arbitrary writes a shell command makes in a read-write workspace — that is the OS sandbox's job (below). +- Tool policy is enforced for **every** file-touching tool — writes (Write/Edit/NotebookEdit) *and* reads (Read/Grep/Glob) — with a boundary-safe containment check (not a `startsWith` prefix) plus a symlink-real check: `readOnly` workspaces deny shell and confine writes to `.agrippa/artifacts`; reads and writes are confined to the workspace, so the agent can't `Read /proc/self/environ`, another run's `/work/runs/`, or the shared artifact store. The static layer cannot bound what a shell command reads or writes in a read-write workspace — that is the OS sandbox's job (below). - The agent subprocess runs with a **scrubbed environment** (`buildScrubbedEnv`): the master `AGRIPPA_SECRET_KEY` and datastore URLs are removed while the Anthropic auth vars the SDK needs are kept. The SDK `sandbox` (bubblewrap) is enabled where the host supports it, `strictMcpConfig` ignores repo `.mcp.json`, and the worker strips repo-supplied `.claude` settings/hooks at checkout so a checked-out repository can't inject hooks or permission overrides. The worker image runs as a non-root user. -- MCP secrets resolve lazily at server spawn and are not logged; `run_events` payloads are scrubbed against known secret values before persistence. +- MCP secrets resolve lazily at server spawn and are not logged; `run_events` payloads are redacted against known secret values (the provider key, resolved MCP tokens) before they are persisted or streamed (`SecretRedactor`). Note: the provider `ANTHROPIC_API_KEY` still lives in the agent subprocess env (the SDK needs it) and one worker UID is shared across runs — keeping the key out of the subprocess and isolating runs from each other require the container layer below. **Explicitly deferred**: per-run container/micro-VM isolation and a fully non-root, network-egress-restricted sandbox. The isolation seam localizes this — the engine hands the executor a `workspaceDir`, an `access` mode, and a signal; whether that directory lives in the worker's filesystem or a jailed container is invisible above the interface. The static containment plus env-scrub plus OS sandbox is adequate for a trusted org running semi-trusted repositories; hostile multi-tenant inputs need the container layer, which is risk #2 in [00-overview](00-overview.md). diff --git a/docs/design/04-execution-runtime.md b/docs/design/04-execution-runtime.md index 3527275..4330736 100644 --- a/docs/design/04-execution-runtime.md +++ b/docs/design/04-execution-runtime.md @@ -17,7 +17,7 @@ The enqueue is a post-commit send, so a narrow dual-write window exists (a crash ## Run State Machine -Pure function in `@agrippa/core` (`transition(state, event) → state | error`); every transition is persisted and audited. The persist step is a **compare-and-swap** on the expected `from` status (`run-lifecycle.transitionRun`), so a late worker finalize can't overwrite a status another path (e.g. a concurrent cancel) already moved on from — the loser of the race simply doesn't write. +Pure function in `@agrippa/core` (`transition(state, event) → state | error`); every transition is persisted and audited. The persist step is a **compare-and-swap** on the expected `from` status (`run-lifecycle.transitionRun`), so a late worker finalize can't overwrite a status another path (e.g. a concurrent cancel) already moved on from — the loser of the race simply doesn't write. Finalization commits the status change, `finishedAt`/`usageTotals`, and the terminal event in **one transaction** (publishing to the bus only after commit), so a crash can't leave a terminal run missing its totals or event; the retry-exhaustion path also goes through the CAS. ``` ┌────────────────────────────┐ @@ -98,7 +98,7 @@ Two independent layers, both enforced: ## Live Progress (SSE) -Ordering rule: the engine writes `run_events` **first** — the per-run monotonic `seq` is allocated by the database inside the INSERT (`run-lifecycle.appendRunEvent`), not from an in-memory `max+1` that a concurrent writer could collide with — then publishes the same event to Redis `run:{id}:events`. +Ordering rule: the engine writes `run_events` **first** — the per-run monotonic `seq` comes from an atomic counter (`runs.next_event_seq`, allocated by `UPDATE … RETURNING` in `run-lifecycle.appendRunEvent`), so it is collision-free and works inside a caller's transaction (the approval decision, which appends its event in the same tx as the decision) — then publishes the same event to Redis `run:{id}:events`. `GET /runs/:id/events` (SSE): @@ -107,4 +107,4 @@ Ordering rule: the engine writes `run_events` **first** — the per-run monotoni 3. flush the buffer, deduplicating by `seq` against the replay, 4. emit each as `id: \nevent: \ndata: `. -Subscribing **before** replaying is what makes reconnection gap-free by construction: an event committed and published in the window between replay and subscribe would otherwise be delivered only at the terminal replay. Redis is optional: with a bus, live events push instantly; without one, the stream falls back to periodic DB replay, which preserves correctness (just with a small latency). Either way, if Redis is briefly down, clients reconnect and replay from Postgres. +Subscribing (and **awaiting** the subscription is live — for Redis, the SUBSCRIBE ack) **before** replaying is what makes reconnection gap-free: an event committed and published in the window between replay and subscribe would otherwise be delivered only at the terminal replay. The bus branch also re-replays Postgres periodically, not only at terminal, so a dropped pub/sub message is recovered mid-run. Redis is optional: with a bus, live events push instantly; without one, the stream falls back to periodic DB replay, which preserves correctness (just with a small latency). Either way, if Redis is briefly down, clients reconnect and replay from Postgres. diff --git a/infra/docker-compose.yml b/infra/docker-compose.yml index e1bbb80..1977101 100644 --- a/infra/docker-compose.yml +++ b/infra/docker-compose.yml @@ -16,6 +16,11 @@ services: AGRIPPA_BASE_URL: ${AGRIPPA_BASE_URL:-http://localhost:3000} AGRIPPA_SECRET_KEY: ${AGRIPPA_SECRET_KEY:?set AGRIPPA_SECRET_KEY (openssl rand -base64 32)} BETTER_AUTH_SECRET: ${BETTER_AUTH_SECRET:?set BETTER_AUTH_SECRET (openssl rand -base64 32)} + # the API chooses the executor at submit, so it must see the same setting + AGRIPPA_EXECUTOR: ${AGRIPPA_EXECUTOR:-claude-agent-sdk} + volumes: + # the API serves artifact downloads, incl. large ones stored on the volume + - workdata:/work depends_on: postgres: condition: service_healthy diff --git a/packages/orchestration/src/compile.test.ts b/packages/orchestration/src/compile.test.ts index c626c65..ea0a011 100644 --- a/packages/orchestration/src/compile.test.ts +++ b/packages/orchestration/src/compile.test.ts @@ -94,6 +94,13 @@ describe("template compiler", () => { ); }); + it("rejects a requires.skills reference to an unknown skill", () => { + const source = mutate((doc) => { + doc.spec.phases[1].steps[0].requires = { skills: ["nonexistent-skill"] }; + }); + expect(issuesOf(source).join()).toContain("requires unknown skill 'nonexistent-skill'"); + }); + it("rejects references to steps that are not defined earlier", () => { const source = mutate((doc) => { doc.spec.phases[0].steps[1].instructions = "look at ${steps.summarize.outputs.x}"; diff --git a/packages/orchestration/src/compile.ts b/packages/orchestration/src/compile.ts index 1777fdb..e54beb3 100644 --- a/packages/orchestration/src/compile.ts +++ b/packages/orchestration/src/compile.ts @@ -93,6 +93,10 @@ export function compileTemplate(sourceYaml: string, options: CompileOptions = {} for (const ref of step.requires?.mcpServers ?? []) { if (!mcpRefs.has(ref)) issues.push(`${where}: requires unknown mcp server '${ref}'`); } + for (const ref of step.requires?.skills ?? []) { + if (!skillSlugs.has(skillSlugOfRef(ref))) + issues.push(`${where}: requires unknown skill '${ref}'`); + } for (const key of step.produces) { if (!artifactKeys.has(key)) { issues.push(`${where}: produces '${key}' which is not in outputs.artifacts`); diff --git a/packages/orchestration/src/engine/bus.ts b/packages/orchestration/src/engine/bus.ts index 8f79a52..b807aeb 100644 --- a/packages/orchestration/src/engine/bus.ts +++ b/packages/orchestration/src/engine/bus.ts @@ -11,9 +11,17 @@ export type BusEvent = { createdAt: string; }; +/** + * A live subscription. `ready` resolves once the underlying transport has + * actually begun delivering (for Redis, once SUBSCRIBE is acknowledged) — the + * SSE handler awaits it before replaying history so no event is lost in the gap + * between replay and an as-yet-inactive subscription. + */ +export type Subscription = { unsubscribe: () => void; ready: Promise }; + export interface RunEventBus { publish(event: BusEvent): Promise; - subscribe(runId: string, listener: (event: BusEvent) => void): () => void; + subscribe(runId: string, listener: (event: BusEvent) => void): Subscription; /** Control channel — today only "cancel". */ publishControl(runId: string, message: string): Promise; subscribeControl(runId: string, listener: (message: string) => void): () => void; @@ -28,11 +36,11 @@ export class InProcessEventBus implements RunEventBus { for (const listener of this.listeners.get(event.runId) ?? []) listener(event); } - subscribe(runId: string, listener: (event: BusEvent) => void): () => void { + subscribe(runId: string, listener: (event: BusEvent) => void): Subscription { const set = this.listeners.get(runId) ?? new Set(); set.add(listener); this.listeners.set(runId, set); - return () => set.delete(listener); + return { unsubscribe: () => set.delete(listener), ready: Promise.resolve() }; } async publishControl(runId: string, message: string): Promise { diff --git a/packages/orchestration/src/engine/deps.ts b/packages/orchestration/src/engine/deps.ts index 5067dee..85533f1 100644 --- a/packages/orchestration/src/engine/deps.ts +++ b/packages/orchestration/src/engine/deps.ts @@ -24,6 +24,9 @@ export interface WorkspaceManager { checkout(runId: string, spec: WorkspaceSpec): Promise; /** git diff against the checkout base — engine-side patch artifacts. */ diff(runId: string): Promise; + /** Empty the artifact convention dir before an attempt, so a failed attempt's + * stale file can't be re-collected by a later successful attempt. */ + clearArtifacts(runId: string): Promise; cleanup(runId: string): Promise; } diff --git a/packages/orchestration/src/engine/engine.integration.test.ts b/packages/orchestration/src/engine/engine.integration.test.ts index 815ed0f..9c06fc7 100644 --- a/packages/orchestration/src/engine/engine.integration.test.ts +++ b/packages/orchestration/src/engine/engine.integration.test.ts @@ -596,15 +596,23 @@ describe.skipIf(!dbUp)("orchestration engine (FakeExecutor compliance suite)", ( it("streams live events over the bus while executing", async () => { const { runId, makeDeps, bus } = await setupFixture(); const seen: string[] = []; - const unsubscribe = bus.subscribe(runId, (event) => seen.push(event.type)); + const subscription = bus.subscribe(runId, (event) => seen.push(event.type)); await executeRun(makeDeps(HAPPY_SCRIPT), runId); - unsubscribe(); + subscription.unsubscribe(); expect(seen[0]).toBe("run.started"); expect(seen).toContain("step.started"); expect(seen).toContain("usage"); expect(seen).toContain("approval.required"); }); + it("clears the artifact dir before each agent step attempt", async () => { + const { runId, makeDeps, workspace } = await setupFixture(); + await executeRun(makeDeps(HAPPY_SCRIPT), runId); + // agent steps ran → clearArtifacts was invoked (so a stale attempt file + // can't be re-collected as a later attempt's result) + expect(workspace.cleared).toContain(runId); + }); + it("redacts known secret values from persisted events", async () => { const secret = "sk-ant-supersecretvalue-1234567890"; const prev = process.env.ANTHROPIC_API_KEY; diff --git a/packages/orchestration/src/engine/engine.ts b/packages/orchestration/src/engine/engine.ts index 2b55cf3..3a20399 100644 --- a/packages/orchestration/src/engine/engine.ts +++ b/packages/orchestration/src/engine/engine.ts @@ -250,6 +250,18 @@ class RunEngine { }) .from(tokenUsage) .where(eq(tokenUsage.runId, run.id)); + // per-phase spend, rebuilt by phase so per-phase budgets survive a resume + // (usage rows carry a step id; run_steps carries the phase) + const perPhaseRows = await db + .select({ + phaseId: runSteps.phaseId, + cost: sql`coalesce(sum(${tokenUsage.costUsd}), 0)`, + }) + .from(tokenUsage) + .innerJoin(runSteps, eq(tokenUsage.stepId, runSteps.id)) + .where(eq(tokenUsage.runId, run.id)) + .groupBy(runSteps.phaseId); + const perPhaseSpent = Object.fromEntries(perPhaseRows.map((r) => [r.phaseId, Number(r.cost)])); const budgets = this.template.spec.budgets; const quota = await this.quotaHeadroom(); this.meter = new BudgetMeter( @@ -261,7 +273,11 @@ class RunEngine { quotaCostUsd: quota.costUsd, quotaTokens: quota.tokens, }, - { costUsd: Number(usageTotals?.cost ?? 0), tokens: Number(usageTotals?.tokens ?? 0) }, + { + costUsd: Number(usageTotals?.cost ?? 0), + tokens: Number(usageTotals?.tokens ?? 0), + perPhaseCostUsd: perPhaseSpent, + }, ); // duration budget survives resume: deadline anchors to the original start @@ -439,10 +455,14 @@ class RunEngine { return; } if (step.requires) { - const authorized = this.authorizedMcpRefs(step.requires.mcpServers); - const ungranted = step.requires.mcpServers.filter((ref) => !authorized.includes(ref)); - const { missing } = await this.deps.resources.mcpServers(authorized); - const unavailable = [...ungranted, ...missing]; + const authorizedMcp = this.authorizedMcpRefs(step.requires.mcpServers); + const ungrantedMcp = step.requires.mcpServers.filter((ref) => !authorizedMcp.includes(ref)); + const { missing } = await this.deps.resources.mcpServers(authorizedMcp); + // a required skill the run isn't authorized for is also unmet — otherwise + // the step would run without the skill it declared it needs + const authorizedSkills = this.authorizedSkillRefs(step.requires.skills); + const ungrantedSkills = step.requires.skills.filter((ref) => !authorizedSkills.includes(ref)); + const unavailable = [...ungrantedMcp, ...missing, ...ungrantedSkills]; if (unavailable.length > 0) { await this.markSkipped( phase, @@ -464,6 +484,9 @@ class RunEngine { await this.runSystemStep(step); await this.completeStep(row, ""); } else { + // start each attempt from a clean artifact dir so a prior attempt's + // stale file isn't collected as this attempt's result + await this.deps.workspace.clearArtifacts(this.run.id); const output = await this.runAgentStep(phase, step, row, attempt); await this.completeStep(row, output); } @@ -743,9 +766,10 @@ class RunEngine { { inline: event.inline, path: event.path }, this.workspaceDir, ); - // a missing/empty source produced no bytes — don't create a zero-byte row - // (and don't mark the key produced, so a required artifact still fails) - if (stored.inline === null && stored.storageRef === null) return; + // a missing OR empty source produced no bytes — don't create a zero-byte row + // (and don't mark the key produced, so a required-but-empty artifact still + // fails the contract rather than passing it) + if (stored.size === 0 && stored.storageRef === null) return; await this.db.insert(artifacts).values({ runId: this.run.id, stepId: row.id, diff --git a/packages/orchestration/src/engine/fakes.ts b/packages/orchestration/src/engine/fakes.ts index a95462d..54d4c99 100644 --- a/packages/orchestration/src/engine/fakes.ts +++ b/packages/orchestration/src/engine/fakes.ts @@ -1,4 +1,4 @@ -import { mkdtempSync } from "node:fs"; +import { mkdtempSync, rmSync } from "node:fs"; import { tmpdir } from "node:os"; import path from "node:path"; import type { Logger, ResolvedMcpServer, ResolvedSkill } from "@agrippa/executor-core"; @@ -29,6 +29,13 @@ export class FakeWorkspaceManager implements WorkspaceManager { return this.diffOutput; } + readonly cleared: string[] = []; + async clearArtifacts(runId: string): Promise { + this.cleared.push(runId); + const dir = this.dirs.get(runId); + if (dir) rmSync(path.join(dir, ".agrippa", "artifacts"), { recursive: true, force: true }); + } + async cleanup(runId: string): Promise { this.cleaned.push(runId); this.dirs.delete(runId); diff --git a/packages/orchestration/src/engine/redis-bus.ts b/packages/orchestration/src/engine/redis-bus.ts index d94c338..2d47221 100644 --- a/packages/orchestration/src/engine/redis-bus.ts +++ b/packages/orchestration/src/engine/redis-bus.ts @@ -1,5 +1,5 @@ import { Redis } from "ioredis"; -import type { BusEvent, RunEventBus } from "./bus"; +import type { BusEvent, RunEventBus, Subscription } from "./bus"; const eventChannel = (runId: string) => `run:${runId}:events`; const controlChannel = (runId: string) => `run:${runId}:control`; @@ -39,17 +39,25 @@ export class RedisEventBus implements RunEventBus { } } - subscribe(runId: string, listener: (event: BusEvent) => void): () => void { + subscribe(runId: string, listener: (event: BusEvent) => void): Subscription { const set = this.listeners.get(runId) ?? new Set(); set.add(listener); this.listeners.set(runId, set); - void this.sub.subscribe(eventChannel(runId)); - return () => { - set.delete(listener); - if (set.size === 0) { - this.listeners.delete(runId); - void this.sub.unsubscribe(eventChannel(runId)); - } + // `ready` resolves once Redis acknowledges SUBSCRIBE; the SSE handler awaits + // it before replaying so an event published in that window isn't lost + const ready = this.sub + .subscribe(eventChannel(runId)) + .then(() => undefined) + .catch(() => undefined); // Redis down → SSE falls back to DB replay + return { + unsubscribe: () => { + set.delete(listener); + if (set.size === 0) { + this.listeners.delete(runId); + void this.sub.unsubscribe(eventChannel(runId)); + } + }, + ready, }; } diff --git a/scripts/backfill-manifest.ts b/scripts/backfill-manifest.ts new file mode 100644 index 0000000..e045fe0 --- /dev/null +++ b/scripts/backfill-manifest.ts @@ -0,0 +1,56 @@ +import { isTerminalRunStatus, type RunStatus } from "@agrippa/core"; +import { createDb, runs, templateVersions } from "@agrippa/db"; +import type { TemplateDoc } from "@agrippa/orchestration"; +import { eq } from "drizzle-orm"; + +/** + * Backfill runs.resource_manifest for runs that predate migration 0002. + * + * Migration 0002 defaults the manifest to `{}`; because the engine resolves + * skills/MCP only from the manifest, a run that was queued or in flight before + * the migration would silently lose its resources. This authorizes each such + * (non-terminal) run's template-declared skills/MCP, matching the pre-manifest + * behavior. Terminal runs never re-execute, so their empty manifest is harmless + * and left alone. Idempotent: runs with a non-empty manifest are skipped. + * + * Run this once during any upgrade of an instance that has live runs: + * bun scripts/backfill-manifest.ts + */ +const db = createDb(); + +const rows = await db + .select({ + id: runs.id, + status: runs.status, + resourceManifest: runs.resourceManifest, + templateVersionId: runs.templateVersionId, + }) + .from(runs); + +let backfilled = 0; +for (const run of rows) { + if (isTerminalRunStatus(run.status as RunStatus)) continue; + if (run.resourceManifest.mcpServers.length > 0 || run.resourceManifest.skills.length > 0) + continue; + + const [version] = await db + .select({ compiled: templateVersions.compiled }) + .from(templateVersions) + .where(eq(templateVersions.id, run.templateVersionId)); + if (!version) continue; + const template = version.compiled as unknown as TemplateDoc; + + await db + .update(runs) + .set({ + resourceManifest: { + mcpServers: template.spec.resources.mcpServers.map((m) => m.ref), + skills: template.spec.resources.skills.map((s) => s.ref.split("@")[0] as string), + }, + }) + .where(eq(runs.id, run.id)); + backfilled += 1; +} + +console.log(`[backfill-manifest] updated ${backfilled} non-terminal run(s)`); +process.exit(0); From 509a5c7b3bcc62e1e7e178f4f10eb1faf609a59e Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 19:20:44 +0800 Subject: [PATCH 15/18] fix: safe artifact-attempt cleanup + unified run-lifecycle finalization MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Round 4 (Codex re-review) — commit 1 of 2: the critical data-loss bug and the finalization races, both by deepening the seams rather than patching around them. Artifact-attempt (R4-A): - CRITICAL: `clearArtifacts` did a recursive `rm` of `/.agrippa/artifacts`, and `.agrippa` wasn't stripped at checkout — a committed `.agrippa -> /work` symlink turned that into deletion of the shared artifact store (Codex reproduced it). Strip `.agrippa` at checkout, remove the whole-dir clear entirely, and instead have the executor delete only *this step's* expected files at attempt start — refusing to act if the artifact dir resolves outside the workspace (an agent-created symlink). Fixes the stale-attempt bug without cross-step deletion or the escape. - `DiskArtifactStore` stats the source before reading and rejects anything over a size cap, then streams accepted files to disk instead of buffering whole files into memory (OOM guard). Run-lifecycle (R4-B): - One `finalizeRun` in the lifecycle module owns terminal transitions: a single tx does the status CAS (optionally requiring cancel_requested=false), finishedAt/totals, and the terminal event. The engine's late-cancel check is now atomic (a cancel committed after the read still wins), and `markRunFailed` routes through it — so a queued run whose setup threw transitions queued→failed (now a legal transition) instead of throwing on an illegal one and stranding the run, and both paths always emit the terminal event. --- apps/worker/src/deps/artifacts.test.ts | 16 ++++ apps/worker/src/deps/artifacts.ts | 33 ++++++-- apps/worker/src/deps/workspace.ts | 21 ++--- apps/worker/src/index.ts | 25 +++--- packages/core/src/run-status.test.ts | 1 + packages/core/src/run-status.ts | 2 +- packages/executor-claude/src/executor.test.ts | 75 ++++++++++++++--- packages/executor-claude/src/executor.ts | 30 ++++++- packages/orchestration/src/engine/deps.ts | 3 - .../src/engine/engine.integration.test.ts | 60 ++++++++++++-- packages/orchestration/src/engine/engine.ts | 83 +++++++++---------- packages/orchestration/src/engine/fakes.ts | 9 +- .../orchestration/src/engine/run-lifecycle.ts | 53 +++++++++++- 13 files changed, 298 insertions(+), 113 deletions(-) diff --git a/apps/worker/src/deps/artifacts.test.ts b/apps/worker/src/deps/artifacts.test.ts index c5b3bcb..fab4257 100644 --- a/apps/worker/src/deps/artifacts.test.ts +++ b/apps/worker/src/deps/artifacts.test.ts @@ -103,4 +103,20 @@ describe("DiskArtifactStore path containment", () => { const round = new Uint8Array(await Bun.file(stored.storageRef as string).arrayBuffer()); expect([...round]).toEqual([...bytes]); // byte-exact, no UTF-8 corruption }); + + it("rejects an artifact over the size cap without buffering it", async () => { + const ws = freshWorkspace(); + await mkdir(path.join(ws, ".agrippa/artifacts"), { recursive: true }); + writeFileSync(path.join(ws, ".agrippa/artifacts/big.md"), "0123456789AB"); // 12 bytes + const prev = process.env.AGRIPPA_MAX_ARTIFACT_BYTES; + process.env.AGRIPPA_MAX_ARTIFACT_BYTES = "8"; + try { + await expect( + store.store("run-1", "big", "markdown", { path: ".agrippa/artifacts/big.md" }, ws), + ).rejects.toThrow(/over the .* limit/); + } finally { + if (prev === undefined) delete process.env.AGRIPPA_MAX_ARTIFACT_BYTES; + else process.env.AGRIPPA_MAX_ARTIFACT_BYTES = prev; + } + }); }); diff --git a/apps/worker/src/deps/artifacts.ts b/apps/worker/src/deps/artifacts.ts index 7fe10a8..0c15db8 100644 --- a/apps/worker/src/deps/artifacts.ts +++ b/apps/worker/src/deps/artifacts.ts @@ -7,6 +7,16 @@ import type { ArtifactStore, StoredArtifact } from "@agrippa/orchestration"; const STORAGE_ROOT = process.env.ARTIFACT_STORAGE_ROOT ?? path.join(tmpdir(), "agrippa-artifacts"); const INLINE_LIMIT = 64 * 1024; +/** Hard cap on a single artifact so an agent can't OOM the worker (env-tunable). */ +const maxArtifactSize = (): number => + Number(process.env.AGRIPPA_MAX_ARTIFACT_BYTES ?? 25 * 1024 * 1024); + +class ArtifactTooLargeError extends Error { + constructor(key: string, size: number) { + super(`artifact '${key}' is ${size} bytes, over the ${maxArtifactSize()}-byte limit`); + this.name = "ArtifactTooLargeError"; + } +} const EMPTY: StoredArtifact = { inline: null, storageRef: null, size: 0, mime: null }; @@ -54,15 +64,19 @@ export class DiskArtifactStore implements ArtifactStore { const file = Bun.file(real); if (!(await file.exists())) return EMPTY; - // `file`-kind artifacts may be binary — read raw bytes and stream them on - // download rather than decoding to UTF-8 (which corrupts non-text content) - if (kind === "file") { - const bytes = new Uint8Array(await file.arrayBuffer()); - if (bytes.byteLength === 0) return EMPTY; - const storageRef = await this.writeToDisk(runId, key, bytes); - return { inline: null, storageRef, size: bytes.byteLength, mime: file.type || null }; + // stat BEFORE reading, so a huge/sparse file is rejected without buffering it + const size = file.size; + if (size === 0) return EMPTY; + if (size > maxArtifactSize()) throw new ArtifactTooLargeError(key, size); + const mime = file.type || null; + + // small text can inline in Postgres; `file`-kind (possibly binary) and any + // large artifact stream straight to disk byte-exact, never fully buffered + if (kind !== "file" && size <= INLINE_LIMIT) { + return this.storeText(runId, key, await file.text(), mime); } - return this.storeText(runId, key, await file.text(), file.type || null); + const storageRef = await this.writeToDisk(runId, key, file); + return { inline: null, storageRef, size, mime }; } private async storeText( @@ -73,6 +87,7 @@ export class DiskArtifactStore implements ArtifactStore { ): Promise { const size = Buffer.byteLength(content); if (size === 0) return EMPTY; + if (size > maxArtifactSize()) throw new ArtifactTooLargeError(key, size); if (size <= INLINE_LIMIT) return { inline: content, storageRef: null, size, mime }; const storageRef = await this.writeToDisk(runId, key, content); return { inline: null, storageRef, size, mime }; @@ -81,7 +96,7 @@ export class DiskArtifactStore implements ArtifactStore { private async writeToDisk( runId: string, key: string, - data: string | Uint8Array, + data: string | Uint8Array | Blob, ): Promise { const dir = path.join(STORAGE_ROOT, runId); await mkdir(dir, { recursive: true }); diff --git a/apps/worker/src/deps/workspace.ts b/apps/worker/src/deps/workspace.ts index 2604c0d..79d80a9 100644 --- a/apps/worker/src/deps/workspace.ts +++ b/apps/worker/src/deps/workspace.ts @@ -8,13 +8,15 @@ import { and, eq } from "drizzle-orm"; const WORKSPACE_ROOT = process.env.WORKSPACE_ROOT ?? path.join(tmpdir(), "agrippa-workspaces"); /** - * Repo-supplied agent configuration that would otherwise be honored by the SDK - * project setting source: hooks run shell, settings grant tool permissions, and - * .mcp.json wires servers. A checked-out repo is untrusted, so these are removed - * before any agent runs. Registry skills are re-materialized into .claude/skills - * afterwards (docs/design/03 §Sandboxing). + * Repo-supplied paths removed before any agent runs (a checked-out repo is + * untrusted): `.claude`/`.mcp.json` would be honored by the SDK project setting + * source (hooks run shell, settings grant permissions, .mcp.json wires servers); + * `.agrippa` is the platform's own artifact convention dir — a committed + * `.agrippa -> /work` symlink would otherwise let workspace-relative artifact + * paths escape to the shared store. Registry skills and the artifact dir are + * re-created fresh afterwards (docs/design/03 §Sandboxing). */ -const REPO_CONFIG_TO_STRIP = [".claude", ".mcp.json"]; +const REPO_CONFIG_TO_STRIP = [".claude", ".mcp.json", ".agrippa"]; async function sanitizeWorkspace(dir: string): Promise { for (const entry of REPO_CONFIG_TO_STRIP) { @@ -110,13 +112,6 @@ export class GitWorkspaceManager implements WorkspaceManager { } } - async clearArtifacts(runId: string): Promise { - await rm(path.join(this.dirFor(runId), ".agrippa", "artifacts"), { - recursive: true, - force: true, - }); - } - async cleanup(runId: string): Promise { if (process.env.AGRIPPA_KEEP_WORKSPACES === "1") return; await rm(this.dirFor(runId), { recursive: true, force: true }); diff --git a/apps/worker/src/index.ts b/apps/worker/src/index.ts index 096efd4..f0a1bdb 100644 --- a/apps/worker/src/index.ts +++ b/apps/worker/src/index.ts @@ -13,10 +13,10 @@ import { durationToMinutes, type EngineDeps, executeRun, + finalizeRun, findStrandedApprovalRuns, InProcessEventBus, RedisEventBus, - transitionRun, } from "@agrippa/orchestration"; import { and, eq, lt, sql } from "drizzle-orm"; import type { Job, JobWithMetadata } from "pg-boss"; @@ -106,17 +106,18 @@ async function scheduleApprovalExpiry(runId: string): Promise { async function markRunFailed(runId: string, err: unknown): Promise { const [run] = await db.select({ status: runs.status }).from(runs).where(eq(runs.id, runId)); if (!run || isTerminalRunStatus(run.status)) return; - // route through the lifecycle CAS so a concurrent transition (e.g. a cancel or - // a resumed worker finalizing) can't be clobbered by this retry-exhaustion path - await db.transaction(async (tx) => { - if (!(await transitionRun(tx, runId, run.status, "failed"))) return; - await tx - .update(runs) - .set({ - finishedAt: new Date(), - error: { code: "internal", message: `retries exhausted: ${String(err).slice(0, 500)}` }, - }) - .where(eq(runs.id, runId)); + // one shared finalization impl: CAS from the *current* status (so a queued run + // whose setup threw before it was claimed transitions queued→failed, not the + // illegal queued→failed of a hard-coded from), and it emits the terminal event + // that the old id-only update omitted + const error = { code: "internal", message: `retries exhausted: ${String(err).slice(0, 500)}` }; + await finalizeRun(db, { + runId, + from: run.status, + to: "failed", + error, + usageTotals: {}, + eventPayload: { error }, }); } diff --git a/packages/core/src/run-status.test.ts b/packages/core/src/run-status.test.ts index 0c94a12..1e9b087 100644 --- a/packages/core/src/run-status.test.ts +++ b/packages/core/src/run-status.test.ts @@ -12,6 +12,7 @@ describe("run state machine", () => { const legal: Array<[RunStatus, RunStatus]> = [ ["queued", "running"], ["queued", "cancelled"], + ["queued", "failed"], // setup threw before the run was claimed ["running", "succeeded"], ["running", "failed"], ["running", "timed_out"], diff --git a/packages/core/src/run-status.ts b/packages/core/src/run-status.ts index 45d4a5d..b50384d 100644 --- a/packages/core/src/run-status.ts +++ b/packages/core/src/run-status.ts @@ -33,7 +33,7 @@ export type StepStatus = (typeof STEP_STATUSES)[number]; * not listed here is illegal and must be rejected. */ const LEGAL_TRANSITIONS: Readonly> = { - queued: ["running", "cancelled"], + queued: ["running", "cancelled", "failed"], running: ["succeeded", "failed", "timed_out", "waiting_approval", "cancelled"], waiting_approval: ["running", "cancelled", "failed"], succeeded: [], diff --git a/packages/executor-claude/src/executor.test.ts b/packages/executor-claude/src/executor.test.ts index 7ea1f98..1410b95 100644 --- a/packages/executor-claude/src/executor.test.ts +++ b/packages/executor-claude/src/executor.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { mkdirSync, mkdtempSync, writeFileSync } from "node:fs"; +import { existsSync, mkdirSync, mkdtempSync, symlinkSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import path from "node:path"; import type { ExecutionContext, ExecutorEvent, StepExecutionRequest } from "@agrippa/executor-core"; @@ -47,13 +47,20 @@ function makeRequest(overrides: Partial = {}): StepExecuti const sdk = (message: unknown) => message as SDKMessage; -function scriptedQuery(messages: unknown[], capture?: { options?: Options; prompt?: string }) { +function scriptedQuery( + messages: unknown[], + capture?: { options?: Options; prompt?: string }, + onStart?: () => void, +) { return (params: { prompt: string; options?: Options }) => { if (capture) { capture.options = params.options; capture.prompt = params.prompt; } return (async function* () { + // runs after the executor has cleared expected artifacts — this is where a + // real agent would write its files + onStart?.(); for (const message of messages) yield sdk(message); })(); }; @@ -165,10 +172,12 @@ describe("claude executor option mapping (docs/design/03)", () => { describe("claude executor event stream", () => { it("translates SDK messages into normalized executor events", async () => { const req = makeRequest(); - // agent wrote an artifact into the convention directory const artifactDir = path.join(req.workspaceDir, ".agrippa/artifacts"); - mkdirSync(artifactDir, { recursive: true }); - writeFileSync(path.join(artifactDir, "fix-report.md"), "# fixed"); + // the agent writes its artifact during the run (after the executor clears) + const writeArtifact = () => { + mkdirSync(artifactDir, { recursive: true }); + writeFileSync(path.join(artifactDir, "fix-report.md"), "# fixed"); + }; const messages = [ { type: "system", subtype: "init", session_id: "sess-1" }, @@ -201,7 +210,7 @@ describe("claude executor event stream", () => { { type: "result", subtype: "success", result: "All done.", is_error: false }, ]; - const executor = createClaudeExecutor(scriptedQuery(messages)); + const executor = createClaudeExecutor(scriptedQuery(messages, undefined, writeArtifact)); const events = await collect(executor.executeStep(req, makeCtx())); expect(events[0]).toEqual({ type: "step.started", sessionId: "sess-1" }); @@ -249,10 +258,11 @@ describe("claude executor event stream", () => { }); const artifactDir = path.join(req.workspaceDir, ".agrippa/artifacts"); mkdirSync(artifactDir, { recursive: true }); - writeFileSync(path.join(artifactDir, "fix-report.md"), "# fixed"); - // the agent also wrote the patch (engine generates it) and a stray file + // stray files that already exist (the patch, which the engine generates, and + // a scratch file); the agent writes fix-report during the run writeFileSync(path.join(artifactDir, "patch"), "diff --git a/x b/x"); writeFileSync(path.join(artifactDir, "scratch.txt"), "junk"); + const writeReport = () => writeFileSync(path.join(artifactDir, "fix-report.md"), "# fixed"); // patch instructions must not tell the agent to author the patch file const { prompt } = buildQueryArgs(req, makeCtx(), new AbortController()); @@ -260,10 +270,14 @@ describe("claude executor event stream", () => { expect(prompt).not.toContain(".agrippa/artifacts/patch"); const executor = createClaudeExecutor( - scriptedQuery([ - { type: "system", subtype: "init", session_id: "s" }, - { type: "result", subtype: "success", result: "done", is_error: false }, - ]), + scriptedQuery( + [ + { type: "system", subtype: "init", session_id: "s" }, + { type: "result", subtype: "success", result: "done", is_error: false }, + ], + undefined, + writeReport, + ), ); const events = await collect(executor.executeStep(req, makeCtx())); const artifacts = events.filter((e) => e.type === "artifact"); @@ -277,6 +291,43 @@ describe("claude executor event stream", () => { ]); }); + it("clears a stale expected artifact before the attempt runs", async () => { + const req = makeRequest(); // expects fix-report (markdown) + const artifactDir = path.join(req.workspaceDir, ".agrippa/artifacts"); + mkdirSync(artifactDir, { recursive: true }); + writeFileSync(path.join(artifactDir, "fix-report.md"), "# stale from attempt 1"); + + // this attempt's agent produces nothing + const executor = createClaudeExecutor( + scriptedQuery([ + { type: "system", subtype: "init", session_id: "s" }, + { type: "result", subtype: "success", result: "done", is_error: false }, + ]), + ); + const events = await collect(executor.executeStep(req, makeCtx())); + expect(events.filter((e) => e.type === "artifact")).toEqual([]); + expect(existsSync(path.join(artifactDir, "fix-report.md"))).toBe(false); + }); + + it("does not clear through a .agrippa symlink that escapes the workspace", async () => { + const req = makeRequest(); + // an agent pointed .agrippa at a shared store holding another run's artifact + const shared = mkdtempSync(path.join(tmpdir(), "shared-store-")); + mkdirSync(path.join(shared, "artifacts"), { recursive: true }); + const victim = path.join(shared, "artifacts", "fix-report.md"); + writeFileSync(victim, "another run's artifact"); + symlinkSync(shared, path.join(req.workspaceDir, ".agrippa")); + + const executor = createClaudeExecutor( + scriptedQuery([ + { type: "system", subtype: "init", session_id: "s" }, + { type: "result", subtype: "success", result: "done", is_error: false }, + ]), + ); + await collect(executor.executeStep(req, makeCtx())); + expect(existsSync(victim)).toBe(true); // shared store untouched + }); + it("maps SDK errors and aborts to step.failed", async () => { const failing = createClaudeExecutor( scriptedQuery([ diff --git a/packages/executor-claude/src/executor.ts b/packages/executor-claude/src/executor.ts index b5d2230..ed56d31 100644 --- a/packages/executor-claude/src/executor.ts +++ b/packages/executor-claude/src/executor.ts @@ -1,4 +1,4 @@ -import { existsSync } from "node:fs"; +import { existsSync, realpathSync, rmSync } from "node:fs"; import path from "node:path"; import { buildScrubbedEnv, @@ -7,6 +7,7 @@ import { type ExecutorEvent, evaluateToolCall, isReadTool, + isWithin, isWriteTool, pathArgOf, realContained, @@ -174,6 +175,28 @@ function* collectArtifacts( } } +/** Delete this step's expected artifact files (patch excluded) before an attempt. */ +function clearExpectedArtifacts( + workspaceDir: string, + expected: StepExecutionRequest["expectedArtifacts"], +): void { + const dir = path.join(workspaceDir, ARTIFACT_DIR); + // refuse to touch the dir if it resolves (via a symlink an agent may have + // created, e.g. `.agrippa -> /work`) outside the workspace — otherwise these + // unlinks would reach the shared artifact store + let realDir: string; + try { + realDir = realpathSync(dir); + } catch { + return; // missing dir → nothing to clear + } + if (!isWithin(realpathSync(workspaceDir), realDir)) return; + for (const a of expected) { + if (a.kind === "patch") continue; + rmSync(path.join(dir, expectedFilename(a)), { force: true }); + } +} + export function createClaudeExecutor(queryFn: QueryFn = sdkQuery as QueryFn): Executor { return { id: "claude-agent-sdk", @@ -188,6 +211,11 @@ export function createClaudeExecutor(queryFn: QueryFn = sdkQuery as QueryFn): Ex if (ctx.signal.aborted) onAbort(); ctx.signal.addEventListener("abort", onAbort, { once: true }); + // remove only THIS step's expected artifact files up front, so a prior + // attempt's stale file isn't collected as this attempt's result — without + // touching other steps' artifacts (or recursively deleting the dir) + clearExpectedArtifacts(req.workspaceDir, req.expectedArtifacts); + const { prompt, options } = buildQueryArgs(req, ctx, abortController); let started = false; let terminal: ExecutorEvent | null = null; diff --git a/packages/orchestration/src/engine/deps.ts b/packages/orchestration/src/engine/deps.ts index 85533f1..5067dee 100644 --- a/packages/orchestration/src/engine/deps.ts +++ b/packages/orchestration/src/engine/deps.ts @@ -24,9 +24,6 @@ export interface WorkspaceManager { checkout(runId: string, spec: WorkspaceSpec): Promise; /** git diff against the checkout base — engine-side patch artifacts. */ diff(runId: string): Promise; - /** Empty the artifact convention dir before an attempt, so a failed attempt's - * stale file can't be re-collected by a later successful attempt. */ - clearArtifacts(runId: string): Promise; cleanup(runId: string): Promise; } diff --git a/packages/orchestration/src/engine/engine.integration.test.ts b/packages/orchestration/src/engine/engine.integration.test.ts index 9c06fc7..95894f4 100644 --- a/packages/orchestration/src/engine/engine.integration.test.ts +++ b/packages/orchestration/src/engine/engine.integration.test.ts @@ -38,6 +38,7 @@ import { import { appendRunEvent, decideApproval, + finalizeRun, findStrandedApprovalRuns, transitionRun, } from "./run-lifecycle"; @@ -605,14 +606,6 @@ describe.skipIf(!dbUp)("orchestration engine (FakeExecutor compliance suite)", ( expect(seen).toContain("approval.required"); }); - it("clears the artifact dir before each agent step attempt", async () => { - const { runId, makeDeps, workspace } = await setupFixture(); - await executeRun(makeDeps(HAPPY_SCRIPT), runId); - // agent steps ran → clearArtifacts was invoked (so a stale attempt file - // can't be re-collected as a later attempt's result) - expect(workspace.cleared).toContain(runId); - }); - it("redacts known secret values from persisted events", async () => { const secret = "sk-ant-supersecretvalue-1234567890"; const prev = process.env.ANTHROPIC_API_KEY; @@ -705,6 +698,57 @@ describe.skipIf(!dbUp)("run-lifecycle module", () => { expect(await findStrandedApprovalRuns(db)).toContain(runId); }); + it("finalizeRun lets a late cancel win over a success (atomic, no read/CAS gap)", async () => { + const { db, runId } = await setupFixture(); + await transitionRun(db, runId, "queued", "running"); + await db.update(runs).set({ cancelRequested: true }).where(eq(runs.id, runId)); + // a success that requires no pending cancel is refused, leaving status running + const r = await finalizeRun(db, { + runId, + from: "running", + to: "succeeded", + requireNotCancelled: true, + error: null, + usageTotals: {}, + eventPayload: {}, + }); + expect(r.outcome).toBe("cancelled_instead"); + const [row] = await db.select({ status: runs.status }).from(runs).where(eq(runs.id, runId)); + expect(row?.status).toBe("running"); + // re-finalizing as cancelled commits + const c = await finalizeRun(db, { + runId, + from: "running", + to: "cancelled", + error: null, + usageTotals: {}, + eventPayload: {}, + }); + expect(c.outcome).toBe("finalized"); + }); + + it("finalizeRun fails a still-queued run and emits the terminal event", async () => { + const { db, runId } = await setupFixture(); + const err = { code: "internal", message: "boom" }; + const r = await finalizeRun(db, { + runId, + from: "queued", + to: "failed", + error: err, + usageTotals: {}, + eventPayload: { error: err }, + }); + expect(r.outcome).toBe("finalized"); + const [run] = await db + .select({ status: runs.status, finishedAt: runs.finishedAt }) + .from(runs) + .where(eq(runs.id, runId)); + expect(run?.status).toBe("failed"); + expect(run?.finishedAt).not.toBeNull(); + const events = await db.select().from(runEvents).where(eq(runEvents.runId, runId)); + expect(events.some((e) => e.type === "run.failed")).toBe(true); + }); + it("decideApproval is a compare-and-swap on pending", async () => { const { db, runId } = await setupFixture(); const [approval] = await db diff --git a/packages/orchestration/src/engine/engine.ts b/packages/orchestration/src/engine/engine.ts index 3a20399..7dd20dd 100644 --- a/packages/orchestration/src/engine/engine.ts +++ b/packages/orchestration/src/engine/engine.ts @@ -37,7 +37,7 @@ import { type TemplateStep, } from "../template-schema"; import type { EngineDeps, RunOutcome } from "./deps"; -import { appendRunEvent, transitionRun } from "./run-lifecycle"; +import { appendRunEvent, finalizeRun, transitionRun } from "./run-lifecycle"; type RunRow = typeof runs.$inferSelect; type StepRow = typeof runSteps.$inferSelect; @@ -484,9 +484,6 @@ class RunEngine { await this.runSystemStep(step); await this.completeStep(row, ""); } else { - // start each attempt from a clean artifact dir so a prior attempt's - // stale file isn't collected as this attempt's result - await this.deps.workspace.clearArtifacts(this.run.id); const output = await this.runAgentStep(phase, step, row, attempt); await this.completeStep(row, output); } @@ -978,52 +975,48 @@ class RunEngine { error: { code: string; message: string } | null, ): Promise { const snapshot = this.meter?.snapshot() ?? { costUsd: 0, tokens: 0, perPhaseCostUsd: {} }; - - // a cancel that landed after the last interrupt check still wins over a - // success — otherwise the API returns "cancel requested" but the run succeeds + const usageTotals = { + costUsd: snapshot.costUsd, + tokens: snapshot.tokens, + perPhaseCostUsd: snapshot.perPhaseCostUsd, + }; + const from = this.run.status; let finalStatus = status; - let finalError = error; - if (status === "succeeded") { - const [row] = await this.db - .select({ cancelRequested: runs.cancelRequested }) - .from(runs) - .where(eq(runs.id, this.run.id)); - if (row?.cancelRequested) { - finalStatus = "cancelled"; - finalError = { code: "cancelled", message: "run cancelled" }; - } - } - const type = `run.${finalStatus}`; - const eventPayload = this.redactor.redact(finalError ? { error: finalError } : {}); - - // status flip + finishedAt/totals + terminal event commit together — a crash - // can no longer leave a terminal run with no finishedAt/totals/event that - // executeRun would then never repair. Publish to the bus only after commit. - const committed = await this.db.transaction(async (tx) => { - // CAS: if another path (e.g. a concurrent cancel) already finalized, bail - if (!(await transitionRun(tx, this.run.id, this.run.status, finalStatus))) return null; - this.run.status = finalStatus; - await tx - .update(runs) - .set({ - finishedAt: new Date(), - error: finalError ?? null, - usageTotals: { - costUsd: snapshot.costUsd, - tokens: snapshot.tokens, - perPhaseCostUsd: snapshot.perPhaseCostUsd, - }, - }) - .where(eq(runs.id, this.run.id)); - return await appendRunEvent(tx, { runId: this.run.id, type, payload: eventPayload }); + let eventPayload = this.redactor.redact(error ? { error } : {}); + + // finalizeRun commits the status CAS + finishedAt/totals + terminal event in + // one tx. For a success we require cancel_requested=false so a cancel that + // landed after the last interrupt check wins atomically (no read/CAS gap). + let result = await finalizeRun(this.db, { + runId: this.run.id, + from, + to: status, + requireNotCancelled: status === "succeeded", + error, + usageTotals, + eventPayload, }); - if (!committed) return; // another path finalized the run + if (result.outcome === "cancelled_instead") { + finalStatus = "cancelled"; + const cancelError = { code: "cancelled", message: "run cancelled" }; + eventPayload = this.redactor.redact({ error: cancelError }); + result = await finalizeRun(this.db, { + runId: this.run.id, + from, + to: "cancelled", + error: cancelError, + usageTotals, + eventPayload, + }); + } + if (result.outcome !== "finalized") return; // another path finalized the run + this.run.status = finalStatus; await this.deps.bus.publish({ runId: this.run.id, - seq: committed.seq, - type, + seq: result.seq, + type: `run.${finalStatus}`, payload: eventPayload, - createdAt: committed.createdAt.toISOString(), + createdAt: result.createdAt.toISOString(), }); try { await this.deps.workspace.cleanup(this.run.id); diff --git a/packages/orchestration/src/engine/fakes.ts b/packages/orchestration/src/engine/fakes.ts index 54d4c99..a95462d 100644 --- a/packages/orchestration/src/engine/fakes.ts +++ b/packages/orchestration/src/engine/fakes.ts @@ -1,4 +1,4 @@ -import { mkdtempSync, rmSync } from "node:fs"; +import { mkdtempSync } from "node:fs"; import { tmpdir } from "node:os"; import path from "node:path"; import type { Logger, ResolvedMcpServer, ResolvedSkill } from "@agrippa/executor-core"; @@ -29,13 +29,6 @@ export class FakeWorkspaceManager implements WorkspaceManager { return this.diffOutput; } - readonly cleared: string[] = []; - async clearArtifacts(runId: string): Promise { - this.cleared.push(runId); - const dir = this.dirs.get(runId); - if (dir) rmSync(path.join(dir, ".agrippa", "artifacts"), { recursive: true, force: true }); - } - async cleanup(runId: string): Promise { this.cleaned.push(runId); this.dirs.delete(runId); diff --git a/packages/orchestration/src/engine/run-lifecycle.ts b/packages/orchestration/src/engine/run-lifecycle.ts index db71166..3e772ab 100644 --- a/packages/orchestration/src/engine/run-lifecycle.ts +++ b/packages/orchestration/src/engine/run-lifecycle.ts @@ -1,5 +1,5 @@ import { canTransitionRun, type RunStatus } from "@agrippa/core"; -import { approvals, type DbOrTx, runEvents, runs } from "@agrippa/db"; +import { approvals, type Db, type DbOrTx, runEvents, runs } from "@agrippa/db"; import { and, eq, sql } from "drizzle-orm"; /** @@ -98,6 +98,57 @@ export async function findStrandedApprovalRuns(db: DbOrTx): Promise { return rows.map((r) => r.id); } +export type FinalizeRunInput = { + runId: string; + from: RunStatus; + to: RunStatus; + /** For a success: the CAS also requires cancel_requested=false, so a cancel that + * landed after the last interrupt check wins atomically instead of racing. */ + requireNotCancelled?: boolean; + error: { code: string; message: string } | null; + usageTotals: Record; + /** Terminal-event payload (caller redacts). */ + eventPayload: Record; +}; + +export type FinalizeResult = + | { outcome: "finalized"; seq: number; createdAt: Date } + | { outcome: "cancelled_instead" } // requireNotCancelled lost to a pending cancel + | { outcome: "lost" }; // another path already finalized the run + +/** + * The single terminal-transition implementation used by both the engine and the + * worker's retry-exhaustion path. In one transaction it CAS-updates the status + * (optionally requiring no pending cancel), writes finishedAt/error/usageTotals, + * and appends the terminal event — so a run can never be left half-finalized and + * both callers always emit the terminal event. + */ +export async function finalizeRun(db: Db, input: FinalizeRunInput): Promise { + const { runId, from, to, requireNotCancelled = false, error, usageTotals, eventPayload } = input; + if (!canTransitionRun(from, to)) throw new Error(`illegal run transition ${from} → ${to}`); + return db.transaction(async (tx): Promise => { + const conds = [eq(runs.id, runId), eq(runs.status, from)]; + if (requireNotCancelled) conds.push(eq(runs.cancelRequested, false)); + const updated = await tx + .update(runs) + .set({ status: to, finishedAt: new Date(), error: error ?? null, usageTotals }) + .where(and(...conds)) + .returning({ id: runs.id }); + if (updated.length === 0) { + const [row] = await tx + .select({ status: runs.status, cancelRequested: runs.cancelRequested }) + .from(runs) + .where(eq(runs.id, runId)); + if (requireNotCancelled && row?.status === from && row.cancelRequested) { + return { outcome: "cancelled_instead" }; + } + return { outcome: "lost" }; + } + const evt = await appendRunEvent(tx, { runId, type: `run.${to}`, payload: eventPayload }); + return { outcome: "finalized", seq: evt.seq, createdAt: evt.createdAt }; + }); +} + /** * Decide a pending approval atomically. The `status = 'pending'` predicate makes * this a compare-and-swap: a user decision and the expiry worker can't overwrite From 780cd17e83d78baf38fdf9782ed82c2e2ecb90c1 Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 19:32:38 +0800 Subject: [PATCH 16/18] fix: symmetric resource resolution, contiguous SSE cursor, split compose volumes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Round 4 (Codex re-review) — commit 2 of 2: resources, streaming, and wiring. Resource-resolution (R4-C): - `requires.skills` only checked authorization, but skill resolution *threw* when no active version existed — so a step failed/retried instead of skipping. Give `skills` the same `{ resolved, missing }` shape as `mcpServers`: an unavailable required skill fails the step, an unavailable optional one is skipped. - scripts/backfill-manifest.ts reconstructed the manifest from the full template (over-granting) and couldn't tell a legacy row from a valid empty one. It now reconstructs from project grants via authorizeResources (idempotent: a valid empty run recomputes empty), skipping runs whose grant was since revoked. SSE (R4-D): - Direct bus delivery advanced a high-water cursor, so a dropped event followed by a later one moved the cursor past the gap and skipped it forever (even on Last-Event-ID reconnect). The bus is now only a wake-up: every event is delivered by an ordered Postgres replay, so the cursor advances contiguously and a dropped message is recovered by the next replay. Compose (R4-E): - Mounting one /work volume on the root API risked initializing it root-owned so the bun worker couldn't write. Split into a worker-only `workspaces` volume and a shared `artifacts` volume; the API runs as bun and mounts only `artifacts`, so ownership is consistent whichever service creates the volume. Docs (ADR-0009, design/04, CHANGELOG) updated. --- CHANGELOG.md | 10 ++-- apps/api/src/routes/execution.ts | 35 ++++--------- apps/worker/src/deps/resources.ts | 20 +++++-- .../0009-security-correctness-deep-modules.md | 3 +- docs/design/04-execution-runtime.md | 2 +- infra/Dockerfile.api | 6 +++ infra/docker-compose.yml | 12 +++-- packages/orchestration/src/engine/deps.ts | 7 ++- .../src/engine/engine.integration.test.ts | 19 ++++++- packages/orchestration/src/engine/engine.ts | 33 ++++++++---- packages/orchestration/src/engine/fakes.ts | 26 +++++++--- scripts/backfill-manifest.ts | 52 ++++++++++++------- 12 files changed, 145 insertions(+), 80 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 2f1916c..2efc57e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,12 +16,12 @@ All notable changes to Agrippa are documented here. The format follows ### Fixed - **Crash recovery for no-retry steps** — a worker that died mid-step no longer silently skips (or spuriously fails) a step without template retries; the crashed attempt no longer consumes the retry budget, and the executor session is carried onto the recovery attempt so resume works. -- **Atomic run lifecycle** — run status transitions are compare-and-swap on the expected status (a late finalize can't overwrite a cancellation), event sequence numbers come from an atomic per-run counter (`runs.next_event_seq`) so they never collide and allocate correctly inside a transaction, and approval decisions are CAS on `pending` with the sweeper re-enqueuing any run left paused by a lost resume enqueue. Finalization commits the status, totals, and terminal event in one transaction (no half-finalized runs), re-checks a late cancel so it wins over a success, and the retry-exhaustion path routes through the CAS. +- **Atomic run lifecycle** — one `finalizeRun` owns every terminal transition: a single transaction does the status CAS (requiring `cancel_requested = false` for a success, so a late cancel wins atomically), finishedAt/totals, and the terminal event. Both the engine and the worker's retry-exhaustion path use it, so a run is never half-finalized and a queued run whose setup threw transitions `queued → failed` (now legal) with a terminal event instead of stranding. Event sequence numbers come from an atomic per-run counter (`runs.next_event_seq`), and approval decisions are CAS on `pending` with the sweeper re-enqueuing any run left paused by a lost resume enqueue. - **Quota accounting** — the engine now counts the same monthly window as the submit gate, excludes the run's own spend from the headroom it checks (no double-count on resume), re-reads project usage at each step boundary so concurrent runs can't jointly overspend, and restores per-phase spend on resume so per-phase budgets aren't reset by a crash. -- **Artifact output contract** — patch steps no longer hand-write the diff (the engine generates it from `git diff` as intended), collected artifacts are validated against the step's declared keys/kinds and matched by exact filename, files from earlier steps aren't re-emitted, the artifact dir is cleared before each attempt so a failed attempt's stale file isn't emitted, missing/empty sources don't create zero-byte rows, and binary (`file`-kind) artifacts are stored byte-exact instead of decoded as text. -- **Authorized resources** — `requires.skills` is now validated by the compiler and enforced by the engine (a step requiring an ungranted skill is skipped, not run without it); `scripts/backfill-manifest.ts` backfills the manifest for runs that predate the manifest migration. -- **SSE live gap** — the events stream subscribes and *awaits* the subscription being live before replaying history (so an event in the replay/subscribe window is delivered live), and re-replays Postgres periodically to recover a dropped pub/sub message mid-run rather than only at terminal (ADR-0007). -- **Production Compose** — the `api` service now mounts the `/work` volume (so it can serve large artifact downloads) and receives `AGRIPPA_EXECUTOR` (so `fake` actually selects the fake executor, since the API chooses the executor at submit). +- **Artifact output contract** — patch steps no longer hand-write the diff (the engine generates it from `git diff`); collected artifacts are validated against the step's declared keys/kinds and matched by exact filename; only the current step's own expected files are cleared before an attempt (never a recursive `rm` a `.agrippa -> /work` symlink could redirect at the shared store, and `.agrippa` is stripped at checkout); missing/empty sources don't create zero-byte rows; binary (`file`-kind) artifacts are stored byte-exact; and ingestion size-caps and streams files instead of buffering them whole. +- **Authorized resources** — `requires.skills` is validated by the compiler and enforced by the engine; `skills` resolution returns `{ resolved, missing }` like `mcpServers`, so an unavailable required skill fails the step and an unavailable optional one is skipped (rather than throwing). `scripts/backfill-manifest.ts` reconstructs the manifest for pre-migration runs from **project grants** (not the full template). +- **SSE live gap** — the events stream subscribes and *awaits* the subscription being live before replaying, and treats the bus purely as a **wake-up** that triggers an ordered Postgres replay — so the cursor advances contiguously and a dropped event can't be skipped past (even on `Last-Event-ID` reconnect) (ADR-0007). +- **Production Compose** — split into a worker-only `workspaces` volume and a shared `artifacts` volume; the `api` runs as the non-root `bun` user and mounts only `artifacts` (consistent ownership whichever service initializes it) and receives `AGRIPPA_EXECUTOR` (the API chooses the executor at submit). ## [0.1.0] — 2026-07-17 diff --git a/apps/api/src/routes/execution.ts b/apps/api/src/routes/execution.ts index 1cab692..ce4ea8a 100644 --- a/apps/api/src/routes/execution.ts +++ b/apps/api/src/routes/execution.ts @@ -437,48 +437,31 @@ export const executionRoutes = new Hono() return row ? isTerminalRunStatus(row.status) : true; }; - // Live: bridge the bus when present, else poll the DB. With a bus we must - // subscribe BEFORE replaying Postgres — an event committed and published - // in the window between replay and subscribe would otherwise be delivered - // only at the terminal replay, contradicting ADR-0007's gap-free - // guarantee. Buffered live events dedupe against the replay by seq. + // Live: bridge the bus when present, else poll the DB. The bus is only a + // WAKE-UP — every event is delivered by an ordered `replay()` from + // Postgres, so the cursor advances contiguously. Sending bus events + // directly would advance a high-water cursor past a dropped seq, and that + // gap would then be skipped forever (even on Last-Event-ID reconnect). if (bus) { - const queue: Array<() => Promise> = []; let notify: (() => void) | null = null; - const subscription = bus.subscribe(run.id, (event) => { - queue.push(async () => { - if (event.seq > cursor) { - await sendRow({ ...event, createdAt: event.createdAt }); - } - }); - notify?.(); - }); + const subscription = bus.subscribe(run.id, () => notify?.()); try { // wait until the subscription is actually live, THEN replay history, // so nothing published in between is dropped (ADR-0007) await subscription.ready; await replay(); - // periodically re-replay from Postgres so a dropped pub/sub message is - // recovered mid-run, not only when the run becomes terminal - let sinceReplay = 0; while (!closed) { - while (queue.length > 0) { - const job = queue.shift(); - if (job) await job(); - } if (await isTerminal()) { - await replay(); // drain anything raced between bus and DB + await replay(); // final ordered drain break; } - if (++sinceReplay >= 5) { - sinceReplay = 0; - await replay(); - } + // sleep until woken by the bus (or a 2 s safety tick), then replay await new Promise((resolve) => { notify = resolve; setTimeout(resolve, 2000); }); notify = null; + await replay(); } } finally { subscription.unsubscribe(); diff --git a/apps/worker/src/deps/resources.ts b/apps/worker/src/deps/resources.ts index e0736cc..43ebe81 100644 --- a/apps/worker/src/deps/resources.ts +++ b/apps/worker/src/deps/resources.ts @@ -21,13 +21,22 @@ const TEMPLATES_DIR = export class DbResourceMaterializer implements ResourceMaterializer { constructor(private readonly db: Db) {} - async skills(refs: string[], workspaceDir: string): Promise { + async skills( + refs: string[], + workspaceDir: string, + ): Promise<{ resolved: ResolvedSkill[]; missing: string[] }> { const resolved: ResolvedSkill[] = []; + const missing: string[] = []; for (const ref of refs) { const slug = skillSlugOfRef(ref); const range = ref.includes("@") ? (ref.split("@")[1] as string) : "*"; const [head] = await this.db.select().from(skills).where(eq(skills.slug, slug)); - if (!head) throw new Error(`skill '${slug}' is not registered`); + if (!head) { + // unregistered or no active matching version → unavailable, not an error; + // the engine treats it symmetrically with a missing MCP server + missing.push(ref); + continue; + } const versions = await this.db .select() .from(skillVersions) @@ -35,7 +44,10 @@ export class DbResourceMaterializer implements ResourceMaterializer { const version = versions .filter((v) => v.status === "active" && Bun.semver.satisfies(v.version, range)) .sort((a, b) => Bun.semver.order(b.version, a.version))[0]; - if (!version) throw new Error(`skill '${slug}' has no version satisfying '${range}'`); + if (!version) { + missing.push(ref); + continue; + } const skillName = slug.split("/").pop() as string; const target = path.join(workspaceDir, ".claude", "skills", skillName); @@ -52,7 +64,7 @@ export class DbResourceMaterializer implements ResourceMaterializer { } resolved.push({ slug, version: version.version, localPath: target }); } - return resolved; + return { resolved, missing }; } async mcpServers(refs: string[]): Promise<{ resolved: ResolvedMcpServer[]; missing: string[] }> { diff --git a/docs/adr/0009-security-correctness-deep-modules.md b/docs/adr/0009-security-correctness-deep-modules.md index 1a8c707..2410cd4 100644 --- a/docs/adr/0009-security-correctness-deep-modules.md +++ b/docs/adr/0009-security-correctness-deep-modules.md @@ -26,6 +26,7 @@ Concentrate each concern behind one deep module whose interface is the test surf ## Consequences - The static isolation layer contains file writes **and reads** to the workspace, refuses shell in read-only workspaces, and redacts known secret values from event payloads. It still cannot bound what a shell command reads or writes in a read-write workspace, nor isolate one run from another at the OS level (all runs share one worker UID and `/work`), nor keep the provider key out of the agent subprocess — those remain the OS sandbox / non-root worker / per-run container + token-proxy's job, which is **explicitly deferred** as a follow-up epic. -- The run-lifecycle module allocates event seq from an atomic per-run counter (`runs.next_event_seq`, migration 0003) so it is collision-free and works inside a caller's transaction; `finalize` commits status + finishedAt/totals + terminal event together. A true execution *lease* (so two at-least-once deliveries can't both resume a `running` run) is still future work. +- The run-lifecycle module allocates event seq from an atomic per-run counter (`runs.next_event_seq`, migration 0003) so it is collision-free and works inside a caller's transaction. One `finalizeRun` owns every terminal transition (status CAS — optionally requiring `cancel_requested = false` so a late cancel wins atomically — plus finishedAt/totals plus the terminal event, in one tx); both the engine and the worker's retry-exhaustion path use it, so a run is never half-finalized and `queued → failed` (setup threw before the run was claimed) is handled uniformly. A true execution *lease* (so two at-least-once deliveries can't both resume a `running` run) is still future work. +- The resource-resolution interface is symmetric: `skills` returns `{ resolved, missing }` like `mcpServers`, so an unavailable required skill (no active version) fails the step and an unavailable optional one is skipped — rather than throwing. The artifact-attempt path clears only a step's *own* expected files (never a recursive dir `rm` that a `.agrippa` symlink could redirect at the shared store) and size-caps ingestion. - Adds a `runs.resource_manifest` column (migration 0002); retries and resumes carry the pinned manifest, so authorization can't drift after submit. Existing runs from before the migration must be backfilled (`scripts/backfill-manifest.ts`). - Grants now genuinely gate optional resources: an optional skill/MCP with no project grant is treated as unavailable, so its dependent step (whether via `requires.mcpServers` or `requires.skills`) is skipped rather than silently privileged. diff --git a/docs/design/04-execution-runtime.md b/docs/design/04-execution-runtime.md index 4330736..bf57bcc 100644 --- a/docs/design/04-execution-runtime.md +++ b/docs/design/04-execution-runtime.md @@ -107,4 +107,4 @@ Ordering rule: the engine writes `run_events` **first** — the per-run monotoni 3. flush the buffer, deduplicating by `seq` against the replay, 4. emit each as `id: \nevent: \ndata: `. -Subscribing (and **awaiting** the subscription is live — for Redis, the SUBSCRIBE ack) **before** replaying is what makes reconnection gap-free: an event committed and published in the window between replay and subscribe would otherwise be delivered only at the terminal replay. The bus branch also re-replays Postgres periodically, not only at terminal, so a dropped pub/sub message is recovered mid-run. Redis is optional: with a bus, live events push instantly; without one, the stream falls back to periodic DB replay, which preserves correctness (just with a small latency). Either way, if Redis is briefly down, clients reconnect and replay from Postgres. +The bus is only a **wake-up**: every event is delivered by an ordered `replay()` from Postgres (`seq > cursor ORDER BY seq`), so the cursor advances contiguously and can never jump past a gap. Sending bus events directly would advance a high-water cursor past a dropped seq, and that gap would then be skipped forever — even on a `Last-Event-ID` reconnect. The handler subscribes (and **awaits** the subscription being live — for Redis, the SUBSCRIBE ack) **before** the first replay, so nothing published in the subscribe/replay window is lost. Redis is optional: with a bus a wake-up makes delivery near-instant; without one the stream ticks the same replay on a timer. Either way a dropped pub/sub message (or a brief Redis outage) is recovered by the next replay, since Postgres is the source of truth. diff --git a/infra/Dockerfile.api b/infra/Dockerfile.api index 9c51b61..e8aadd9 100644 --- a/infra/Dockerfile.api +++ b/infra/Dockerfile.api @@ -15,5 +15,11 @@ COPY --from=build /app /app ENV NODE_ENV=production ENV AGRIPPA_WEB_DIST=/app/apps/web/dist ENV AGRIPPA_TEMPLATES_DIR=/app/templates + +# Run as the non-root `bun` user, and own the shared artifact store so a freshly +# created `artifacts` volume is bun-owned whichever service initializes it first +# (the worker writes it, the API reads it). /app stays root-owned. +RUN mkdir -p /work/artifacts && chown -R bun:bun /work +USER bun EXPOSE 3000 CMD ["bun", "apps/api/src/index.ts"] diff --git a/infra/docker-compose.yml b/infra/docker-compose.yml index 1977101..98e3f6c 100644 --- a/infra/docker-compose.yml +++ b/infra/docker-compose.yml @@ -19,8 +19,10 @@ services: # the API chooses the executor at submit, so it must see the same setting AGRIPPA_EXECUTOR: ${AGRIPPA_EXECUTOR:-claude-agent-sdk} volumes: - # the API serves artifact downloads, incl. large ones stored on the volume - - workdata:/work + # the API serves artifact downloads; it only needs the shared artifact + # store (not run workspaces). Both images run as the `bun` user and chown + # this dir, so the volume is bun-owned whichever container initializes it. + - artifacts:/work/artifacts depends_on: postgres: condition: service_healthy @@ -50,7 +52,8 @@ services: AGRIPPA_EXECUTOR: ${AGRIPPA_EXECUTOR:-claude-agent-sdk} WORKER_SLOTS: ${WORKER_SLOTS:-2} volumes: - - workdata:/work + - workspaces:/work/runs # throwaway per-run checkouts (worker only) + - artifacts:/work/artifacts # shared with the API for downloads depends_on: postgres: condition: service_healthy @@ -78,4 +81,5 @@ services: volumes: pgdata: - workdata: + workspaces: # per-run checkouts, worker-only + artifacts: # artifact store, shared worker↔api (both run as bun) diff --git a/packages/orchestration/src/engine/deps.ts b/packages/orchestration/src/engine/deps.ts index 5067dee..150c4e1 100644 --- a/packages/orchestration/src/engine/deps.ts +++ b/packages/orchestration/src/engine/deps.ts @@ -28,8 +28,11 @@ export interface WorkspaceManager { } export interface ResourceMaterializer { - /** Materialize the step's skills into the workspace; returns their disk locations. */ - skills(refs: string[], workspaceDir: string): Promise; + /** Materialize the step's skills into the workspace; missing = unregistered or no active version. */ + skills( + refs: string[], + workspaceDir: string, + ): Promise<{ resolved: ResolvedSkill[]; missing: string[] }>; /** Resolve step MCP refs against the registry + secrets; missing = unregistered/disabled. */ mcpServers(refs: string[]): Promise<{ resolved: ResolvedMcpServer[]; missing: string[] }>; } diff --git a/packages/orchestration/src/engine/engine.integration.test.ts b/packages/orchestration/src/engine/engine.integration.test.ts index 95894f4..8a49a7f 100644 --- a/packages/orchestration/src/engine/engine.integration.test.ts +++ b/packages/orchestration/src/engine/engine.integration.test.ts @@ -70,7 +70,7 @@ type Fixture = { }; }; -type DepsOptions = { mcpServers?: string[] }; +type DepsOptions = { mcpServers?: string[]; skills?: string[] }; type FixtureOptions = { params?: Record; @@ -196,7 +196,10 @@ async function setupFixture(options: FixtureOptions = {}): Promise { executor, bus, workspace, - resources: new FakeResourceMaterializer({ mcpServers: opts.mcpServers ?? [] }), + resources: new FakeResourceMaterializer({ + mcpServers: opts.mcpServers ?? [], + ...(opts.skills !== undefined ? { skills: opts.skills } : {}), + }), artifacts: new InMemoryArtifactStore(), logger: silentLogger, }; @@ -594,6 +597,18 @@ describe.skipIf(!dbUp)("orchestration engine (FakeExecutor compliance suite)", ( expect(deps.executor.requests.some((r) => r.stepId === "open-pr")).toBe(false); }); + it("fails a step whose required skill has no available version", async () => { + const { db, runId, makeDeps } = await setupFixture(); + await executeRun(makeDeps(HAPPY_SCRIPT), runId); + await approve(db, runId); + // no skills resolve — implement-fix's required builtin/git-workflow is missing + expect(await executeRun(makeDeps(HAPPY_SCRIPT, { skills: [] }), runId)).toBe("failed"); + const [run] = await db.select().from(runs).where(eq(runs.id, runId)); + expect((run?.error as { message?: string } | null)?.message).toContain( + "required resources unavailable", + ); + }); + it("streams live events over the bus while executing", async () => { const { runId, makeDeps, bus } = await setupFixture(); const seen: string[] = []; diff --git a/packages/orchestration/src/engine/engine.ts b/packages/orchestration/src/engine/engine.ts index 7dd20dd..55b3322 100644 --- a/packages/orchestration/src/engine/engine.ts +++ b/packages/orchestration/src/engine/engine.ts @@ -457,12 +457,16 @@ class RunEngine { if (step.requires) { const authorizedMcp = this.authorizedMcpRefs(step.requires.mcpServers); const ungrantedMcp = step.requires.mcpServers.filter((ref) => !authorizedMcp.includes(ref)); - const { missing } = await this.deps.resources.mcpServers(authorizedMcp); - // a required skill the run isn't authorized for is also unmet — otherwise - // the step would run without the skill it declared it needs + const { missing: missingMcp } = await this.deps.resources.mcpServers(authorizedMcp); + // a required skill must be BOTH authorized (in the manifest) and available + // (has an active version) — otherwise the step runs without what it needs const authorizedSkills = this.authorizedSkillRefs(step.requires.skills); const ungrantedSkills = step.requires.skills.filter((ref) => !authorizedSkills.includes(ref)); - const unavailable = [...ungrantedMcp, ...missing, ...ungrantedSkills]; + const { missing: missingSkills } = await this.deps.resources.skills( + authorizedSkills, + this.workspaceDir, + ); + const unavailable = [...ungrantedMcp, ...missingMcp, ...ungrantedSkills, ...missingSkills]; if (unavailable.length > 0) { await this.markSkipped( phase, @@ -629,23 +633,32 @@ class RunEngine { // resolve only what the run is authorized for — ungranted optional // resources are dropped here, never resolved from the global registry - const skills = await this.deps.resources.skills( + const { resolved: skills, missing: missingSkills } = await this.deps.resources.skills( this.authorizedSkillRefs(step.skills), this.workspaceDir, ); - const { resolved: mcpServers, missing } = await this.deps.resources.mcpServers( + const { resolved: mcpServers, missing: missingMcp } = await this.deps.resources.mcpServers( this.authorizedMcpRefs(step.mcpServers), ); // register the resolved MCP credentials so they're redacted from any event this.redactor.add(mcpServers.flatMap(mcpSecretValues)); - const optionalRefs = new Set( + // an unavailable *required* resource fails the step; optional ones are dropped + const optionalMcp = new Set( this.template.spec.resources.mcpServers.filter((m) => m.optional).map((m) => m.ref), ); - const hardMissing = missing.filter((ref) => !optionalRefs.has(ref)); + const optionalSkill = new Set( + this.template.spec.resources.skills + .filter((s) => s.optional) + .map((s) => s.ref.split("@")[0] as string), + ); + const hardMissing = [ + ...missingMcp.filter((ref) => !optionalMcp.has(ref)), + ...missingSkills.filter((ref) => !optionalSkill.has(ref.split("@")[0] as string)), + ]; if (hardMissing.length > 0) { - throw new StepFailed(`required MCP servers unavailable: ${hardMissing.join(", ")}`, { + throw new StepFailed(`required resources unavailable: ${hardMissing.join(", ")}`, { code: "tool_error", - message: `required MCP servers unavailable: ${hardMissing.join(", ")}`, + message: `required resources unavailable: ${hardMissing.join(", ")}`, }); } diff --git a/packages/orchestration/src/engine/fakes.ts b/packages/orchestration/src/engine/fakes.ts index a95462d..e077b76 100644 --- a/packages/orchestration/src/engine/fakes.ts +++ b/packages/orchestration/src/engine/fakes.ts @@ -38,12 +38,26 @@ export class FakeWorkspaceManager implements WorkspaceManager { export class FakeResourceMaterializer implements ResourceMaterializer { constructor(private readonly available: { skills?: string[]; mcpServers?: string[] } = {}) {} - async skills(refs: string[], workspaceDir: string): Promise { - return refs.map((ref) => ({ - slug: ref.split("@")[0] as string, - version: "1.0.0", - localPath: path.join(workspaceDir, ".claude/skills", ref.split("@")[0] as string), - })); + async skills( + refs: string[], + workspaceDir: string, + ): Promise<{ resolved: ResolvedSkill[]; missing: string[] }> { + const allowed = this.available.skills; // undefined = all available + const resolved: ResolvedSkill[] = []; + const missing: string[] = []; + for (const ref of refs) { + const slug = ref.split("@")[0] as string; + if (allowed === undefined || allowed.includes(slug) || allowed.includes(ref)) { + resolved.push({ + slug, + version: "1.0.0", + localPath: path.join(workspaceDir, ".claude/skills", slug), + }); + } else { + missing.push(ref); + } + } + return { resolved, missing }; } async mcpServers(refs: string[]): Promise<{ resolved: ResolvedMcpServer[]; missing: string[] }> { diff --git a/scripts/backfill-manifest.ts b/scripts/backfill-manifest.ts index e045fe0..37cf746 100644 --- a/scripts/backfill-manifest.ts +++ b/scripts/backfill-manifest.ts @@ -1,33 +1,43 @@ import { isTerminalRunStatus, type RunStatus } from "@agrippa/core"; -import { createDb, runs, templateVersions } from "@agrippa/db"; -import type { TemplateDoc } from "@agrippa/orchestration"; +import { createDb, mcpServers, runs, skills, templateVersions } from "@agrippa/db"; +import { authorizeResources, SubmitError, type TemplateDoc } from "@agrippa/orchestration"; import { eq } from "drizzle-orm"; /** * Backfill runs.resource_manifest for runs that predate migration 0002. * * Migration 0002 defaults the manifest to `{}`; because the engine resolves - * skills/MCP only from the manifest, a run that was queued or in flight before - * the migration would silently lose its resources. This authorizes each such - * (non-terminal) run's template-declared skills/MCP, matching the pre-manifest - * behavior. Terminal runs never re-execute, so their empty manifest is harmless - * and left alone. Idempotent: runs with a non-empty manifest are skipped. + * skills/MCP only from the manifest, a run queued or in flight before the + * migration would silently lose its resources. This recomputes each such + * (non-terminal) run's manifest **from the project's grants** — exactly what + * `authorizeResources` produces at submit — rather than granting every template + * resource. That makes it correct for a legacy row and idempotent for a valid + * new run that legitimately has an empty manifest (it recomputes empty). * - * Run this once during any upgrade of an instance that has live runs: + * Run once during any upgrade of an instance that has live runs: * bun scripts/backfill-manifest.ts */ const db = createDb(); +const skillRows = await db.select({ id: skills.id, slug: skills.slug }).from(skills); +const mcpRows = await db.select({ id: mcpServers.id, slug: mcpServers.slug }).from(mcpServers); +const registry = { + skillIdBySlug: new Map(skillRows.map((s) => [s.slug, s.id])), + mcpIdBySlug: new Map(mcpRows.map((m) => [m.slug, m.id])), +}; + const rows = await db .select({ id: runs.id, status: runs.status, + projectId: runs.projectId, resourceManifest: runs.resourceManifest, templateVersionId: runs.templateVersionId, }) .from(runs); let backfilled = 0; +let skipped = 0; for (const run of rows) { if (isTerminalRunStatus(run.status as RunStatus)) continue; if (run.resourceManifest.mcpServers.length > 0 || run.resourceManifest.skills.length > 0) @@ -40,17 +50,21 @@ for (const run of rows) { if (!version) continue; const template = version.compiled as unknown as TemplateDoc; - await db - .update(runs) - .set({ - resourceManifest: { - mcpServers: template.spec.resources.mcpServers.map((m) => m.ref), - skills: template.spec.resources.skills.map((s) => s.ref.split("@")[0] as string), - }, - }) - .where(eq(runs.id, run.id)); - backfilled += 1; + try { + const manifest = await authorizeResources(db, run.projectId, template, registry); + await db.update(runs).set({ resourceManifest: manifest }).where(eq(runs.id, run.id)); + backfilled += 1; + } catch (err) { + // a required grant was revoked since submit — leave the run's manifest empty + // and report it rather than fabricating an authorization it no longer has + if (err instanceof SubmitError) { + console.warn(`[backfill-manifest] run ${run.id}: ${err.code} — left empty`); + skipped += 1; + } else { + throw err; + } + } } -console.log(`[backfill-manifest] updated ${backfilled} non-terminal run(s)`); +console.log(`[backfill-manifest] updated ${backfilled} run(s), skipped ${skipped}`); process.exit(0); From 1d444f634aace5dc4feb64b2d90fc0acc976ccc2 Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 19:47:57 +0800 Subject: [PATCH 17/18] fix(security): allow-list the agent subprocess env; harden the manifest backfill MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Round 5 — the two open items from the PR review (CodeRabbit/Greptile); the rest of their findings were already addressed in rounds 3–4. - buildScrubbedEnv was a denylist, so any worker variable that evaded the secret name heuristic reached the agent subprocess — including NODE_OPTIONS (code injection) and future DSNs/credentials. Switch to an explicit allow-list: only the SDK auth variables and a fixed set of system essentials (PATH/HOME/locale/ TLS-trust) pass through; everything else is dropped. - backfill-manifest.ts guards against a hand-set `{}` manifest (optional-chained length checks) and its comment no longer claims migration 0002 produces `{}` (it backfills the full empty shape). Artifact per-attempt staging (committing artifacts on the emitting event rather than promoting them on step success) is deferred and documented in ADR-0009 — it's moot with the current executors, which collect only on step.completed. --- CHANGELOG.md | 2 +- .../0009-security-correctness-deep-modules.md | 3 +- packages/executor-core/src/isolation.test.ts | 13 ++++- packages/executor-core/src/isolation.ts | 58 ++++++++++++------- scripts/backfill-manifest.ts | 11 ++-- 5 files changed, 58 insertions(+), 29 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 2efc57e..d06aac8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,7 +8,7 @@ All notable changes to Agrippa are documented here. The format follows ### Security -- **Executor isolation seam** — one enforceable place (`packages/executor-core/isolation.ts`) decides every tool call and scrubs the subprocess environment. Read-only workspaces now actually deny shell and confine writes to the artifact directory; read-write workspaces confine writes to the workspace with a boundary-safe check (the previous `startsWith` let a sibling `-evil` path through, and `Bash` bypassed the check entirely). **Reads (Read/Grep/Glob) are confined to the workspace too**, so the agent can't read `/proc/self/environ`, another run's directory, or the shared artifact store. The SDK subprocess runs with the platform secrets (`AGRIPPA_SECRET_KEY`, datastore URLs) stripped from its environment (via an explicit allowlist so a namespaced `*_TOKEN`/`*_KEY` can't ride along), the OS `sandbox` enabled where available (bubblewrap installed in the worker image), `strictMcpConfig`, and repo-supplied `.claude`/`.mcp.json` removed after checkout so a checked-out repository can't inject hooks or permission overrides. The worker image runs as a non-root user with `/app` kept root-owned. +- **Executor isolation seam** — one enforceable place (`packages/executor-core/isolation.ts`) decides every tool call and scrubs the subprocess environment. Read-only workspaces now actually deny shell and confine writes to the artifact directory; read-write workspaces confine writes to the workspace with a boundary-safe check (the previous `startsWith` let a sibling `-evil` path through, and `Bash` bypassed the check entirely). **Reads (Read/Grep/Glob) are confined to the workspace too**, so the agent can't read `/proc/self/environ`, another run's directory, or the shared artifact store. The SDK subprocess environment is **allow-listed** — only the SDK auth variables and a fixed set of system essentials pass through, so platform secrets, DSNs, and injection vectors like `NODE_OPTIONS` are all dropped — with the OS `sandbox` enabled where available (bubblewrap installed in the worker image), `strictMcpConfig`, and repo-supplied `.claude`/`.mcp.json` removed after checkout so a checked-out repository can't inject hooks or permission overrides. The worker image runs as a non-root user with `/app` kept root-owned. - **Event-payload secret redaction** — known secret values (the provider key, resolved MCP tokens) are redacted from every event before it is persisted or streamed over SSE, so a secret the agent echoes into output can't leak through the timeline. - **Artifact path containment** — artifact ingestion resolves sources through `realpath` and rejects any that escape the workspace, closing a symlink disclosure (e.g. `ln -s /proc/self/environ`) that could exfiltrate secrets or other runs' files through the download endpoint. - **Cross-tenant resource authorization** — submission rejects a `repoConnectionId` that isn't owned by the project, and the worker loads repo connections scoped to the run's project. Optional skills/MCP servers are now grant-checked: an authorized-resource manifest is pinned onto the run at submit (required grants enforced, optional resources included only when granted) and the worker resolves resources only from it, so a project without a grant can no longer receive the platform's global credential (e.g. the shared GitHub token). diff --git a/docs/adr/0009-security-correctness-deep-modules.md b/docs/adr/0009-security-correctness-deep-modules.md index 2410cd4..5c92196 100644 --- a/docs/adr/0009-security-correctness-deep-modules.md +++ b/docs/adr/0009-security-correctness-deep-modules.md @@ -27,6 +27,7 @@ Concentrate each concern behind one deep module whose interface is the test surf - The static isolation layer contains file writes **and reads** to the workspace, refuses shell in read-only workspaces, and redacts known secret values from event payloads. It still cannot bound what a shell command reads or writes in a read-write workspace, nor isolate one run from another at the OS level (all runs share one worker UID and `/work`), nor keep the provider key out of the agent subprocess — those remain the OS sandbox / non-root worker / per-run container + token-proxy's job, which is **explicitly deferred** as a follow-up epic. - The run-lifecycle module allocates event seq from an atomic per-run counter (`runs.next_event_seq`, migration 0003) so it is collision-free and works inside a caller's transaction. One `finalizeRun` owns every terminal transition (status CAS — optionally requiring `cancel_requested = false` so a late cancel wins atomically — plus finishedAt/totals plus the terminal event, in one tx); both the engine and the worker's retry-exhaustion path use it, so a run is never half-finalized and `queued → failed` (setup threw before the run was claimed) is handled uniformly. A true execution *lease* (so two at-least-once deliveries can't both resume a `running` run) is still future work. -- The resource-resolution interface is symmetric: `skills` returns `{ resolved, missing }` like `mcpServers`, so an unavailable required skill (no active version) fails the step and an unavailable optional one is skipped — rather than throwing. The artifact-attempt path clears only a step's *own* expected files (never a recursive dir `rm` that a `.agrippa` symlink could redirect at the shared store) and size-caps ingestion. +- The resource-resolution interface is symmetric: `skills` returns `{ resolved, missing }` like `mcpServers`, so an unavailable required skill (no active version) fails the step and an unavailable optional one is skipped — rather than throwing. The artifact-attempt path clears only a step's *own* expected files (never a recursive dir `rm` that a `.agrippa` symlink could redirect at the shared store) and size-caps ingestion. Artifacts are still committed on the emitting event rather than staged and promoted per attempt; this is moot with the current executors (the Claude executor collects only on `step.completed`), and full per-attempt staging is deferred until the artifact module is deepened further. +- The agent subprocess environment is allow-listed (SDK auth + a fixed set of system essentials), not denylisted, so `NODE_OPTIONS` and any future non-secret-looking variable cannot reach it. - Adds a `runs.resource_manifest` column (migration 0002); retries and resumes carry the pinned manifest, so authorization can't drift after submit. Existing runs from before the migration must be backfilled (`scripts/backfill-manifest.ts`). - Grants now genuinely gate optional resources: an optional skill/MCP with no project grant is treated as unavailable, so its dependent step (whether via `requires.mcpServers` or `requires.skills`) is skipped rather than silently privileged. diff --git a/packages/executor-core/src/isolation.test.ts b/packages/executor-core/src/isolation.test.ts index 822e5e5..786dde1 100644 --- a/packages/executor-core/src/isolation.test.ts +++ b/packages/executor-core/src/isolation.test.ts @@ -75,10 +75,11 @@ describe("evaluateToolCall — read-only workspace", () => { }); describe("buildScrubbedEnv", () => { - it("drops platform secrets but keeps allow-listed SDK auth and system vars", () => { + it("allow-lists only SDK auth + system vars, dropping everything else", () => { const env = buildScrubbedEnv({ PATH: "/usr/bin", HOME: "/home/bun", + LANG: "en_US.UTF-8", ANTHROPIC_API_KEY: "sk-ant-xxx", ANTHROPIC_BASE_URL: "https://api.anthropic.com", AGRIPPA_SECRET_KEY: "master", @@ -87,14 +88,20 @@ describe("buildScrubbedEnv", () => { REDIS_URL: "redis://x", GITHUB_TOKEN: "ghp_x", SOME_PASSWORD: "p", - // namespaced secrets must NOT ride along just because of their prefix ANTHROPIC_PRIVATE_KEY: "leak", CLAUDE_ADMIN_TOKEN: "leak", + // a code-injection vector that a name-heuristic denylist would have missed + NODE_OPTIONS: "--require /tmp/evil.js", + // an arbitrary non-secret var still must not pass through + SOME_INTERNAL_URL: "http://internal", }); + // kept: system essentials + explicit SDK auth expect(env.PATH).toBe("/usr/bin"); expect(env.HOME).toBe("/home/bun"); + expect(env.LANG).toBe("en_US.UTF-8"); expect(env.ANTHROPIC_API_KEY).toBe("sk-ant-xxx"); expect(env.ANTHROPIC_BASE_URL).toBe("https://api.anthropic.com"); + // dropped: secrets, namespaced secrets, NODE_OPTIONS, and any unlisted var expect(env.AGRIPPA_SECRET_KEY).toBeUndefined(); expect(env.DATABASE_URL).toBeUndefined(); expect(env.BETTER_AUTH_SECRET).toBeUndefined(); @@ -103,6 +110,8 @@ describe("buildScrubbedEnv", () => { expect(env.SOME_PASSWORD).toBeUndefined(); expect(env.ANTHROPIC_PRIVATE_KEY).toBeUndefined(); expect(env.CLAUDE_ADMIN_TOKEN).toBeUndefined(); + expect(env.NODE_OPTIONS).toBeUndefined(); + expect(env.SOME_INTERNAL_URL).toBeUndefined(); }); }); diff --git a/packages/executor-core/src/isolation.ts b/packages/executor-core/src/isolation.ts index 41528f4..e935d66 100644 --- a/packages/executor-core/src/isolation.ts +++ b/packages/executor-core/src/isolation.ts @@ -168,9 +168,37 @@ const SDK_AUTH_ALLOW = new Set([ ]); /** - * Platform secrets that must never reach the agent subprocess: leaking - * `AGRIPPA_SECRET_KEY` decrypts every stored credential, and the datastore URLs - * grant direct access to run/tenant data. + * System variables the CLI/agent legitimately needs (locale, temp dir, TLS trust + * roots). Notably absent: `NODE_OPTIONS`/`BUN_*`, which can inject code into the + * subprocess, and anything not enumerated here. + */ +const SYSTEM_ENV_ALLOW = new Set([ + "PATH", + "HOME", + "LANG", + "LANGUAGE", + "LC_ALL", + "LC_CTYPE", + "TZ", + "TMPDIR", + "TEMP", + "TMP", + "TERM", + "USER", + "LOGNAME", + "HOSTNAME", + "PWD", + "SHELL", + "SSL_CERT_FILE", + "SSL_CERT_DIR", + "NODE_EXTRA_CA_CERTS", + "CURL_CA_BUNDLE", +]); + +/** + * Platform secrets whose VALUES are redacted from event payloads (below). The + * env allow-list already keeps these out of the subprocess; this set feeds the + * redactor so their values can't be echoed back through SSE either. */ const SECRET_ENV_KEYS = new Set([ "AGRIPPA_SECRET_KEY", @@ -180,18 +208,13 @@ const SECRET_ENV_KEYS = new Set([ "REDIS_URL", ]); -/** Heuristic secret-name match (applied to every non-allowlisted variable). */ -function looksSecret(key: string): boolean { - return /(SECRET|PASSWORD|PRIVATE_KEY|CREDENTIAL|_TOKEN$|_KEY$)/i.test(key); -} - /** - * Build the subprocess environment for the agent, dropping platform secrets. - * The SDK's `env` option REPLACES the child environment wholesale, so we start - * from the worker env and remove what the agent must not see, rather than - * allow-listing (which would starve the CLI of PATH/HOME/locale it needs). The - * SDK auth variables are kept via an explicit allowlist so the secret heuristic - * doesn't drop them. + * Build the subprocess environment for the agent. The SDK's `env` option + * REPLACES the child environment wholesale, so we **allow-list**: only the SDK + * auth variables and a fixed set of system essentials pass through, and + * everything else — platform secrets, DSNs, `NODE_OPTIONS`, and any future + * variable — is dropped. (A denylist would silently forward the next injection + * vector or credential that doesn't match a name heuristic.) */ export function buildScrubbedEnv( source: Record = process.env, @@ -199,12 +222,7 @@ export function buildScrubbedEnv( const out: Record = {}; for (const [key, value] of Object.entries(source)) { if (value === undefined) continue; - if (SDK_AUTH_ALLOW.has(key)) { - out[key] = value; - continue; - } - if (SECRET_ENV_KEYS.has(key) || looksSecret(key)) continue; - out[key] = value; + if (SDK_AUTH_ALLOW.has(key) || SYSTEM_ENV_ALLOW.has(key)) out[key] = value; } return out; } diff --git a/scripts/backfill-manifest.ts b/scripts/backfill-manifest.ts index 37cf746..3f43ee9 100644 --- a/scripts/backfill-manifest.ts +++ b/scripts/backfill-manifest.ts @@ -6,9 +6,9 @@ import { eq } from "drizzle-orm"; /** * Backfill runs.resource_manifest for runs that predate migration 0002. * - * Migration 0002 defaults the manifest to `{}`; because the engine resolves - * skills/MCP only from the manifest, a run queued or in flight before the - * migration would silently lose its resources. This recomputes each such + * Migration 0002 backfilled the manifest to an empty `{mcpServers:[],skills:[]}`; + * because the engine resolves skills/MCP only from the manifest, a run queued or + * in flight before the migration would silently lose its resources. This recomputes each such * (non-terminal) run's manifest **from the project's grants** — exactly what * `authorizeResources` produces at submit — rather than granting every template * resource. That makes it correct for a legacy row and idempotent for a valid @@ -40,8 +40,9 @@ let backfilled = 0; let skipped = 0; for (const run of rows) { if (isTerminalRunStatus(run.status as RunStatus)) continue; - if (run.resourceManifest.mcpServers.length > 0 || run.resourceManifest.skills.length > 0) - continue; + // defensive against a hand-set `{}` row: an already-populated manifest is left alone + const manifest = run.resourceManifest ?? { mcpServers: [], skills: [] }; + if ((manifest.mcpServers?.length ?? 0) > 0 || (manifest.skills?.length ?? 0) > 0) continue; const [version] = await db .select({ compiled: templateVersions.compiled }) From da8f08b3124f3ba14eeb2a6ed33d2eed133556ee Mon Sep 17 00:00:00 2001 From: John Hu Date: Fri, 17 Jul 2026 20:00:10 +0800 Subject: [PATCH 18/18] fix: guard the artifact-size env and mount the API artifact store read-only MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two quick-wins from the latest PR review (CodeRabbit); the review's other threads were stale re-anchors of issues already fixed in rounds 3–5. - AGRIPPA_MAX_ARTIFACT_BYTES=invalid parsed to NaN, and `size > NaN` is always false, silently disabling the artifact size cap. Parse once and fall back to the default on any non-finite / non-positive value. - The API mounts the shared artifact volume but only reads it, so mount it read-only — a compromised API process can't modify or delete worker artifacts. --- apps/worker/src/deps/artifacts.test.ts | 22 ++++++++++++++++++++++ apps/worker/src/deps/artifacts.ts | 10 ++++++++-- infra/docker-compose.yml | 8 +++++--- 3 files changed, 35 insertions(+), 5 deletions(-) diff --git a/apps/worker/src/deps/artifacts.test.ts b/apps/worker/src/deps/artifacts.test.ts index fab4257..0c3dc38 100644 --- a/apps/worker/src/deps/artifacts.test.ts +++ b/apps/worker/src/deps/artifacts.test.ts @@ -119,4 +119,26 @@ describe("DiskArtifactStore path containment", () => { else process.env.AGRIPPA_MAX_ARTIFACT_BYTES = prev; } }); + + it("falls back to the default cap when the size env is not a valid number", async () => { + const ws = freshWorkspace(); + await mkdir(path.join(ws, ".agrippa/artifacts"), { recursive: true }); + writeFileSync(path.join(ws, ".agrippa/artifacts/small.md"), "hello"); + const prev = process.env.AGRIPPA_MAX_ARTIFACT_BYTES; + process.env.AGRIPPA_MAX_ARTIFACT_BYTES = "invalid"; // NaN must not disable the cap + try { + // a normal small file still stores (default cap applies, not NaN) + const stored = await store.store( + "run-1", + "small", + "markdown", + { path: ".agrippa/artifacts/small.md" }, + ws, + ); + expect(stored.inline).toBe("hello"); + } finally { + if (prev === undefined) delete process.env.AGRIPPA_MAX_ARTIFACT_BYTES; + else process.env.AGRIPPA_MAX_ARTIFACT_BYTES = prev; + } + }); }); diff --git a/apps/worker/src/deps/artifacts.ts b/apps/worker/src/deps/artifacts.ts index 0c15db8..d841d7b 100644 --- a/apps/worker/src/deps/artifacts.ts +++ b/apps/worker/src/deps/artifacts.ts @@ -8,8 +8,14 @@ import type { ArtifactStore, StoredArtifact } from "@agrippa/orchestration"; const STORAGE_ROOT = process.env.ARTIFACT_STORAGE_ROOT ?? path.join(tmpdir(), "agrippa-artifacts"); const INLINE_LIMIT = 64 * 1024; /** Hard cap on a single artifact so an agent can't OOM the worker (env-tunable). */ -const maxArtifactSize = (): number => - Number(process.env.AGRIPPA_MAX_ARTIFACT_BYTES ?? 25 * 1024 * 1024); +const DEFAULT_MAX_ARTIFACT_SIZE = 25 * 1024 * 1024; +const maxArtifactSize = (): number => { + const raw = process.env.AGRIPPA_MAX_ARTIFACT_BYTES; + if (raw === undefined) return DEFAULT_MAX_ARTIFACT_SIZE; + const n = Number(raw); + // a bad value (e.g. "invalid" → NaN, or 0/negative) must not silently disable the cap + return Number.isFinite(n) && n > 0 ? n : DEFAULT_MAX_ARTIFACT_SIZE; +}; class ArtifactTooLargeError extends Error { constructor(key: string, size: number) { diff --git a/infra/docker-compose.yml b/infra/docker-compose.yml index 98e3f6c..39f999f 100644 --- a/infra/docker-compose.yml +++ b/infra/docker-compose.yml @@ -20,9 +20,11 @@ services: AGRIPPA_EXECUTOR: ${AGRIPPA_EXECUTOR:-claude-agent-sdk} volumes: # the API serves artifact downloads; it only needs the shared artifact - # store (not run workspaces). Both images run as the `bun` user and chown - # this dir, so the volume is bun-owned whichever container initializes it. - - artifacts:/work/artifacts + # store (not run workspaces), and only reads it — mount read-only so a + # compromised API can't modify or delete worker-produced artifacts. Both + # images run as `bun` and chown this dir, so the volume is bun-owned + # whichever container initializes it. + - artifacts:/work/artifacts:ro depends_on: postgres: condition: service_healthy