diff --git a/.claude/calibration.json b/.claude/calibration.json index 9179b2851..03ba1fc90 100644 --- a/.claude/calibration.json +++ b/.claude/calibration.json @@ -1,236 +1,254 @@ { - "calibratedAt": "2026-09-13", + "calibratedAt": "2026-09-14", "workerEngine": "codex", "workerCommand": "codex", - "workerModel": "gpt-5.6-sol", - "workerArgs": [ - "exec", - "-c", - "windows.sandbox=\"unelevated\"", - "--dangerously-bypass-approvals-and-sandbox", - "-c", - "model_reasoning_effort=\"high\"", - "--model", - "gpt-5.6-sol" - ], - "workerModelSource": "resolveWorkerInvocation(config.worker, ..., \"default\") in tools/lib/orchestrator-config.mjs, the exact vector launch-worker.mjs launches: engine args, then models.default args, then the model", + "workerTiers": { + "default": { + "model": "gpt-5.6-sol", + "args": [ + "exec", + "-c", + "windows.sandbox=\"unelevated\"", + "--dangerously-bypass-approvals-and-sandbox", + "-c", + "model_reasoning_effort=\"high\"", + "--model", + "gpt-5.6-sol" + ] + }, + "mechanical": { + "model": "gpt-5.6-sol", + "args": [ + "exec", + "-c", + "windows.sandbox=\"unelevated\"", + "--dangerously-bypass-approvals-and-sandbox", + "-c", + "model_reasoning_effort=\"medium\"", + "--model", + "gpt-5.6-sol" + ] + } + }, + "workerTierVerdict": "default handles product, design, architecture, and ambiguous orders requiring judgment; mechanical handles merge-forward work, known conflict lists, and reviewer-directed test experiments", + "workerModelSource": "resolveWorkerInvocation(config.worker, ..., tier) in tools/lib/orchestrator-config.mjs for every configured tier, with each exact launch vector: engine args, then selected profile args, then the model", "entries": { - ".agents/skills/merge-prs/SKILL.md": { - "model": null, - "effort": null, - "digest": "fa4a10a59a391992", - "calibratedAt": "2026-09-09", - "verdict": "current: a pointer with no behaviour, so it declares no model and no effort and inherits whatever the Codex host runs. Its digest is the whole verdict: the frontmatter name and description decide whether Codex finds this skill at all, and the body names the one canonical definition both hosts read." - }, - ".agents/skills/orchestrate/SKILL.md": { - "model": null, - "effort": null, - "digest": "d55e8bc268758fcb", - "calibratedAt": "2026-09-09", - "verdict": "current: a pointer with no behaviour, so it declares no model and no effort and inherits whatever the Codex host runs. Its digest is the whole verdict, and it matters most here: orchestrate is the entry point every other piece of work passes through, so a pointer that stops resolving takes the whole queue with it." - }, - ".agents/skills/ticket/SKILL.md": { - "model": null, - "effort": null, - "digest": "e1f13a268ec17004", - "calibratedAt": "2026-09-09", - "verdict": "current: a pointer with no behaviour, so it declares no model and no effort and inherits whatever the Codex host runs. Its digest is the whole verdict, because a ticket is the prompt (D2) and a host that reads a forked definition writes a forked ticket." - }, ".claude/agents/Explore.md": { "model": "haiku", "effort": null, "digest": "74e69f2e9ecb53a7", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "current: locates code and returns excerpts, never reviews it, so haiku is the right floor and effort is undeclared because breadth is passed in the prompt." }, ".claude/agents/audit-readonly.md": { "model": "haiku", "effort": null, "digest": "ccf83adcaf2a327b", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "current: a read-only fan-out finder over Read/Grep/Glob, so haiku is the right floor and no effort is declared because the work is enumeration rather than judgement." }, ".claude/agents/completeness-critic.md": { "model": "sonnet", "effort": "high", "digest": "18fa9e64efcdb3a1", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "current: its whole job is to falsify a completion claim across a surface inventory, which is judgement, so sonnet at high effort is right and must not fall to haiku." }, ".claude/agents/design-reviewer.md": { "model": "sonnet", "effort": "medium", "digest": "2948735644b388c8", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "current: judges a diff against DESIGN.md with the spec in front of it, so sonnet at medium effort is right; the rules are written down, which is what keeps it off high." }, ".claude/agents/design-specialist.md": { "model": "inherit", "effort": "high", "digest": "0238593f7564142c", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "current: shapes the UI half of a ticket and must refuse to improvise where the design system cannot meet a need, so it inherits the session model at high effort deliberately." }, ".claude/agents/product-manager.md": { "model": "inherit", "effort": "high", "digest": "06e63d937fc86cb1", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "current: runs the eight-category edge-case pass and decides how many tickets a request becomes, so it inherits the session model at high effort." }, ".claude/agents/web-researcher.md": { "model": "sonnet", "effort": "medium", "digest": "b0eba80af5cdd41f", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "current: verifies load-bearing facts against live pages for one narrow slice, so sonnet at medium effort is right; it has no Agent tool, which is the structural cap that matters more than the model." }, ".claude/skills/android-generate/SKILL.md": { "model": null, "effort": "low", "digest": "4aab9a05b8f27d13", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "current: a mechanical gradle build and emulator install, so low effort is right." }, ".claude/skills/android-release/SKILL.md": { "model": null, "effort": "low", "digest": "511b9e409e86a2d9", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "current: dispatches one workflow with computed version numbers, so low effort is right." }, ".claude/skills/audit-code-quality/SKILL.md": { "model": null, "effort": null, "digest": "2b278674dc8f8764", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "undeclared, inherits the session: a judgement-level debt audit that opens tickets, which argues for an explicit high. Left undeclared in this first pass because declaring it changes behaviour and cost, and eleven skills are in the same position; that is the follow-up this pass names rather than a change smuggled into the mechanism." }, ".claude/skills/audit-performance/SKILL.md": { "model": null, "effort": null, "digest": "97dae7a146ee0fc6", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "undeclared, inherits the session: same shape as audit-code-quality and the same follow-up." }, ".claude/skills/audit-security/SKILL.md": { "model": null, "effort": null, "digest": "e92ff1f4ff0b8b21", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "undeclared, inherits the session: same shape as audit-code-quality and the same follow-up. This is the one where an inherited low effort would cost the most, because a missed authz hole is not visible in the output." }, ".claude/skills/audit-tests/SKILL.md": { "model": null, "effort": null, "digest": "291bf85e61e31045", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "undeclared, inherits the session: same shape as audit-code-quality and the same follow-up." }, ".claude/skills/deep-research/SKILL.md": { "model": null, "effort": null, "digest": "a7bbf3c505b096e8", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "undeclared, inherits the session: it fans out web-researcher subagents that carry their own sonnet/medium tuning, so the orchestrating half inheriting the session is defensible today." }, ".claude/skills/dev-server/SKILL.md": { "model": null, "effort": "low", "digest": "b95013bea459e4fc", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "current: brings up Docker, the API and the web server in dependency order with readiness gates, which is mechanical, so low effort is right." }, ".claude/skills/drift-review/SKILL.md": { "model": null, "effort": null, "digest": "1fae48b525fe2fe2", - "calibratedAt": "2026-09-12", + "calibratedAt": "2026-09-14", "verdict": "undeclared, inherits the session: it judges repeated evidence against the current workflow files, but every result remains a staged candidate for human review." }, ".claude/skills/handoff/SKILL.md": { "model": null, "effort": "high", "digest": "3f7239924dc41147", - "calibratedAt": "2026-09-13", + "calibratedAt": "2026-09-14", "verdict": "current: high effort, and it earns it. It decides what survives into a spec that outlives every session, and under-thinking it is how a rule Thomas set in week one disappears by week four." }, ".claude/skills/investigate/SKILL.md": { "model": null, "effort": null, "digest": "9e633595abd89253", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "undeclared, inherits the session: root-causing a production incident across Sentry, Render, Postgres and the LSP is judgement, so this is a follow-up candidate." }, ".claude/skills/lesson/SKILL.md": { "model": null, "effort": null, "digest": "f3af59784bce4557", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "undeclared, inherits the session: it graduates a correction into a rule, which is judgement, but it always runs with Thomas present, so an inherited effort is checked by a human in the moment." }, ".claude/skills/merge-prs/SKILL.md": { "model": null, "effort": null, "digest": "30a5ce2efa3fb3ea", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "undeclared, inherits the session: the dangerous half of this skill is mechanical (an exact-head preflight, an ordered admin squash), and its safety comes from the preflight rather than from reasoning depth." }, ".claude/skills/orchestrate/SKILL.md": { "model": null, "effort": "high", "digest": "4df860787a7b5091", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "current: high effort, and it earns it: it plans the queue, verifies delivery from artifacts and clears the review, and it is the entry point every other piece of work passes through." }, ".claude/skills/prod-readiness/SKILL.md": { "model": null, "effort": null, "digest": "24aa498f98faf0c3", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "undeclared, inherits the session: it consolidates four child audits into one honest launch verdict, which is judgement, so this is a follow-up candidate." }, ".claude/skills/progress/SKILL.md": { "model": null, "effort": "medium", "digest": "976dcf419b3253d8", - "calibratedAt": "2026-09-13", + "calibratedAt": "2026-09-14", "verdict": "current: medium effort, because it reads live git and ticket state and must judge whether a part-built screen is honestly described, which low effort gets wrong by rounding up." }, ".claude/skills/questions/SKILL.md": { "model": null, "effort": "high", "digest": "430ce323de282429", - "calibratedAt": "2026-09-13", + "calibratedAt": "2026-09-14", "verdict": "current: high effort, because the filter decides what NOT to ask, and a wrong call either wastes his attention or ships a guess as a decision." }, ".claude/skills/second-opinion/SKILL.md": { "model": null, "effort": null, "digest": "8977e10e62d65f75", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "current with nothing to declare: the reasoning happens in the other model, by construction. Declaring an effort here would tune the wrong side of the call." }, ".claude/skills/sleep/SKILL.md": { "model": null, "effort": null, "digest": "3acd8155b2528682", - "calibratedAt": "2026-09-11", + "calibratedAt": "2026-09-14", "verdict": "undeclared, inherits the session: it takes every decision alone overnight, which is the strongest argument for an explicit high in the follow-up, and the weakest place to guess it in this pass." }, ".claude/skills/ticket/SKILL.md": { "model": null, "effort": "high", "digest": "ce7376bb28eca9c8", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "current: high effort, and it earns it: a ticket is the prompt (D2), so a shallow ticket is a shallow implementation, and the cost lands on whoever executes it." }, ".claude/skills/validate/SKILL.md": { "model": null, "effort": "low", "digest": "e1ef5a36050d37f1", - "calibratedAt": "2026-09-09", + "calibratedAt": "2026-09-14", "verdict": "current: runs lint, type-check and tests across both repos, so low effort is right." + }, + ".agents/skills/merge-prs/SKILL.md": { + "model": null, + "effort": null, + "digest": "fa4a10a59a391992", + "calibratedAt": "2026-09-14", + "verdict": "current: a pointer with no behaviour, so it declares no model and no effort and inherits whatever the Codex host runs. Its digest is the whole verdict: the frontmatter name and description decide whether Codex finds this skill at all, and the body names the one canonical definition both hosts read." + }, + ".agents/skills/orchestrate/SKILL.md": { + "model": null, + "effort": null, + "digest": "d55e8bc268758fcb", + "calibratedAt": "2026-09-14", + "verdict": "current: a pointer with no behaviour, so it declares no model and no effort and inherits whatever the Codex host runs. Its digest is the whole verdict, and it matters most here: orchestrate is the entry point every other piece of work passes through, so a pointer that stops resolving takes the whole queue with it." + }, + ".agents/skills/ticket/SKILL.md": { + "model": null, + "effort": null, + "digest": "e1f13a268ec17004", + "calibratedAt": "2026-09-14", + "verdict": "current: a pointer with no behaviour, so it declares no model and no effort and inherits whatever the Codex host runs. Its digest is the whole verdict, because a ticket is the prompt (D2) and a host that reads a forked definition writes a forked ticket." } } } diff --git a/.claude/orchestrator.json b/.claude/orchestrator.json index c005c3e3f..e1a3aca75 100644 --- a/.claude/orchestrator.json +++ b/.claude/orchestrator.json @@ -13,6 +13,10 @@ "default": { "model": "gpt-5.6-sol", "args": ["-c", "model_reasoning_effort=\"high\""] + }, + "mechanical": { + "model": "gpt-5.6-sol", + "args": ["-c", "model_reasoning_effort=\"medium\""] } } } @@ -31,6 +35,7 @@ "parallelTickets": 3, "cloudParallelTasks": 8, "reviewFixAttempts": 3, + "workerLaunchesPerBranch": 2, "workerLogMegabytes": 512 }, "cloud": { diff --git a/tools/README.md b/tools/README.md index 4d45a090b..3adc1ea90 100644 --- a/tools/README.md +++ b/tools/README.md @@ -25,7 +25,7 @@ name or git status. Read `CONVENTIONS.md` before adding one. | `complete-ticket.mjs` | Owns the explicit post-merge transition. `--preflight` proves the open ticket and configured project item without writing. Completion sets Status Done and closes the issue as completed. `--repair-status` is the inverse, for a ticket GitHub already closed from a merge commit: it reconciles the board Status with the close reason (completed to Done, not planned to Canceled, duplicate to Duplicate), writes nothing else, and is a no-op when the column already agrees. `--cancel --reason-file` closes an open ticket whose work is GONE rather than done: it posts the required reason, sets Status Canceled and closes as not planned, so deleted work is never recorded as shipped. | `node tools/complete-ticket.mjs --issue [--preflight \| --repair-status \| --cancel --reason-file ]` | | `plan-queue.mjs` | Resolves a scope (explicit tickets, a GitHub milestone, or the configured board) into ONE ordered execution plan: admission and deferrals, DAG-safe waves, transitive fan-out (`unlocks`), and stackable dependency chains. A ticket whose same-repo blockers do not form one chain has no branch that can carry them, so it defers as `UNSTACKABLE_BLOCKERS_IN_QUEUE` and the rest of the board still plans; cross-repo blockers order but never stack. A still-open blocker outside the queue defers the ticket. It plans only, and writes nothing anywhere. | `node tools/plan-queue.mjs (--tickets ORB-1,#123 \| --project \| --board)` (`--format markdown`) | | `compose-prompt.mjs` | Builds one worker prompt with the ticket and comments, a brief, and one finishing contract. Both modes require an unchanged-test observation and a strengthened-test failure before fixing a missed regression, with evidence delivered to the PR body. Local workers push and open a PR; `--cloud` selects commit-only container delivery for the configured Cloud repository and carries evidence in handoff `testResults`. The orchestrator owns CI waiting in both modes. Refuses output inside a declared repo. | `node tools/compose-prompt.mjs --issue --repo --out [--cloud]` | -| `launch-worker.mjs` | Spawns one headless worker and supervises it in the foreground. Both clocks kill the complete process tree, and this process is itself the run's wake source. `--measurement` swaps the no-progress clock for the measurement one and leaves the hard ceiling alone. `--hard-ceiling-minutes ` replaces the configured ceiling for this one launch, for a ticket that legitimately outruns the fleet-wide default. `--dry-run` prints the resolved plan and spawns nothing. GitHub credentials are owner-selected per child environment; global account state is never switched. It launches a worker; it never merges, reviews, or moves a ticket. | `node tools/launch-worker.mjs --issue --worktree --prompt ` (`--measurement`, `--hard-ceiling-minutes `, `--dry-run`) | +| `launch-worker.mjs` | Spawns one headless worker and supervises it in the foreground. Both clocks kill the complete process tree, and this process is itself the run's wake source. `--tier ` selects the configured order profile. Each non-dry launch is recorded in the checkout-local Git ledger, and a branch exceeding `caps.workerLaunchesPerBranch` needs `--relaunch-reason `. `--measurement` swaps the no-progress clock for the measurement one and leaves the hard ceiling alone. `--hard-ceiling-minutes ` replaces the configured ceiling for this launch. `--dry-run` prints the resolved plan and spawns nothing. GitHub credentials are owner-selected per child environment; global account state is never switched. It launches a worker; it never merges, reviews, or moves a ticket. | `node tools/launch-worker.mjs --issue --worktree --prompt ` (`--tier `, `--relaunch-reason `, `--measurement`, `--hard-ceiling-minutes `, `--dry-run`) | | `submit-cloud-worker.mjs` | Submits one Codex Cloud implementation from the explicitly pushed branch, after a strict dash check, and persists the task, base SHA, order hashes, deadline, and target worktree in both the scratchpad and shared Git state. Submission returns immediately. An empty-diff failure permits one reserved retry of the unchanged order with the commit instruction first; failed or uncertain retries retain ticket ownership. Its separate `--watch` mode registers a live wake source, records local abandonment at the deadline, and stays responsible until the remote task is terminal before resuming the unattended scheduler. It never applies a result. | `node tools/submit-cloud-worker.mjs --issue --env --branch --order --worktree ` or `node tools/submit-cloud-worker.mjs --watch ` | | `materialize-cloud-result.mjs` | Reads one task through `codex cloud list --json`, quarantines late results, and serially applies a ready diff at the receipt's exact base SHA. Recovery authenticates raw patch bytes. New submissions require a committed Cloud handoff, preserved in the receipt and returned on retries. With a handoff, `pullRequestBody` returns a `## Test evidence` section containing its `testResults` verbatim for PR delivery. Exit 10 resolves invalid metadata as terminally unusable while preserving the staged patch for manual delivery, or blocks a pending decision. It reports Git status and staged stat from disk. An empty result records CLOUD_TASK_EMPTY, retains ticket ownership, and reports title, summary counts, named targets, and the single retry state. | `node tools/materialize-cloud-result.mjs --receipt [--allow-abandoned]` | | `verify-delivery.mjs` | Derives delivery solely from bounded Git and GitHub children: clean/pushed/current PR, exact current-head CI state, and GitHub compare `behind_by=0`. `OUT_OF_DATE` reports base/head SHAs and behind count. File and line counts remain `sizeAdvisory` and never block. Completed CI passes only on the confirmed `SUCCESS`, `NEUTRAL`, or `SKIPPED` allowlist; every other or future conclusion fails closed. Failed CI retains run/check ID, details URL, workflow, name, status and conclusion; newest rerun wins. With no required checks, a nonempty rollup must remain unchanged across 60 seconds of observation in the current invocation; use `--wait-ci` or receive `CI_PENDING` with the registration reason. | `node tools/verify-delivery.mjs --issue --worktree

--branch --repo ` (`--base`, `--wait-ci`, `--command-timeout-seconds`) | diff --git a/tools/__tests__/check-calibration.mjs b/tools/__tests__/check-calibration.mjs index 91818630e..4162e36b7 100644 --- a/tools/__tests__/check-calibration.mjs +++ b/tools/__tests__/check-calibration.mjs @@ -39,7 +39,7 @@ const stageHarness = (label, { stamp, model = "gpt-5.6-sol", engine = "codex", c for (const [relativePath, body] of Object.entries(extraFiles)) write(join(fixture, relativePath), body) write( join(fixture, ".claude", "orchestrator.json"), - `${JSON.stringify({ caps: { parallelTickets: 3 }, worker: engine, workers: { [engine]: { command, args: engineArgs, models: { default: { model, args } } } } }, null, 2)}\n`, + `${JSON.stringify({ caps: { parallelTickets: 3 }, worker: engine, workers: { [engine]: { command, args: engineArgs, models: { default: { model, args }, mechanical: { model, args: ["-c", 'model_reasoning_effort="medium"'] } } } } }, null, 2)}\n`, ) if (stamp !== null) write(join(fixture, ".claude", "calibration.json"), `${JSON.stringify(stamp, null, 2)}\n`) return fixture @@ -53,10 +53,12 @@ const currentStamp = (overrides = {}) => { calibratedAt, workerEngine: "codex", workerCommand: "codex", - workerModel: "gpt-5.6-sol", - // The RESOLVED vector launch-worker.mjs launches: engine args, then profile args, then the model. - workerArgs: ["exec", "-c", 'model_reasoning_effort="high"', "--model", "gpt-5.6-sol"], - workerModelSource: "resolveWorkerInvocation(config.worker, ..., default) in tools/lib/orchestrator-config.mjs", + workerTiers: { + default: { model: "gpt-5.6-sol", args: ["exec", "-c", 'model_reasoning_effort="high"', "--model", "gpt-5.6-sol"] }, + mechanical: { model: "gpt-5.6-sol", args: ["exec", "-c", 'model_reasoning_effort="medium"', "--model", "gpt-5.6-sol"] }, + }, + workerTierVerdict: "default for judgment; mechanical for specified procedures", + workerModelSource: "resolveWorkerInvocation(config.worker, ..., tier) in tools/lib/orchestrator-config.mjs", entries: { ".claude/agents/design-reviewer.md": { model: "sonnet", effort: "medium", digest: digestOf(AGENT), calibratedAt, verdict: "current" }, ".claude/skills/lesson/SKILL.md": { model: null, effort: null, digest: digestOf(UNTUNED_SKILL), calibratedAt, verdict: "undeclared, inherits the session" }, @@ -75,7 +77,7 @@ const withEntries = (mutate) => { export const cases = () => { check(TOOL, "a stamp covering every agent and skill file exits 0", ["--root", stageHarness("clean", { stamp: currentStamp() })], { status: 0, - stdout: /3 calibrated file\(s\) stamped .* against codex gpt-5\.6-sol \["exec","-c","model_reasoning_effort=\\"high\\"","--model","gpt-5\.6-sol"\]/, + stdout: /3 calibrated file\(s\) stamped .* against codex tiers default, mechanical/, }) // The denominator is a glob, so a file added without a verdict is the case that catches the failure @@ -113,7 +115,7 @@ export const cases = () => { TOOL, "editing the worker model without a fresh stamp exits 1", ["--root", stageHarness("model-moved", { stamp: currentStamp(), model: "gpt-6-next" })], - { status: 1, stderr: /the worker model is gpt-6-next and the stamp was taken against gpt-5\.6-sol/ }, + { status: 1, stderr: /the default tier worker model is gpt-6-next and the stamp was taken against gpt-5\.6-sol/ }, ) check( @@ -224,7 +226,7 @@ export const cases = () => { TOOL, "changing only the reasoning effort in the profile args exits 1", ["--root", stageHarness("args-changed", { stamp: currentStamp(), args: ["-c", 'model_reasoning_effort="low"'] })], - { status: 1, stderr: /this is the whole resolved launch vector, engine args included/ }, + { status: 1, stderr: /default tier worker args[\s\S]*whole resolved launch vector, engine args included/ }, ) /** * The half this gate could not see. `resolveWorkerInvocation` prepends `engine.args` to every launch, @@ -239,7 +241,7 @@ export const cases = () => { "--root", stageHarness("engine-args-changed", { stamp: currentStamp(), args: [], engineArgs: ["exec", "-c", 'model_reasoning_effort="low"'] }), ], - { status: 1, stderr: /this is the whole resolved launch vector, engine args included/ }, + { status: 1, stderr: /default tier worker args[\s\S]*whole resolved launch vector, engine args included/ }, ) /** * And the same configuration with the stamp reseeded against it passes, so the case above proves the @@ -251,7 +253,12 @@ export const cases = () => { [ "--root", stageHarness("engine-args-reseeded", { - stamp: currentStamp({ workerArgs: ["exec", "-c", 'model_reasoning_effort="low"', "--model", "gpt-5.6-sol"] }), + stamp: currentStamp({ + workerTiers: { + default: { model: "gpt-5.6-sol", args: ["exec", "-c", 'model_reasoning_effort="low"', "--model", "gpt-5.6-sol"] }, + mechanical: { model: "gpt-5.6-sol", args: ["exec", "-c", 'model_reasoning_effort="low"', "-c", 'model_reasoning_effort="medium"', "--model", "gpt-5.6-sol"] }, + }, + }), args: [], engineArgs: ["exec", "-c", 'model_reasoning_effort="low"'], }), @@ -481,7 +488,7 @@ export const cases = () => { ) write( join(emptyTree, ".claude", "calibration.json"), - `${JSON.stringify({ calibratedAt: today(), workerEngine: "codex", workerCommand: "codex", workerModel: "gpt-5.6-sol", workerArgs: ["--model", "gpt-5.6-sol"], entries: {} }, null, 2)}\n`, + `${JSON.stringify({ calibratedAt: today(), workerEngine: "codex", workerCommand: "codex", workerTiers: { default: { model: "gpt-5.6-sol", args: ["--model", "gpt-5.6-sol"] } }, workerTierVerdict: "default for judgment", entries: {} }, null, 2)}\n`, ) check(TOOL, "a tree with no agent and no skill exits 2 rather than reporting a vacuous green", ["--root", emptyTree], { status: 2, diff --git a/tools/__tests__/launch-worker.mjs b/tools/__tests__/launch-worker.mjs index f3f5cc7f7..499368851 100644 --- a/tools/__tests__/launch-worker.mjs +++ b/tools/__tests__/launch-worker.mjs @@ -1,5 +1,5 @@ import { spawn, spawnSync } from "node:child_process" -import { appendFileSync, cpSync, existsSync, mkdirSync, readFileSync, renameSync, rmSync, watch, writeFileSync } from "node:fs" +import { appendFileSync, cpSync, existsSync, mkdirSync, readFileSync, readdirSync, renameSync, rmSync, watch, writeFileSync } from "node:fs" import { dirname, join } from "node:path" import { processIsRunning, T, check, orcaEnv, realOrchestratorConfig, run, stage, stageRepo, stageWithConfig, TOOLS_DIR } from "./_harness.mjs" @@ -27,7 +27,15 @@ const launchConfig = ({ engine = {}, timeouts = {}, caps = {} } = {}) => { * that only sleeps. `args` is the script PATH and not `-e`, because node refuses the `--model` * the launcher appends after an `--eval` script ("bad option: --model"), measured. */ -const stubEngine = (script) => ({ engine: { args: [script], models: { default: { model: "gate-stub", args: [] } } } }) +const stubEngine = (script) => ({ + engine: { + args: [script], + models: { + default: { model: "gate-stub", args: ["--gate-effort=high"] }, + mechanical: { model: "gate-stub", args: ["--gate-effort=medium"] }, + }, + }, +}) const SLEEPER = stage("launch-worker/sleeping-worker.js", "setTimeout(() => {}, 60000)\n") const IMMEDIATE = stage("launch-worker/immediate-worker.js", "process.exit(0)\n") @@ -40,8 +48,9 @@ const launch = (label, config) => { const repo = stageRepo(`launch-worker-${label}`) if (!repo) return null repo.git(["remote", "set-url", "origin", `https://github.com/test-owner/${label}.git`]) - const staged = stageWithConfig(`launch-worker-${label}`, TOOL, config) - return { ...staged, worktree: repo.path, prompt: stage(`launch-worker/${label}-prompt.md`, "the work order, verbatim\n") } + const configured = { ...config, repos: { ...config.repos, [config.cloud.repositoryKey]: repo.path } } + const staged = stageWithConfig(`launch-worker-${label}`, TOOL, configured) + return { ...staged, worktree: repo.path, git: repo.git, prompt: stage(`launch-worker/${label}-prompt.md`, "the work order, verbatim\n") } } const githubAuthEnv = () => orcaEnv([{ match: "auth token --user test-owner", stdout: "test-github-token" }]) @@ -209,6 +218,9 @@ export const cases = async () => { check(TOOL, "refuses a malformed ticket reference", ["--issue", "ticket-201", "--worktree", fixture.worktree, "--prompt", fixture.prompt], { status: 2, stderr: /--issue must be ORB-N, #N, or N/ }, options) check(TOOL, "accepts a post-migration #N reference", ["--issue", "#9001", "--worktree", fixture.worktree, "--prompt", fixture.prompt, "--dry-run"], { status: 0, stdout: /"issue": "#9001"/ }, options) check(TOOL, "accepts a post-migration plain number and normalizes it", ["--issue", "9001", "--worktree", fixture.worktree, "--prompt", fixture.prompt, "--dry-run"], { status: 0, stdout: /"issue": "#9001"/ }, options) + check(TOOL, "refuses a tier outside the two order profiles", [...argv, "--tier", "expensive", "--dry-run"], { status: 2, stderr: /--tier must be default or mechanical/ }, options) + check(TOOL, "refuses a tier flag with no value", [...argv, "--tier"], { status: 2, stderr: /--tier must be default or mechanical/ }, options) + check(TOOL, "refuses an empty relaunch reason", [...argv, "--relaunch-reason", "", "--dry-run"], { status: 2, stderr: /--relaunch-reason must be non-empty text/ }, options) check(TOOL, "refuses a missing worktree flag", ["--issue", "ORB-201", "--prompt", fixture.prompt], { status: 2, stderr: /--worktree is required/ }, options) check(TOOL, "refuses a missing prompt flag", ["--issue", "ORB-201", "--worktree", fixture.worktree], { status: 2, stderr: /--prompt is required/ }, options) check(TOOL, "refuses a worktree that does not exist", ["--issue", "ORB-201", "--worktree", join(fixture.base, "absent"), "--prompt", fixture.prompt], { status: 2, stderr: /worktree not found/ }, options) @@ -260,15 +272,59 @@ export const cases = async () => { JSON.stringify(plan), ) T( - `${TOOL}: the resolved implementer is gpt-5.6-sol at high reasoning effort (D21)`, + `${TOOL}: the default tier resolves gpt-5.6-sol at high reasoning effort`, plan !== null && plan.model === "gpt-5.6-sol" && plan.args.includes('model_reasoning_effort="high"'), JSON.stringify(plan?.args), ) + const mechanicalDryRun = check(TOOL, "--dry-run resolves the mechanical tier", [...argv, "--tier", "mechanical", "--dry-run"], { status: 0 }, options) + const mechanicalPlan = JSON.parse(mechanicalDryRun.stdout) + T( + `${TOOL}: each tier reports itself and resolves a different argument vector`, + mechanicalPlan.tier === "mechanical" && + mechanicalPlan.args.includes('model_reasoning_effort="medium"') && + JSON.stringify(mechanicalPlan.args) !== JSON.stringify(plan?.args), + JSON.stringify({ default: plan?.args, mechanical: mechanicalPlan.args }), + ) T( `${TOOL}: the worker is handed the prompt PATH and the branch, never the prompt text`, plan !== null && plan.branch === "main" && plan.args.at(-1).includes(fixture.prompt) && !plan.args.at(-1).includes("the work order, verbatim"), JSON.stringify(plan?.args?.at(-1)), ) + + const capped = launch("branch-cap", launchConfig({ ...stubEngine(IMMEDIATE), caps: { workerLaunchesPerBranch: 2 } })) + const cappedArgs = ["--issue", "ORB-201", "--worktree", capped.worktree, "--prompt", capped.prompt] + const firstLaunch = check(TOOL, "the first launch below the branch cap succeeds", cappedArgs, { status: 0 }, { path: capped.path, env: githubAuthEnv() }) + const secondLaunch = check(TOOL, "the second launch at the branch cap succeeds", [...cappedArgs, "--tier", "mechanical"], { status: 0 }, { path: capped.path, env: githubAuthEnv() }) + discardLog(firstLaunch.stdout) + discardLog(secondLaunch.stdout) + const refusedLaunch = check( + TOOL, + "a third launch exits 2 and names both earlier launches", + cappedArgs, + { status: 2, stderr: /worker launch cap 2 reached[\s\S]*Earlier launches:[\s\S]*1\.[\s\S]*tier=default[\s\S]*2\.[\s\S]*tier=mechanical/ }, + { path: capped.path }, + ) + const allowedLaunch = check( + TOOL, + "a third launch with a recorded reason succeeds", + [...cappedArgs, "--tier", "mechanical", "--relaunch-reason", "Pullfrog supplied a second experiment"], + { status: 0 }, + { path: capped.path, env: githubAuthEnv() }, + ) + discardLog(allowedLaunch.stdout) + const ledgerPath = join(capped.worktree, ".git", "orbit-worker-launches") + const launchLedger = readdirSync(ledgerPath) + .map((name) => JSON.parse(readFileSync(join(ledgerPath, name), "utf8"))) + .sort((left, right) => left.timestamp.localeCompare(right.timestamp)) + T( + `${TOOL}: the checkout-local ledger records required fields and the override reason`, + launchLedger.length === 3 && + launchLedger.every((row) => row.repositoryKey === "ui" && row.branch === "main" && /^[0-9a-f]{40}$/.test(row.headSha) && typeof row.timestamp === "string") && + launchLedger[2].relaunchReason === "Pullfrog supplied a second experiment" && + refusedLaunch.stderr.includes(launchLedger[0].headSha) && + capped.git(["status", "--short"]).stdout.trim() === "", + JSON.stringify({ launchLedger, status: capped.git(["status", "--short"]).stdout }), + ) /** * Both clocks are read from config.timeouts, and the only way to prove that is to move them: * a hardcoded 45 and 10 minutes would leave a sleeping worker running until the suite's own diff --git a/tools/__tests__/orchestrator-config.mjs b/tools/__tests__/orchestrator-config.mjs index 8b4a492d7..21e832583 100644 --- a/tools/__tests__/orchestrator-config.mjs +++ b/tools/__tests__/orchestrator-config.mjs @@ -96,10 +96,16 @@ export const cases = async () => { JSON.stringify(invocation), ) T( - `${NAME}: the shipped implementer is gpt-5.6-sol at high reasoning effort (D21)`, + `${NAME}: the shipped default implementer is gpt-5.6-sol at high reasoning effort`, invocation.model === "gpt-5.6-sol" && invocation.args.includes('model_reasoning_effort="high"'), `.claude/orchestrator.json resolved ${invocation.model} with ${JSON.stringify(invocation.args)}`, ) + const mechanicalInvocation = resolveWorkerInvocation(engineName, engine, "mechanical") + T( + `${NAME}: the shipped mechanical implementer keeps the model and lowers reasoning effort`, + mechanicalInvocation.model === invocation.model && mechanicalInvocation.args.includes('model_reasoning_effort="medium"'), + JSON.stringify(mechanicalInvocation), + ) const readAndFail = (label, config) => thrown(() => readOrchestratorConfig(configUrl(label, JSON.stringify(config)))) T( @@ -202,6 +208,13 @@ export const cases = async () => { /caps\.reviewFixAttempts must be a positive number/.test(readAndFail("zero-review-fixes", { ...real, caps: { ...real.caps, reviewFixAttempts: 0 } }) ?? ""), "a zero review fixer bound was accepted", ) + T( + `${NAME}: a branch launch cap must be a positive integer`, + /caps\.workerLaunchesPerBranch must be a positive integer/.test( + readAndFail("fractional-launch-cap", { ...real, caps: { ...real.caps, workerLaunchesPerBranch: 1.5 } }) ?? "", + ), + "a fractional worker launch cap was accepted", + ) T( `${NAME}: a cloud cap outside the measured 4 through 8 range is refused`, /caps\.cloudParallelTasks must be an integer from 4 through 8/.test( @@ -248,6 +261,7 @@ export const cases = async () => { real.timeouts.receiptLockSeconds === 1 && real.caps.cloudParallelTasks === 8 && real.caps.parallelTickets === 3 && + real.caps.workerLaunchesPerBranch === 2 && real.caps.workerLogMegabytes === 512, JSON.stringify({ cloud: real.cloud, timeouts: real.timeouts, caps: real.caps }), ) diff --git a/tools/__tests__/reseed-calibration.mjs b/tools/__tests__/reseed-calibration.mjs index b505eb021..e1a9b6445 100644 --- a/tools/__tests__/reseed-calibration.mjs +++ b/tools/__tests__/reseed-calibration.mjs @@ -55,7 +55,7 @@ const stage = (label, stamp) => { write( join(fixture, ".claude", "orchestrator.json"), `${JSON.stringify( - { worker: "codex", workers: { codex: { command: "codex", args: ["exec"], models: { default: { model: "gpt-5.6-sol", args: [] } } } } }, + { worker: "codex", workers: { codex: { command: "codex", args: ["exec"], models: { default: { model: "gpt-5.6-sol", args: [] }, mechanical: { model: "gpt-5.6-sol", args: ["-c", 'model_reasoning_effort="medium"'] } } } } }, null, 2, )}\n`, @@ -149,7 +149,9 @@ export const cases = () => { * against the model that reads it. Carrying dates forward there would be the same silent renewal * in the opposite direction. */ - const moved = stage("worker-moved", backdated(first, 200, { workerModel: "gpt-6-next" })) + const moved = stage("worker-moved", backdated(first, 200, { + workerTiers: { ...first.workerTiers, default: { ...first.workerTiers.default, model: "gpt-6-next" } }, + })) check(TOOL, "a moved worker model renews every verdict, because it decays all of them at once", ["--root", moved], { status: 0, stdout: new RegExp(`${COUNT} verdict\\(s\\) renewed, 0 carried forward`), @@ -172,7 +174,7 @@ export const cases = () => { for (const file of FILES.filter((entry) => entry.path !== A_SKILL)) write(join(deleted, ...file.path.split("/")), file.body) write( join(deleted, ".claude", "orchestrator.json"), - `${JSON.stringify({ worker: "codex", workers: { codex: { command: "codex", args: ["exec"], models: { default: { model: "gpt-5.6-sol", args: [] } } } } }, null, 2)}\n`, + `${JSON.stringify({ worker: "codex", workers: { codex: { command: "codex", args: ["exec"], models: { default: { model: "gpt-5.6-sol", args: [] }, mechanical: { model: "gpt-5.6-sol", args: ["-c", 'model_reasoning_effort="medium"'] } } } } }, null, 2)}\n`, ) check(TOOL, "a verdict whose file is gone is refused, so a deleted skill cannot leave one behind", ["--root", deleted], { nonZero: true, diff --git a/tools/__tests__/run-state.mjs b/tools/__tests__/run-state.mjs index f6140cd12..c661334eb 100644 --- a/tools/__tests__/run-state.mjs +++ b/tools/__tests__/run-state.mjs @@ -1,9 +1,10 @@ import { existsSync, mkdirSync, writeFileSync } from "node:fs" import { join } from "node:path" +import { Worker } from "node:worker_threads" import { T, root } from "./_harness.mjs" -const { clearWakeSource, readRunState, readWakeSources, registerWakeSource, runStatePath, wakeSourceDirectory, writeRunState } = await import("../lib/run-state.mjs") +const { clearWakeSource, readRunState, readWakeSources, readWorkerLaunches, registerWakeSource, reserveWorkerLaunch, runStatePath, wakeSourceDirectory, workerLaunchDirectory, writeRunState } = await import("../lib/run-state.mjs") const TOOL = "lib/run-state.mjs" @@ -14,7 +15,97 @@ const stageCheckout = (label) => { return repoRoot } -export const cases = () => { +const reserveTogether = (moduleUrl, repoRoot, launch, cap) => { + const signal = new SharedArrayBuffer(2 * Int32Array.BYTES_PER_ELEMENT) + const workerSource = ` +const fs = require("node:fs") +const { syncBuiltinESMExports } = require("node:module") +const { parentPort, workerData } = require("node:worker_threads") +const signal = new Int32Array(workerData.signal) +const originalOpenSync = fs.openSync +let synchronized = false +fs.openSync = (...args) => { + if (!synchronized && String(args[0]).includes("orbit-worker-launches") && args[1] === "wx") { + synchronized = true + if (Atomics.add(signal, 0, 1) + 1 === 2) { + Atomics.store(signal, 1, 1) + Atomics.notify(signal, 1, 2) + } else { + while (Atomics.load(signal, 1) === 0) Atomics.wait(signal, 1, 0) + } + } + return originalOpenSync(...args) +} +syncBuiltinESMExports() +import(workerData.moduleUrl).then(({ reserveWorkerLaunch }) => { + parentPort.postMessage(reserveWorkerLaunch(workerData.launch, workerData.cap, workerData.repoRoot)) +}) +` + return Promise.all([1, 2].map((index) => new Promise((resolve, reject) => { + const worker = new Worker(workerSource, { + eval: true, + execArgv: [], + workerData: { + signal, + moduleUrl, + repoRoot, + cap, + launch: { ...launch, timestamp: `2026-09-14T00:0${index}:00.000Z` }, + }, + }) + worker.once("message", resolve) + worker.once("error", reject) + }))) +} + +const reserveAfterLaterReturns = (moduleUrl, repoRoot, launch) => { + const signal = new SharedArrayBuffer(2 * Int32Array.BYTES_PER_ELEMENT) + const workerSource = ` +const fs = require("node:fs") +const { syncBuiltinESMExports } = require("node:module") +const { parentPort, workerData } = require("node:worker_threads") +const signal = new Int32Array(workerData.signal) +if (workerData.earlier) { + const originalOpenSync = fs.openSync + fs.openSync = (...args) => { + if (String(args[0]).includes("orbit-worker-launches") && args[1] === "wx") { + Atomics.store(signal, 0, 1) + Atomics.notify(signal, 0) + while (Atomics.load(signal, 1) === 0) Atomics.wait(signal, 1, 0) + } + return originalOpenSync(...args) + } + syncBuiltinESMExports() +} else { + while (Atomics.load(signal, 0) === 0) Atomics.wait(signal, 0, 0) +} +import(workerData.moduleUrl).then(({ reserveWorkerLaunch }) => { + const reservation = reserveWorkerLaunch(workerData.launch, 1, workerData.repoRoot) + if (!workerData.earlier) { + Atomics.store(signal, 1, 1) + Atomics.notify(signal, 1) + } + parentPort.postMessage(reservation) +}) +` + return Promise.all([true, false].map((earlier) => new Promise((resolve, reject) => { + const worker = new Worker(workerSource, { + eval: true, + execArgv: [], + workerData: { + signal, + moduleUrl, + repoRoot, + earlier, + launch: { ...launch, timestamp: earlier ? "2026-09-14T00:01:00.000Z" : "2026-09-14T00:02:00.000Z" }, + }, + }) + worker.once("message", resolve) + worker.once("error", reject) + }))) +} + +export const cases = async () => { const repoRoot = stageCheckout("basic") T(`${TOOL}: no run has written a record, so there is no state and no wake source`, readRunState(repoRoot) === null && readWakeSources(repoRoot).length === 0) @@ -59,6 +150,49 @@ export const cases = () => { writeRunState({ sessionId: "s2", sleep: true, remaining: ["ORB-9"], pullRequests: [] }, repoRoot) T(`${TOOL}: a new session starts with a fresh readiness ledger`, readRunState(repoRoot)?.readinessLedger?.length === 0, JSON.stringify(readRunState(repoRoot))) + const launch = { repositoryKey: "ui", branch: "chore/test", headSha: "a".repeat(40), tier: "default", timestamp: "2026-09-14T00:00:00.000Z", relaunchReason: null } + const firstReservation = reserveWorkerLaunch(launch, 1, repoRoot) + const refusedReservation = reserveWorkerLaunch({ ...launch, timestamp: "2026-09-14T00:01:00.000Z" }, 1, repoRoot) + const reasonedReservation = reserveWorkerLaunch({ ...launch, timestamp: "2026-09-14T00:02:00.000Z", relaunchReason: "known conflict list" }, 1, repoRoot) + T( + `${TOOL}: the worker ledger records below-cap launches, refuses the cap, and records a deliberate override`, + firstReservation.allowed && !refusedReservation.allowed && reasonedReservation.allowed && + readWorkerLaunches(repoRoot).length === 2 && + readWorkerLaunches(repoRoot)[1].relaunchReason === "known conflict list" && + existsSync(workerLaunchDirectory(repoRoot)), + JSON.stringify(readWorkerLaunches(repoRoot)), + ) + + const raceRoot = stageCheckout("launch-race") + const racingReservations = await reserveTogether(new URL("../lib/run-state.mjs", import.meta.url).href, raceRoot, launch, 1) + const raceLedger = readWorkerLaunches(raceRoot) + T( + `${TOOL}: simultaneous reservations admit exactly one launch and retain its record`, + racingReservations.filter((reservation) => reservation.allowed).length === 1 && raceLedger.length === 1, + JSON.stringify({ racingReservations, raceLedger }), + ) + + const freeSlotRoot = stageCheckout("launch-race-free-slot") + reserveWorkerLaunch(launch, 2, freeSlotRoot) + const contenders = await reserveTogether(new URL("../lib/run-state.mjs", import.meta.url).href, freeSlotRoot, launch, 2) + const freeSlotLedger = readWorkerLaunches(freeSlotRoot) + T( + `${TOOL}: simultaneous contenders use the one free slot instead of both yielding it`, + contenders.filter((reservation) => reservation.allowed).length === 1 && + contenders.filter((reservation) => !reservation.allowed).length === 1 && + freeSlotLedger.length === 2, + JSON.stringify({ contenders, freeSlotLedger }), + ) + + const delayedCreateRoot = stageCheckout("launch-race-delayed-create") + const delayedCreateReservations = await reserveAfterLaterReturns(new URL("../lib/run-state.mjs", import.meta.url).href, delayedCreateRoot, launch) + const delayedCreateLedger = readWorkerLaunches(delayedCreateRoot) + T( + `${TOOL}: a contender returning before an earlier contender creates cannot exceed the cap`, + delayedCreateReservations.filter((reservation) => reservation.allowed).length === 1 && delayedCreateLedger.length === 1, + JSON.stringify({ delayedCreateReservations, delayedCreateLedger }), + ) + registerWakeSource({ pid: process.pid, what: "worker ORB-1" }, repoRoot) registerWakeSource({ pid: process.ppid, what: "worker ORB-2" }, repoRoot) T( @@ -100,6 +234,7 @@ export const cases = () => { try { registerWakeSource({ pid: 5252, what: "worker ORB-9" }, notADirectory) clearWakeSource(5252, notADirectory) + reserveWorkerLaunch(launch, 1, notADirectory) } catch { threw = true } diff --git a/tools/check-calibration.mjs b/tools/check-calibration.mjs index 1fab9617d..d54aaf09f 100644 --- a/tools/check-calibration.mjs +++ b/tools/check-calibration.mjs @@ -59,10 +59,9 @@ const USAGE = `usage: check-calibration.mjs [--root ] 3. each entry's recorded model and effort match what the file declares today, AND its recorded digest matches the file's complete normalized content, so rewriting a prompt body invalidates the verdict that was written about the old text - 4. the stamp's workerCommand, workerModel and workerArgs match what launch-worker.mjs actually - resolves, taken from resolveWorkerInvocation itself rather than rebuilt here, so workerArgs is - the WHOLE argument vector: engine-level args, then the models.default profile args, then the - model. Comparing the profile half alone let engine-level tuning move without reseeding + 4. the stamp's workerCommand and every workerTiers vector match what launch-worker.mjs actually + resolves, taken from resolveWorkerInvocation itself rather than rebuilt here. Each args value + is the WHOLE vector: engine-level args, then the selected profile args, then the model 5. EVERY entry's own calibratedAt is a real date, is not in the future, is no older than the stamp date, and is at most 90 days old, the backstop for a model alias whose target moved without its declared string changing. Per entry rather than stamp-wide, because one date for @@ -86,24 +85,18 @@ const fail = (code, message) => { /** The backstop for a model alias whose target moved without the declared string changing. */ const MAX_AGE_DAYS = 90 /** - * The profile tier `launch-worker.mjs` resolves the implementer from, at `:122`: - * `resolveWorkerInvocation(engineName, engine, "default")`. - * * The ENGINE is read from `config.worker` rather than hardcoded, because that is what * `launch-worker.mjs:115-116` does (`const engineName = config.worker`, then * `config.workers[engineName]`). Naming `codex` here would have compared the wrong profile the moment * the engine switched, and read a path that no longer exists. * - * The stamp records the RESOLVED argument vector alongside its `model`, because the reasoning effort - * lives in the args (`model_reasoning_effort="high"`), so a model string that never moves can still - * have its tuning changed underneath. An args-only edit has to go red too. + * The stamp records every RESOLVED argument vector alongside its `model`, because the reasoning + * effort lives in the args. A model string that never moves can still have its tuning changed. * * The whole vector, not the profile half: `resolveWorkerInvocation` launches * `[...engine.args, ...profile.args, "--model", model]`, so tuning declared at the ENGINE level is - * just as load-bearing as tuning declared in the profile. Stamping only `models.default.args` left - * engine-level effort outside the gate entirely. + * just as load-bearing as tuning declared in the profile. */ -const AUTHORITATIVE_TIER = "default" let repositoryRoot = resolve(dirname(fileURLToPath(import.meta.url)), "..") const positional = process.argv.slice(2) @@ -127,13 +120,23 @@ if (stamp === null || typeof stamp !== "object" || Array.isArray(stamp)) fail(2, if (typeof stamp.calibratedAt !== "string" || !/^\d{4}-\d{2}-\d{2}$/.test(stamp.calibratedAt)) { fail(2, `check-calibration: calibratedAt must be a YYYY-MM-DD date, got ${JSON.stringify(stamp.calibratedAt)}`) } -if (typeof stamp.workerModel !== "string" || stamp.workerModel === "") fail(2, "check-calibration: workerModel must be a non-empty string") if (typeof stamp.workerEngine !== "string" || stamp.workerEngine === "") fail(2, "check-calibration: workerEngine must be a non-empty string") if (typeof stamp.workerCommand !== "string" || stamp.workerCommand === "") { fail(2, "check-calibration: workerCommand must be a non-empty string, because the executable is what actually runs the work") } -if (!Array.isArray(stamp.workerArgs) || stamp.workerArgs.some((argument) => typeof argument !== "string")) { - fail(2, "check-calibration: workerArgs must be an array of strings, because it is the resolved launch vector and the reasoning effort lives in it") +if (stamp.workerTiers === null || typeof stamp.workerTiers !== "object" || Array.isArray(stamp.workerTiers)) { + fail(2, "check-calibration: workerTiers must be an object keyed by configured tier") +} +for (const [tier, profile] of Object.entries(stamp.workerTiers)) { + if (profile === null || typeof profile !== "object" || Array.isArray(profile) || typeof profile.model !== "string" || profile.model === "") { + fail(2, `check-calibration: workerTiers.${tier}.model must be a non-empty string`) + } + if (!Array.isArray(profile.args) || profile.args.some((argument) => typeof argument !== "string")) { + fail(2, `check-calibration: workerTiers.${tier}.args must be an array of strings`) + } +} +if (typeof stamp.workerTierVerdict !== "string" || stamp.workerTierVerdict.trim() === "") { + fail(2, "check-calibration: workerTierVerdict must name which orders use each tier") } if (stamp.entries === null || typeof stamp.entries !== "object" || Array.isArray(stamp.entries)) { fail(2, "check-calibration: entries must be an object keyed by repository-relative path") @@ -253,20 +256,22 @@ if (typeof configuredEngine !== "string" || configuredEngine === "") { fail(2, "check-calibration: .claude/orchestrator.json declares no `worker`, so there is no engine to resolve the implementer profile from") } /** - * Resolved by the CANONICAL resolver, never rebuilt here. `resolveWorkerInvocation` prepends the - * engine's own args before the profile's, so reading `models.default.args` alone stamped half of what - * launches: an engine-level `-c model_reasoning_effort="low"` beside an empty profile args array left - * this gate green while every worker launched at low effort. A gate that cannot see the tuning it - * exists to pin is the gate-that-cannot-fail this tool was written to undo. + * Resolved by the CANONICAL resolver, never rebuilt here. Every configured tier is stamped because + * either can now launch a worker, and omitting one would let its effort drift without failing CI. */ -let configuredInvocation +const configuredTierNames = Object.keys(orchestrator?.workers?.[configuredEngine]?.models ?? {}).sort() +const configuredTiers = {} try { - configuredInvocation = resolveWorkerInvocation(configuredEngine, orchestrator?.workers?.[configuredEngine], AUTHORITATIVE_TIER) + if (configuredTierNames.length === 0) { + resolveWorkerInvocation(configuredEngine, orchestrator?.workers?.[configuredEngine]) + } + for (const tier of configuredTierNames) { + const invocation = resolveWorkerInvocation(configuredEngine, orchestrator?.workers?.[configuredEngine], tier) + configuredTiers[tier] = { model: invocation.model, args: invocation.args } + } } catch (error) { fail(2, `check-calibration: ${error.message}, so the model-match assertion has nothing to compare against`) } -const configuredModel = configuredInvocation.model -const configuredArgs = configuredInvocation.args /** * The EXECUTABLE, which `resolveWorkerInvocation` does not return: `launch-worker.mjs:194` spawns * `engine.command` and the invocation only describes what is passed TO it. Swapping that command while @@ -280,13 +285,20 @@ if (stamp.workerEngine !== configuredEngine) { if (stamp.workerCommand !== configuredCommand) { problems.push(`the worker command is ${JSON.stringify(configuredCommand ?? null)} and the stamp was taken against ${JSON.stringify(stamp.workerCommand)}; a different executable is a different implementer, so recalibrate`) } -if (stamp.workerModel !== configuredModel) { - problems.push(`the worker model is ${configuredModel} and the stamp was taken against ${stamp.workerModel}; recalibrate in the same pull request that moved it`) +const stampedTierNames = Object.keys(stamp.workerTiers).sort() +if (JSON.stringify(stampedTierNames) !== JSON.stringify(configuredTierNames)) { + problems.push(`the configured worker tiers are ${configuredTierNames.join(", ") || "none"} and the stamp names ${stampedTierNames.join(", ") || "none"}; recalibrate every selectable launch vector`) } -if (JSON.stringify(stamp.workerArgs) !== JSON.stringify(configuredArgs)) { - problems.push( - `the worker args are ${JSON.stringify(configuredArgs)} and the stamp was taken against ${JSON.stringify(stamp.workerArgs)}; this is the whole resolved launch vector, engine args included, so an args-only change decays the tuning exactly like a model change`, - ) +for (const tier of configuredTierNames) { + const configured = configuredTiers[tier] + const stamped = stamp.workerTiers[tier] + if (!stamped) continue + if (stamped.model !== configured.model) { + problems.push(`the ${tier} tier worker model is ${configured.model} and the stamp was taken against ${stamped.model}; recalibrate in the same pull request that moved it`) + } + if (JSON.stringify(stamped.args) !== JSON.stringify(configured.args)) { + problems.push(`the ${tier} tier worker args are ${JSON.stringify(configured.args)} and the stamp was taken against ${JSON.stringify(stamped.args)}; this is the whole resolved launch vector, engine args included`) + } } /** @@ -354,5 +366,5 @@ if (problems.length > 0) { const oldestVerdictAge = Object.values(stamp.entries).reduce((oldest, entry) => Math.max(oldest, ageInDays(entry.calibratedAt, "entry calibratedAt")), 0) console.log( - `check-calibration: ${files.length} calibrated file(s) stamped ${stamp.calibratedAt} against ${configuredEngine} ${configuredModel} ${JSON.stringify(configuredArgs)}, oldest verdict ${oldestVerdictAge} day(s) old.`, + `check-calibration: ${files.length} calibrated file(s) stamped ${stamp.calibratedAt} against ${configuredEngine} tiers ${configuredTierNames.join(", ")}, oldest verdict ${oldestVerdictAge} day(s) old.`, ) diff --git a/tools/launch-worker.mjs b/tools/launch-worker.mjs index d74cfe08b..257cab091 100644 --- a/tools/launch-worker.mjs +++ b/tools/launch-worker.mjs @@ -13,14 +13,14 @@ */ import { spawn, spawnSync } from "node:child_process" -import { closeSync, existsSync, mkdirSync, openSync, readFileSync, readdirSync, renameSync, statSync, writeFileSync } from "node:fs" +import { closeSync, existsSync, mkdirSync, openSync, readFileSync, readdirSync, realpathSync, renameSync, statSync, writeFileSync } from "node:fs" import { tmpdir } from "node:os" import { delimiter, dirname, extname, join, resolve } from "node:path" import { githubEnvironment, redactSecrets } from "./lib/github-auth.mjs" import { resolveTicket } from "./lib/github-issues.mjs" import { readOrchestratorConfig, resolveWorkerInvocation } from "./lib/orchestrator-config.mjs" -import { clearWakeSource, registerWakeSource } from "./lib/run-state.mjs" +import { clearWakeSource, registerWakeSource, reserveWorkerLaunch } from "./lib/run-state.mjs" const USAGE = `usage: launch-worker.mjs --issue --worktree --prompt [options] @@ -35,6 +35,10 @@ const USAGE = `usage: launch-worker.mjs --issue --worktree - --hard-ceiling-minutes this ticket's hard ceiling, replacing timeouts.hardCeilingMinutes for this one launch. For a ticket that legitimately outruns the fleet-wide default + --tier + worker profile for this order (default: default) + --relaunch-reason + deliberate reason for launching after this branch reaches its configured cap --dry-run print the resolved plan as JSON and exit 0, spawning nothing --help, -h print this usage and exit 0 @@ -66,9 +70,19 @@ const issueArgument = argOf("--issue") const worktreeArg = argOf("--worktree") const promptArg = argOf("--prompt") const hardCeilingArg = argOf("--hard-ceiling-minutes") +const tierValue = argOf("--tier") +const tierArgument = tierValue ?? "default" +const relaunchReasonArgument = argOf("--relaunch-reason") const measurement = process.argv.includes("--measurement") const dryRun = process.argv.includes("--dry-run") +if ((process.argv.includes("--tier") && typeof tierValue !== "string") || !new Set(["default", "mechanical"]).has(tierArgument)) { + fail(2, `${USAGE}\n\n--tier must be default or mechanical, got "${tierArgument}"`) +} +if (process.argv.includes("--relaunch-reason") && (typeof relaunchReasonArgument !== "string" || relaunchReasonArgument.startsWith("--") || relaunchReasonArgument.trim() === "")) { + fail(2, `${USAGE}\n\n--relaunch-reason must be non-empty text`) +} + let issue try { const resolvedTicket = resolveTicket(issueArgument) @@ -116,10 +130,9 @@ const engineName = config.worker const engine = config.workers[engineName] if (!engine.command) fail(2, `.claude/orchestrator.json names worker "${engineName}" but carries no command for it`) -/** One ticket, one worker, one model: "default" is the only tier this launcher ever resolves. */ let invocation try { - invocation = resolveWorkerInvocation(engineName, engine, "default") + invocation = resolveWorkerInvocation(engineName, engine, tierArgument) } catch (error) { fail(2, error.message) } @@ -255,6 +268,37 @@ if (dryRun) { process.exit(0) } +const gitRepositoryIdentity = (directory) => { + const result = spawnSync("git", ["-C", directory, "rev-parse", "--git-common-dir"], { encoding: "utf8", windowsHide: true }) + if (result.error || result.status !== 0 || result.stdout.trim() === "") return null + try { + const identity = realpathSync.native(resolve(directory, result.stdout.trim())) + return process.platform === "win32" ? identity.toLowerCase() : identity + } catch { + return null + } +} +const repositoryIdentity = gitRepositoryIdentity(runDirectory) +if (!repositoryIdentity) fail(2, `could not resolve the Git repository identity for ${runDirectory}`) +const repositoryKey = Object.entries(config.repos ?? {}).find(([, repository]) => + typeof repository === "string" && gitRepositoryIdentity(repository) === repositoryIdentity)?.[0] +if (!repositoryKey) fail(2, `${runDirectory} does not belong to a repository configured in .claude/orchestrator.json`) + +const timestamp = new Date().toISOString() +const reservation = reserveWorkerLaunch({ + repositoryKey, + branch, + headSha: startHead, + tier: invocation.tier, + timestamp, + relaunchReason: relaunchReasonArgument, +}, config.caps.workerLaunchesPerBranch, runDirectory) +if (!reservation.allowed) { + const earlier = reservation.earlierLaunches.map((launch, index) => + ` ${index + 1}. ${launch.timestamp} tier=${launch.tier} head=${launch.headSha}`).join("\n") + fail(2, `worker launch cap ${config.caps.workerLaunchesPerBranch} reached for ${repositoryKey} branch ${branch}. Earlier launches:\n${earlier}\nPass --relaunch-reason "" to record and allow another launch.`) +} + console.error(`starting the ${engineName} worker for ${issue} in ${runDirectory}; log: ${logFile}`) const startedAt = new Date().toISOString() /** diff --git a/tools/lib/orchestrator-config.mjs b/tools/lib/orchestrator-config.mjs index 7a0fdb130..ba8f9793c 100644 --- a/tools/lib/orchestrator-config.mjs +++ b/tools/lib/orchestrator-config.mjs @@ -131,6 +131,9 @@ export const readOrchestratorConfig = (configUrl = DEFAULT_CONFIG_URL, baseBranc positive(config.timeouts?.noProgressMinutes, "timeouts.noProgressMinutes") positive(config.timeouts?.pollSeconds, "timeouts.pollSeconds") positive(config.caps?.reviewFixAttempts, "caps.reviewFixAttempts") + if (!Number.isInteger(config.caps?.workerLaunchesPerBranch) || config.caps.workerLaunchesPerBranch <= 0) { + throw new Error(".claude/orchestrator.json caps.workerLaunchesPerBranch must be a positive integer") + } if (!Number.isInteger(config.caps?.cloudParallelTasks) || config.caps.cloudParallelTasks < 4 || config.caps.cloudParallelTasks > 8) { throw new Error(".claude/orchestrator.json caps.cloudParallelTasks must be an integer from 4 through 8") } @@ -146,11 +149,8 @@ export const readOrchestratorConfig = (configUrl = DEFAULT_CONFIG_URL, baseBranc } /** - * `tier` is a plain string, not a label array. The tier:cheap / tier:deep label machinery is gone - * with the wave planner that set it: one ticket, one worker, one model. D21 fixes the implementer - * at one model @ high, gpt-5.6-sol since 2026-09-07, so "default" is the only tier any launch resolves. - * The harness no longer runs a reviewer of its own: Pullfrog reviews in GitHub Actions and publishes the - * `pullfrog-approval` required check, so there is no second tier to declare. + * `tier` is a plain string, not a label array. The caller chooses the configured profile for the + * order in front of it; the resolver owns the one argument path shared by every tier. */ export const resolveWorkerInvocation = (engineName, engine, tier = "default") => { if (!isRecord(engine)) { diff --git a/tools/lib/run-state.mjs b/tools/lib/run-state.mjs index 85d245166..9c39bf570 100644 --- a/tools/lib/run-state.mjs +++ b/tools/lib/run-state.mjs @@ -7,13 +7,14 @@ * artifact trail identical to a run that finished. A queue that ends silently is worse than one that * fails loudly, because nobody looks for it. * - * Two files, in `.git/`, because that directory is per-checkout, never committed, always writable, + * These records live in `.git/`, because that directory is per-checkout, never committed, writable, * and needs no gitignore entry: * * .git/orbit-orchestrate-run.json the ORCHESTRATOR is its only writer: which session, whether * --sleep is on, and which tickets remain. * .git/orbit-wake-sources/.json one file per live wake source, written by launch-worker.mjs * when it starts and removed when it exits. + * .git/orbit-worker-launches/.json one file per attempted worker launch that was admitted. * * One file per wake source rather than an array in one file: under `--parallel` three launchers write * at once, and a read-modify-write on a shared array loses entries. A crashed launcher leaks its file @@ -24,7 +25,8 @@ */ import { spawnSync } from "node:child_process" -import { existsSync, mkdirSync, readFileSync, readdirSync, rmSync, statSync, writeFileSync } from "node:fs" +import { createHash, randomUUID } from "node:crypto" +import { closeSync, existsSync, mkdirSync, openSync, readFileSync, readdirSync, rmSync, statSync, writeFileSync } from "node:fs" import { join, resolve } from "node:path" import { fileURLToPath } from "node:url" @@ -54,6 +56,86 @@ export const REPO_ROOT = fileURLToPath(new URL("../..", import.meta.url)) export const runStatePath = (repoRoot = REPO_ROOT) => join(gitDirectoryOf(repoRoot), "orbit-orchestrate-run.json") export const wakeSourceDirectory = (repoRoot = REPO_ROOT) => join(gitDirectoryOf(repoRoot), "orbit-wake-sources") +export const workerLaunchDirectory = (repoRoot = REPO_ROOT) => join(gitDirectoryOf(repoRoot), "orbit-worker-launches") + +const readWorkerLaunchRecords = (repoRoot = REPO_ROOT) => { + const directory = workerLaunchDirectory(repoRoot) + let names + try { + names = readdirSync(directory).filter((name) => name.endsWith(".json")) + } catch { + return [] + } + const records = [] + for (const name of names) { + try { + records.push({ name, launch: JSON.parse(readFileSync(join(directory, name), "utf8")) }) + } catch { + /* an unreadable launch cannot prove that the branch cap was reached */ + } + } + return records.sort((left, right) => + String(left.launch?.timestamp).localeCompare(String(right.launch?.timestamp)) || left.name.localeCompare(right.name)) +} + +export const readWorkerLaunches = (repoRoot = REPO_ROOT) => readWorkerLaunchRecords(repoRoot).map((record) => record.launch) + +const readLaunchFiles = (paths) => paths.flatMap((path) => { + try { + return [JSON.parse(readFileSync(path, "utf8"))] + } catch { + return [] + } +}) + +const claimLaunchFile = (path, launch) => { + let fileDescriptor + try { + fileDescriptor = openSync(path, "wx") + } catch { + return { claimed: false, occupied: existsSync(path), recorded: false } + } + try { + writeFileSync(fileDescriptor, `${JSON.stringify(launch, null, 2)}\n`) + return { claimed: true, occupied: true, recorded: true } + } catch { + return { claimed: true, occupied: true, recorded: false } + } finally { + try { + closeSync(fileDescriptor) + } catch { + /* the exclusive create already claimed the slot, so closing remains fail-soft */ + } + } +} + +/** Each exact slot name can be claimed once, so no read-then-decide race can exceed the cap. */ +export const reserveWorkerLaunch = (launch, cap, repoRoot = REPO_ROOT) => { + const directory = workerLaunchDirectory(repoRoot) + try { + mkdirSync(directory, { recursive: true }) + } catch { + return { allowed: true, earlierLaunches: [], recorded: false } + } + const branchKey = createHash("sha256").update(`${launch.repositoryKey}\0${launch.branch}`).digest("hex") + const slotPaths = Array.from({ length: cap }, (_, index) => join(directory, `${branchKey}-slot-${index + 1}.json`)) + for (let index = 0; index < slotPaths.length; index++) { + const claim = claimLaunchFile(slotPaths[index], launch) + if (claim.claimed) { + return { allowed: true, earlierLaunches: readLaunchFiles(slotPaths.slice(0, index)), recorded: claim.recorded } + } + if (!claim.occupied) { + return { allowed: true, earlierLaunches: readLaunchFiles(slotPaths.slice(0, index)), recorded: false } + } + } + const earlierLaunches = readLaunchFiles(slotPaths) + if (launch.relaunchReason === null) { + return { allowed: false, earlierLaunches, recorded: false } + } + const overridePath = join(directory, `${branchKey}-override-${randomUUID()}.json`) + const override = claimLaunchFile(overridePath, launch) + return { allowed: true, earlierLaunches, recorded: override.recorded } +} /** The orchestrator's own run record, or null when no run has written one. */ export const readRunState = (repoRoot = REPO_ROOT) => { diff --git a/tools/reseed-calibration.mjs b/tools/reseed-calibration.mjs index d98b2471d..1059d5961 100644 --- a/tools/reseed-calibration.mjs +++ b/tools/reseed-calibration.mjs @@ -95,7 +95,7 @@ const VERDICTS = { ".claude/skills/drift-review/SKILL.md": "undeclared, inherits the session: it judges repeated evidence against the current workflow files, but every result remains a staged candidate for human review.", ".claude/skills/handoff/SKILL.md": - "current: high effort, and it earns it: it decides what a fresh session cannot rediscover, and under-thinking it is how a handoff loses the one fact written nowhere else.", + "current: high effort, and it earns it. It decides what survives into a spec that outlives every session, and under-thinking it is how a rule Thomas set in week one disappears by week four.", ".claude/skills/investigate/SKILL.md": "undeclared, inherits the session: root-causing a production incident across Sentry, Render, Postgres and the LSP is judgement, so this is a follow-up candidate.", ".claude/skills/lesson/SKILL.md": @@ -106,6 +106,10 @@ const VERDICTS = { "current: high effort, and it earns it: it plans the queue, verifies delivery from artifacts and clears the review, and it is the entry point every other piece of work passes through.", ".claude/skills/prod-readiness/SKILL.md": "undeclared, inherits the session: it consolidates four child audits into one honest launch verdict, which is judgement, so this is a follow-up candidate.", + ".claude/skills/progress/SKILL.md": + "current: medium effort, because it reads live git and ticket state and must judge whether a part-built screen is honestly described, which low effort gets wrong by rounding up.", + ".claude/skills/questions/SKILL.md": + "current: high effort, because the filter decides what NOT to ask, and a wrong call either wastes his attention or ships a guess as a decision.", ".claude/skills/second-opinion/SKILL.md": "current with nothing to declare: the reasoning happens in the other model, by construction. Declaring an effort here would tune the wrong side of the call.", ".claude/skills/sleep/SKILL.md": @@ -158,8 +162,7 @@ if (extra.length > 0) throw new Error(`verdict written for a file that is not in const config = JSON.parse(readFileSync(join(root, ".claude", "orchestrator.json"), "utf8")) // The ENGINE comes from config.worker, the same key launch-worker.mjs:115 reads. The invocation comes // from resolveWorkerInvocation itself rather than being rebuilt here, so the stamp records the WHOLE -// vector that launches: engine args, then the models.default profile args, then the model. Reading -// models.default.args alone left engine-level reasoning effort outside the gate entirely. +// vector that launches: engine args, then the selected profile args, then the model. /** * The canonical resolver, imported from THIS tool's own directory rather than from `--root`. Loading * it out of the target tree made the pass unrunnable against any root that is not a full checkout, @@ -168,12 +171,14 @@ const config = JSON.parse(readFileSync(join(root, ".claude", "orchestrator.json" * being stamped. */ const workerEngine = config.worker -const invocation = resolveWorkerInvocation(workerEngine, config.workers[workerEngine], "default") // The executable itself, which resolveWorkerInvocation does not return: launch-worker.mjs spawns // engine.command and the invocation only describes what is passed TO it. const workerCommand = config.workers[workerEngine].command -const workerModel = invocation.model -const workerArgs = invocation.args +const workerTiers = Object.fromEntries(Object.keys(config.workers[workerEngine].models).sort().map((tier) => { + const invocation = resolveWorkerInvocation(workerEngine, config.workers[workerEngine], tier) + return [tier, { model: invocation.model, args: invocation.args }] +})) +const workerTierVerdict = "default handles product, design, architecture, and ambiguous orders requiring judgment; mechanical handles merge-forward work, known conflict lists, and reviewer-directed test experiments" const now = new Date() const calibratedAt = `${now.getUTCFullYear()}-${String(now.getUTCMonth() + 1).padStart(2, "0")}-${String(now.getUTCDate()).padStart(2, "0")}` @@ -188,8 +193,8 @@ const previous = existsSync(stampPath) ? JSON.parse(readFileSync(stampPath, "utf const workerMoved = previous.workerEngine !== workerEngine || previous.workerCommand !== workerCommand || - previous.workerModel !== workerModel || - JSON.stringify(previous.workerArgs) !== JSON.stringify(workerArgs) + JSON.stringify(previous.workerTiers) !== JSON.stringify(workerTiers) || + previous.workerTierVerdict !== workerTierVerdict const entries = {} let renewed = 0 @@ -221,14 +226,14 @@ const stamp = { calibratedAt, workerEngine, workerCommand, - workerModel, - workerArgs, + workerTiers, + workerTierVerdict, workerModelSource: - 'resolveWorkerInvocation(config.worker, ..., "default") in tools/lib/orchestrator-config.mjs, the exact vector launch-worker.mjs launches: engine args, then models.default args, then the model', + "resolveWorkerInvocation(config.worker, ..., tier) in tools/lib/orchestrator-config.mjs for every configured tier, with each exact launch vector: engine args, then selected profile args, then the model", entries, } writeFileSync(stampPath, `${JSON.stringify(stamp, null, 2)}\n`, "utf8") console.log( - `stamped ${files.length} file(s) at ${calibratedAt} against ${workerEngine} ${workerModel} ${JSON.stringify(workerArgs)}; ${renewed} verdict(s) renewed, ${files.length - renewed} carried forward.`, + `stamped ${files.length} file(s) at ${calibratedAt} against ${workerEngine} tiers ${Object.keys(workerTiers).join(", ")}; ${renewed} verdict(s) renewed, ${files.length - renewed} carried forward.`, )