diff --git a/.changeset/review-roster-opus-5.md b/.changeset/review-roster-opus-5.md new file mode 100644 index 00000000..fa09caf2 --- /dev/null +++ b/.changeset/review-roster-opus-5.md @@ -0,0 +1,7 @@ +--- +"review": minor +--- + +Move the reviewer roster to Opus 5 (`claude-opus-5`). `pattern-triage` stays on Sonnet 4.6 as the cheap first pass. + +The specialist lenses were kept off Fable 5 because cyber safety classifiers can refuse benign security-focused analysis, and a refused lens is a silent coverage hole: it surfaces as a missing agent result, not an error. Opus 5 can also return `stop_reason: "refusal"` on cyber-adjacent input, so this move re-opens that hole rather than closing it. The detector is the weekly drift corpus, where a refusing lens craters must-catch recall on security-adjacent cases while every other metric looks normal. On that signature, put the `security-auth` lens back on `claude-opus-4-8` first. Moving `security-auth` with the roster rather than carving it out pre-emptively is deliberate: the refusal hazard is intermittent and unproven on Opus 5, the one-hop runtime fallback (a refused agent re-dispatches once on `claude-opus-4-8`, recorded as `fellBackTo`) bounds each occurrence, and a uniform roster leaves one signature to watch and one revert to make instead of a permanent carve-out justified by a hazard that may not materialize. diff --git a/.github/workflows/review-eval-drift.yml b/.github/workflows/review-eval-drift.yml index 5c60bea2..c1df25ce 100644 --- a/.github/workflows/review-eval-drift.yml +++ b/.github/workflows/review-eval-drift.yml @@ -48,7 +48,7 @@ on: # 29855626692-29855643020: $97.19 / 90), so a drift week with both # arms on this roster is ~$194; 220 clears it with real headroom, # which matters now that a budget-skipped tail is a red run (the - # caseAsymmetry gate). If the swap is reverted, dial back toward 160. + # caseAsymmetry gate). description: "Total hard budget across both arms and all repeats" required: false default: "220" diff --git a/package.json b/package.json index dccd2a8a..6521a46e 100644 --- a/package.json +++ b/package.json @@ -10,7 +10,7 @@ "build": "tsc -p actions/tsconfig.json" }, "devDependencies": { - "@anthropic-ai/claude-agent-sdk": "^0.3.205", + "@anthropic-ai/claude-agent-sdk": "^0.3.219", "@changesets/cli": "^2.29.8", "@khanacademy/eslint-config": "^0.1.0", "@swc-node/register": "^1.11.1", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 18d1721b..14d14744 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -9,8 +9,8 @@ importers: .: devDependencies: '@anthropic-ai/claude-agent-sdk': - specifier: ^0.3.205 - version: 0.3.205(@anthropic-ai/sdk@0.110.0(zod@4.4.3))(@modelcontextprotocol/sdk@1.29.0(zod@4.4.3))(zod@4.4.3) + specifier: ^0.3.219 + version: 0.3.220(@anthropic-ai/sdk@0.110.0(zod@4.4.3))(@modelcontextprotocol/sdk@1.29.0(zod@4.4.3))(zod@4.4.3) '@changesets/cli': specifier: ^2.29.8 version: 2.29.8(@types/node@25.3.3) @@ -134,41 +134,81 @@ packages: cpu: [arm64] os: [darwin] + '@anthropic-ai/claude-agent-sdk-darwin-arm64@0.3.220': + resolution: {integrity: sha512-7VxlbEosK7DODiOnsjoVd0DSJzbnaPrM2jelMHI0y8zx1UnLS3WC6EFUXbvy74F2sXqEznh2tzn7EKWInaRN6Q==} + cpu: [arm64] + os: [darwin] + '@anthropic-ai/claude-agent-sdk-darwin-x64@0.3.205': resolution: {integrity: sha512-G6ETPmL5mNzJ2DFsWxG3jmsmrXgZX1N2ZCJvxaGUUpjTsKZJ4Tup1cWYvcd/m7o5fYZmx9REmgzTwsAIc1fdPQ==} cpu: [x64] os: [darwin] + '@anthropic-ai/claude-agent-sdk-darwin-x64@0.3.220': + resolution: {integrity: sha512-X9RwDsSmbF6ultKZroaip+DL8WRgC64gHbrAwrRlAFSPNZV7zmJyP2ur8rW7KrxqmtuehdMMkw8+SAC/6hD2PA==} + cpu: [x64] + os: [darwin] + '@anthropic-ai/claude-agent-sdk-linux-arm64-musl@0.3.205': resolution: {integrity: sha512-91fgdG4aTnQ29sKOcUqgH4+tKCW2ut6PWGRSYmXNDbROasJm1rAlPdzC5brdu/e4c0CDSNV6TWyE5JCjaS/jlQ==} cpu: [arm64] os: [linux] + '@anthropic-ai/claude-agent-sdk-linux-arm64-musl@0.3.220': + resolution: {integrity: sha512-OHoZOZ8Cf2TBr6oXIXPwyvUxj9jrq2w8E4poA8dMpacXszcPSPiCQCMuuOh4aWJzfeJE1+TtWxhKMVb2csXyZQ==} + cpu: [arm64] + os: [linux] + '@anthropic-ai/claude-agent-sdk-linux-arm64@0.3.205': resolution: {integrity: sha512-CXzySK3PV3EizCRPXnxPqeaAtgrBFDnMFOVpMe36oC3U16yDb1b1tAJGqZi/7uFrVvAiaXvnSFxhUWnDDSaO+A==} cpu: [arm64] os: [linux] + '@anthropic-ai/claude-agent-sdk-linux-arm64@0.3.220': + resolution: {integrity: sha512-WkROPwWskqhKR9XgnmseHQ6rLi9zM9qt57IWoToIjL/eXOqDWipp7JXZ1L5ud+LrA42dunHPZfBwD/vXZ+A7LA==} + cpu: [arm64] + os: [linux] + '@anthropic-ai/claude-agent-sdk-linux-x64-musl@0.3.205': resolution: {integrity: sha512-vvsb7GlnA8CTSVvvTkrXjcSeRKqxSM7p/tU3Od9ICAZeWHglptekEyzLEApzLuLbI5ewfFF/F0q3NwOBbo18dg==} cpu: [x64] os: [linux] + '@anthropic-ai/claude-agent-sdk-linux-x64-musl@0.3.220': + resolution: {integrity: sha512-K+FWj+LcGhC1Z7wqeWoLxm1iemcba5xKpLLFVwYm4V6HyMx3ruYd/2r2TiQtjT+JWeNFWIys0ScHiItR6vWAiA==} + cpu: [x64] + os: [linux] + '@anthropic-ai/claude-agent-sdk-linux-x64@0.3.205': resolution: {integrity: sha512-siS+1iNqBSlGFZZvJY6+mhzZ/6/ec/TbX9GMuwmTF0E6fxGhIIp797jJxR1q8r6FAq7d39mEoRNhC0Ffo60uNQ==} cpu: [x64] os: [linux] + '@anthropic-ai/claude-agent-sdk-linux-x64@0.3.220': + resolution: {integrity: sha512-tkTJFnpR9VifvWX2fmkCAPkT6+8Wk/gVu8B5jsVekKZPiZoWRHmMXO30BnZn+f0TZhgYP+82PSX3S8crH1kn+w==} + cpu: [x64] + os: [linux] + '@anthropic-ai/claude-agent-sdk-win32-arm64@0.3.205': resolution: {integrity: sha512-SpP5zF68weFez/6pKrGzq/UVAJDMDNphWqmkLfOpWTDBL5xy6XlIZw5Bl4EXoVnfi2VLFkwuffNeFe+9SdX7kw==} cpu: [arm64] os: [win32] + '@anthropic-ai/claude-agent-sdk-win32-arm64@0.3.220': + resolution: {integrity: sha512-rIwgq0UwQExWl6KrHUyC4w5KwpL9l6nd95aUTx6RitexaAuEw//xtfTVLnuE4hDDQZFkzEwpdKc3nxDWoGcUbA==} + cpu: [arm64] + os: [win32] + '@anthropic-ai/claude-agent-sdk-win32-x64@0.3.205': resolution: {integrity: sha512-kg2kkXyeSoFLruO3Ic2IruLxzBR0xCUtmlJHdWi3SYW7JhAKNJg4fcrdJsWcardmEw23Y2UDGDJbRyxqSVx6wg==} cpu: [x64] os: [win32] + '@anthropic-ai/claude-agent-sdk-win32-x64@0.3.220': + resolution: {integrity: sha512-MuOuXhbr66HlGaWXD2f3w0k2PsvmnbkwcUZ0dAe2poFLdl72GC2dapwwOBefxm9QmoNqk9+jmv/dSKGOVWyvLw==} + cpu: [x64] + os: [win32] + '@anthropic-ai/claude-agent-sdk@0.3.205': resolution: {integrity: sha512-ft6iBw9kXudsusiXNpeybIPBJ07Z3tqp1ROSg5cEJqgA+9i+JJj2sRfQth+QD+lyenbbAU8yPieLxIimvfBhtw==} engines: {node: '>=18.0.0'} @@ -177,6 +217,14 @@ packages: '@modelcontextprotocol/sdk': ^1.29.0 zod: ^4.0.0 + '@anthropic-ai/claude-agent-sdk@0.3.220': + resolution: {integrity: sha512-glc7SdwPkOkLw8oxwLo9PKTdLJGqW/PIR4urWXFoRtX9YllwozsEVc5Tc1+EvLSkfrsxPJqQWqOgpjUOQXf1oA==} + engines: {node: '>=18.0.0'} + peerDependencies: + '@anthropic-ai/sdk': '>=0.93.0' + '@modelcontextprotocol/sdk': ^1.29.0 + zod: ^4.0.0 + '@anthropic-ai/sdk@0.110.0': resolution: {integrity: sha512-hOP4bNYXDFHDxxiEgzlILXrxZIYCDnhe8sry0RDRKD/QnsEpvZcQpablCdm9X/WuD/YgOiSIkkqsL1mLLlTqJw==} hasBin: true @@ -3495,27 +3543,51 @@ snapshots: '@anthropic-ai/claude-agent-sdk-darwin-arm64@0.3.205': optional: true + '@anthropic-ai/claude-agent-sdk-darwin-arm64@0.3.220': + optional: true + '@anthropic-ai/claude-agent-sdk-darwin-x64@0.3.205': optional: true + '@anthropic-ai/claude-agent-sdk-darwin-x64@0.3.220': + optional: true + '@anthropic-ai/claude-agent-sdk-linux-arm64-musl@0.3.205': optional: true + '@anthropic-ai/claude-agent-sdk-linux-arm64-musl@0.3.220': + optional: true + '@anthropic-ai/claude-agent-sdk-linux-arm64@0.3.205': optional: true + '@anthropic-ai/claude-agent-sdk-linux-arm64@0.3.220': + optional: true + '@anthropic-ai/claude-agent-sdk-linux-x64-musl@0.3.205': optional: true + '@anthropic-ai/claude-agent-sdk-linux-x64-musl@0.3.220': + optional: true + '@anthropic-ai/claude-agent-sdk-linux-x64@0.3.205': optional: true + '@anthropic-ai/claude-agent-sdk-linux-x64@0.3.220': + optional: true + '@anthropic-ai/claude-agent-sdk-win32-arm64@0.3.205': optional: true + '@anthropic-ai/claude-agent-sdk-win32-arm64@0.3.220': + optional: true + '@anthropic-ai/claude-agent-sdk-win32-x64@0.3.205': optional: true + '@anthropic-ai/claude-agent-sdk-win32-x64@0.3.220': + optional: true + '@anthropic-ai/claude-agent-sdk@0.3.205(@anthropic-ai/sdk@0.110.0(zod@4.4.3))(@modelcontextprotocol/sdk@1.29.0(zod@4.4.3))(zod@4.4.3)': dependencies: '@anthropic-ai/sdk': 0.110.0(zod@4.4.3) @@ -3531,6 +3603,21 @@ snapshots: '@anthropic-ai/claude-agent-sdk-win32-arm64': 0.3.205 '@anthropic-ai/claude-agent-sdk-win32-x64': 0.3.205 + '@anthropic-ai/claude-agent-sdk@0.3.220(@anthropic-ai/sdk@0.110.0(zod@4.4.3))(@modelcontextprotocol/sdk@1.29.0(zod@4.4.3))(zod@4.4.3)': + dependencies: + '@anthropic-ai/sdk': 0.110.0(zod@4.4.3) + '@modelcontextprotocol/sdk': 1.29.0(zod@4.4.3) + zod: 4.4.3 + optionalDependencies: + '@anthropic-ai/claude-agent-sdk-darwin-arm64': 0.3.220 + '@anthropic-ai/claude-agent-sdk-darwin-x64': 0.3.220 + '@anthropic-ai/claude-agent-sdk-linux-arm64': 0.3.220 + '@anthropic-ai/claude-agent-sdk-linux-arm64-musl': 0.3.220 + '@anthropic-ai/claude-agent-sdk-linux-x64': 0.3.220 + '@anthropic-ai/claude-agent-sdk-linux-x64-musl': 0.3.220 + '@anthropic-ai/claude-agent-sdk-win32-arm64': 0.3.220 + '@anthropic-ai/claude-agent-sdk-win32-x64': 0.3.220 + '@anthropic-ai/sdk@0.110.0(zod@4.4.3)': dependencies: json-schema-to-ts: 3.1.1 diff --git a/workflows/review/README.md b/workflows/review/README.md index a50edd67..765fbc85 100644 --- a/workflows/review/README.md +++ b/workflows/review/README.md @@ -516,7 +516,9 @@ Fable's cyber classifiers can refuse benign security analysis, while `correctness-reviewer` — the default roster's load-bearing recall agent — was moved *onto* Fable 5 for its recall gain. Eval run 30656579898 caught it refusing `incident-auth-bypass` and `adversarial-injection-approve` outright, -at 5,207 tokens (so not a context limit). +at 5,207 tokens (so not a context limit). The roster has since moved to Opus 5, +which carries its own elevated cyber safeguards, so the hazard moved with it +rather than being resolved by the pin change. Refusals are **intermittent**: probe run 30658862532 saw the same Fable pin clear both cases that run 30656579898 blocked. The ordinary retry still cannot @@ -527,7 +529,7 @@ refusing pin to a model with a different refusal profile: | Pinned model | Falls back to | Basis | | --- | --- | --- | | `claude-fable-5` | `claude-opus-4-8` | measured (run 30656579898) | -| `claude-opus-5` | `claude-opus-4-8` | pre-emptive; #294 notes Opus 5 also ships elevated cyber safeguards | +| `claude-opus-5` | `claude-opus-4-8` | pre-emptive; Opus 5 ships elevated cyber safeguards and can also return `stop_reason: "refusal"` | Rules: **one hop**, never back to a model that already refused, and **no fallback for an unlisted pin** — an unmapped model's refusal stands and is @@ -552,19 +554,19 @@ sub-agent models — this table is the human-facing summary: | Role | Model | Effort | Why | | --- | --- | --- | --- | -| orchestrator | `claude-opus-4-8` | high | Owns every GitHub/safe-output decision | +| orchestrator | `claude-opus-5` | high | Owns every GitHub/safe-output decision | | `pattern-triage` | `claude-sonnet-4-6` | medium | Cheap first-pass triage | -| `thread-reconciler` | `claude-opus-4-8` | medium | Reconciliation | -| `correctness-reviewer` | `claude-fable-5` | high | Whole-change reviewer; bug-finding recall is the load-bearing metric | -| `skill-auditor` | `claude-opus-4-8` | high | Whole-change reviewer | -| `holistic` | `claude-opus-4-8` | high | Opt-in whole-change reviewer (`enable` in `ROUTING`) | -| `completeness` | `claude-opus-4-8` | high | Opt-in whole-change reviewer (`enable` in `ROUTING`) | -| `test-adequacy` | `claude-opus-4-8` | high | Opt-in whole-change reviewer (`enable` in `ROUTING`) | -| `conventions` | `claude-opus-4-8` | medium | Opt-in advisory targeted check (`enable` in `ROUTING`) | -| `documentation` | `claude-opus-4-8` | medium | Opt-in advisory targeted check (`enable` in `ROUTING`) | -| `first-principles` | `claude-fable-5` | high | Opt-in advisory-only; reviews the change's justification | -| `claim-validator` | `claude-opus-4-8` | xhigh | Adversarial claim validation; stays Opus (the Fable arm did not improve precision) | -| specialist lenses | `claude-opus-4-8` | high | Opt-in via `lens=` in `ROUTING`; the security & auth lens is xhigh | +| `thread-reconciler` | `claude-opus-5` | medium | Reconciliation | +| `correctness-reviewer` | `claude-opus-5` | high | Whole-change reviewer; bug-finding recall is the load-bearing metric | +| `skill-auditor` | `claude-opus-5` | high | Whole-change reviewer | +| `holistic` | `claude-opus-5` | high | Opt-in whole-change reviewer (`enable` in `ROUTING`) | +| `completeness` | `claude-opus-5` | high | Opt-in whole-change reviewer (`enable` in `ROUTING`) | +| `test-adequacy` | `claude-opus-5` | high | Opt-in whole-change reviewer (`enable` in `ROUTING`) | +| `conventions` | `claude-opus-5` | medium | Opt-in advisory targeted check (`enable` in `ROUTING`) | +| `documentation` | `claude-opus-5` | medium | Opt-in advisory targeted check (`enable` in `ROUTING`) | +| `first-principles` | `claude-opus-5` | high | Opt-in advisory-only; reviews the change's justification | +| `claim-validator` | `claude-opus-5` | xhigh | Adversarial claim validation; stays Opus (the Fable arm did not improve precision) | +| specialist lenses | `claude-opus-5` | high | Opt-in via `lens=` in `ROUTING`; the security & auth lens is xhigh | Only the orchestrator and the default roster (`pattern-triage`, `correctness-reviewer`, `skill-auditor`, `thread-reconciler`, `claim-validator`) diff --git a/workflows/review/lib/model-pricing.test.ts b/workflows/review/lib/model-pricing.test.ts new file mode 100644 index 00000000..46ee6eed --- /dev/null +++ b/workflows/review/lib/model-pricing.test.ts @@ -0,0 +1,78 @@ +/** + * Mechanical pricing gates for the model pins in review.md (#294 review + * feedback: the merge-ordering constraint "do not ship an un-priced pin" + * rested entirely on human memory of a draft PR). + * + * Two hazards, both prose-only until this file: + * + * - The `models.providers` overlay matches per model and an unlisted model + * silently bills at FULL list price (the overlay's own MAINTENANCE note); + * the overlay omitting `claude-sonnet-4-6` shipped exactly that way in an + * earlier draft of #314. + * - On the stable toolchain (gh-aw v0.83.x -> firewall v0.27.42) the + * `providers` block is dropped silently and the api-proxy's credit guard + * rejects a model its curated table does not price with a 400 before the + * request reaches the model (#266). `claude-opus-5` is not in that table, + * so the `default-ai-credits-pricing` fallback is load-bearing for every + * dispatch until a gh-aw release defaults the firewall to v0.27.43+. + * + * DELETE the fallback test (only it) together with the fallback block when + * the toolchain moves; the coverage test is permanent. + */ +import {readFileSync} from "node:fs"; +import {join} from "node:path"; + +import {describe, it, expect} from "vitest"; + +const reviewMd = readFileSync(join(__dirname, "..", "review.md"), "utf8"); + +/** The workflow frontmatter (between the first pair of --- fences). */ +const frontmatter = reviewMd.split(/^---$/m)[1] ?? ""; + +/** + * Every model pin in the file: the engine's indented `model:` line in the + * frontmatter plus each sub-agent's `model:` line in its block frontmatter. + */ +const pins = [ + ...new Set( + [...reviewMd.matchAll(/^\s*model:\s*(claude-[a-z0-9.-]+)\s*$/gm)].map( + (match) => match[1], + ), + ), +]; + +/** + * The models the `providers` overlay prices: bare `claude-*:` mapping keys in + * the frontmatter (only the providers block declares them). + */ +const priced = new Set( + [...frontmatter.matchAll(/^\s+(claude-[a-z0-9.-]+):\s*$/gm)].map( + (match) => match[1], + ), +); + +describe("model pricing coverage (review.md frontmatter)", () => { + it("finds the pins and the overlay (guards the extraction itself)", () => { + // 22 agents plus the engine; a collapse to zero means the regexes + // rotted, not that the roster emptied. + expect(pins.length).toBeGreaterThanOrEqual(2); + expect(priced.size).toBeGreaterThanOrEqual(2); + }); + + it("prices every pinned model in the providers overlay", () => { + const unpriced = pins.filter((pin) => !priced.has(pin)); + // An unlisted model silently bills at full list price; add an entry + // at 50% of its Anthropic list rate (see the MAINTENANCE note). + expect(unpriced).toEqual([]); + }); + + it("keeps the stable-toolchain credit-guard fallback while claude-opus-5 is pinned", () => { + // Firewall v0.27.42's curated table does not price claude-opus-5; + // without `default-ai-credits-pricing` every dispatch 400s on the + // stable toolchain. Delete this test with the fallback block once a + // gh-aw release defaults the firewall to v0.27.43+. + if (pins.includes("claude-opus-5")) { + expect(frontmatter).toContain("default-ai-credits-pricing:"); + } + }); +}); diff --git a/workflows/review/review.md b/workflows/review/review.md index 2d7aa221..93ad2e38 100644 --- a/workflows/review/review.md +++ b/workflows/review/review.md @@ -179,7 +179,7 @@ observability: # job-level timeout-minutes still bounds the run. engine: id: claude - model: claude-opus-4-8 + model: claude-opus-5 env: BASH_DEFAULT_TIMEOUT_MS: "60000" BASH_MAX_TIMEOUT_MS: "1200000" @@ -343,23 +343,60 @@ post-steps: # Do NOT collapse these to a bare `claude-opus-4` prefix: prefix matching would # also capture opus-4-0/4-1, which list at 3x the 4-5+ rate. models: + # claude-opus-5 (the engine and roster pin) is NOT in the firewall + # api-proxy's curated AI-credits pricing table at v0.27.42, the release + # gh-aw v0.83.4 defaults to (that table carries claude-opus-4-5 through 4-8 + # and claude-fable-5, and stops there). The proxy's AI-credits guard rejects + # an un-priced model with a 400 BEFORE the request reaches the model, so + # without this fallback every dispatch fails on the stable toolchain: the + # #266 failure that killed the first-principles dispatch on every run, + # except the whole roster runs the un-priced model rather than two opt-in + # agents. Units differ from the overlay below and the two are not + # interchangeable: `default-ai-credits-pricing` is $/1M tokens and feeds + # the credit guard; `providers` is $/token. Rates here are Anthropic LIST + # (Opus 5 lists at exactly Opus 4.8's price), deliberately not Khan's 50% + # rate: in the only window where this fallback binds (stable gh-aw v0.83.x, + # firewall v0.27.42, which drops the `providers` block silently) every + # other model bills at list from the curated table, so list keeps Opus 5 + # denominated consistently with the rest of the roster. Two caveats: the + # default-pricing path does not bill cache writes ($6.25/M real), so credit + # accounting under-counts that component while it binds; and the fallback + # applies to ANY un-priced model, so a typo'd model id bills at Opus rates + # instead of failing loudly. + # + # REMOVE THIS FALLBACK when a gh-aw release defaults the firewall to + # v0.27.43 or later: that release carries a curated claude-opus-5 entry at + # the same list rates and bills cache writes, and the recompile also makes + # the `providers` overlay below live (the higher-precedence source). Do NOT + # reach for `sandbox.agent.version: v0.27.43` to get there early; a version + # is pinned here only to hold a release BACK, never to move one forward. + # + # MINIMUM COMPILER: gh-aw >= v0.83.0 for `models.default-ai-credits-pricing`. + # $/1M tokens. `input` and `output` are the only rates the schema accepts, + # so the cache rates are the proxy's derivations, not ours. + default-ai-credits-pricing: + input: 5.0 + output: 25.0 providers: anthropic: models: - # Current engine model. + # The refusal-fallback target (the one-hop re-dispatch of #315) and + # an `engine:` override candidate; the engine until the roster moved + # to Opus 5. claude-opus-4-8: cost: input: "2.5e-06" output: "1.25e-05" cache_read: "2.5e-07" cache_write: "3.125e-06" - # Engine models a consumer may select via an `engine:` override. + # Current engine and roster model. claude-opus-5: cost: input: "2.5e-06" output: "1.25e-05" cache_read: "2.5e-07" cache_write: "3.125e-06" + # Engine models a consumer may select via an `engine:` override. claude-sonnet-5: cost: input: "1e-06" @@ -1186,11 +1223,12 @@ return `{"findings": [], "hunts": [...]}` with the hunt states still recorded. --- name: correctness-reviewer description: Classifies each changed file's risk and reviews the diff for correctness defects; returns JSON. -model: claude-fable-5 +model: claude-opus-5 # effort: high — launch default (whole-change reviewer). gh-aw has no per-agent # effort field yet; the per-role model/effort table lives in the README. -# Fable 5: bug-finding recall is this workflow's load-bearing metric, and -# stronger real-defect detection is Fable's headline gain over Opus 4.8. +# Opus 5: bug-finding recall is this workflow's load-bearing metric; the +# 2026-07-20 A/B recall gain that lived on Fable 5 is carried by Opus 5 at +# Opus 4.8's per-token price. See the roster table in the README. --- You are a correctness-focused code reviewer. You have **no GitHub access** — read the diff and file list from disk and return your result as JSON only. @@ -1381,7 +1419,7 @@ before (run 29943085279 carried its one-line fix under `suggested_patch`): --- name: skill-auditor description: Evaluates the diff against the repo's best-practice skills and returns findings as JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: high — launch default (whole-change reviewer). --- You audit a PR diff for best-practice "skill" violations. You have **no GitHub @@ -1545,7 +1583,7 @@ Return ONLY this JSON object (no prose, no code fence): --- name: thread-reconciler description: Decides which of the workflow's earlier review threads the current code has addressed; returns thread ids. -model: claude-opus-4-8 +model: claude-opus-5 # effort: medium — launch default (reconciliation). --- You decide which earlier review threads the current code has resolved. You have **no @@ -1604,7 +1642,7 @@ Return ONLY this JSON object (no prose, no code fence): --- name: claim-validator description: Re-checks each candidate review comment against the actual code and the repo's best-practice skills, and drops or corrects the ones that are wrong; returns JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: xhigh — launch default (claim-validator). Deliberately NOT moved to # Fable 5 with the correctness reviewer: in the 2026-07-20 pooled A/B the # Fable validator did not offset the higher flag rate (noise 43% -> 49%, one @@ -1791,7 +1829,7 @@ Every input `id` must appear exactly once. --- name: holistic description: Reviews the change as a whole — is the overall approach sound and coherent — and returns findings as JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: high — launch default (whole-change reviewer). --- You are the **holistic** reviewer. Your single mandate is to **judge the @@ -1867,7 +1905,7 @@ scenario). If the change hangs together, return {"findings": []}. --- name: completeness description: Checks the change against its stated intent (PR description + linked ticket/doc) and returns findings as JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: high — launch default (whole-change reviewer). --- You are the **completeness** reviewer. Your single mandate is to **check @@ -1938,7 +1976,7 @@ If the change matches its intent, return {"findings": []}. --- name: test-adequacy description: Evaluates whether the changed behavior is adequately tested and returns findings as JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: high — launch default (whole-change reviewer). --- You are the **test-adequacy** reviewer. Your job is to judge whether the **changed @@ -1999,10 +2037,11 @@ If the changed behavior is adequately tested, return {"findings": []}. --- name: first-principles description: A diverse-perspective, advisory-only sanity check on whether the change should exist as written; returns findings as JSON. -model: claude-fable-5 -# effort: high — launch default. Ran on Fable 5 (claude-fable-5) from day one; -# the correctness reviewer joined it after the 2026-07-20 A/B. Advisory-only, -# never blocks. +model: claude-opus-5 +# effort: high — launch default. Ran on Fable 5 (claude-fable-5) from day one, +# partly to be the one non-Opus reviewer; the correctness reviewer joined it +# after the 2026-07-20 A/B, and it moved to Opus 5 with the roster. +# Advisory-only, never blocks. --- You are the **first-principles** reviewer. Your single mandate is to review the **justification for the change, not the change itself**: where `holistic` asks @@ -2072,7 +2111,7 @@ If you have nothing worth raising, return {"findings": []}. --- name: conventions description: Advisory, opt-in check of repo-specific conventions; returns findings as JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: medium — launch default (advisory, opt-in targeted check). --- You are the **conventions** reviewer. You check the change against this repository's @@ -2138,7 +2177,7 @@ not worth flagging). If nothing deviates from repo conventions, return --- name: documentation description: Advisory, opt-in check that code comments and prose docs in the diff document intent rather than restate code; returns findings as JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: medium — launch default (advisory, opt-in targeted check). Sibling of # `conventions`: same shape, same cost profile, different subject matter. --- @@ -2295,7 +2334,7 @@ maintain for nothing). If nothing in the change fails the policy, return --- name: security-auth description: Specialist security & auth lens — reviews touched files for authorization, secrets, injection, and unsafe-deserialization defects; returns structured findings as JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: xhigh — launch default. The security & auth lens is the one specialist # lens pinned to xhigh (per-role table in the README). gh-aw has no # per-agent effort field yet; this annotation and the README table are the authoritative @@ -2383,7 +2422,7 @@ Conventional-Comment `label` is emitted (the orchestrator computes it from --- name: ai-safety-moderation description: Specialist AI safety & moderation lens — reviews AI/generation paths for missing moderation, prompt-injection surfaces, and PII exposure; returns structured findings as JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: high — launch default (specialist lens). --- You are the **AI safety & moderation** specialist lens. You review only AI/model and @@ -2453,7 +2492,7 @@ Conventional-Comment `label` is emitted (the orchestrator computes it from --- name: mass-comms-coppa description: Specialist mass-comms & COPPA lens — reviews bulk-communication paths for audience/consent/age-gating defects; returns structured findings as JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: high — launch default (specialist lens). --- You are the **mass-comms & COPPA** specialist lens. You review only bulk-communication @@ -2521,7 +2560,7 @@ Conventional-Comment `label` is emitted (the orchestrator computes it from --- name: caching-resource description: Specialist caching & resource lens — reviews caching and resource-management paths for key-scoping, invalidation, and exhaustion defects; returns structured findings as JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: high — launch default (specialist lens). --- You are the **caching & resource** specialist lens. You review only caching and @@ -2596,7 +2635,7 @@ Conventional-Comment `label` is emitted (the orchestrator computes it from --- name: data-migrations description: Specialist data & migrations lens — reviews schema/migration/backfill changes for compatibility and safety defects; returns structured findings as JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: high — launch default (specialist lens). --- You are the **data & migrations** specialist lens. You review only schema changes, @@ -2666,7 +2705,7 @@ Conventional-Comment `label` is emitted (the orchestrator computes it from --- name: concurrency-async description: Specialist concurrency & async lens — reviews concurrent/async code for races, unawaited work, and idempotency defects; returns structured findings as JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: high — launch default (specialist lens). --- You are the **concurrency & async** specialist lens. You review only concurrent and @@ -2735,7 +2774,7 @@ Conventional-Comment `label` is emitted (the orchestrator computes it from --- name: api-federation-compat description: Specialist API & federation compatibility lens — reviews public API and GraphQL/federation changes for breaking-change defects; returns structured findings as JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: high — launch default (specialist lens). --- You are the **API & federation compatibility** specialist lens. You review only changes to @@ -2804,7 +2843,7 @@ Conventional-Comment `label` is emitted (the orchestrator computes it from --- name: cross-deploy-serialization description: Specialist cross-deploy serialization lens — reviews persisted/queued/cached serialized shapes for rolling-deploy compatibility defects; returns structured findings as JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: high — launch default (specialist lens). --- You are the **cross-deploy serialization** specialist lens. You review only changes to @@ -2877,7 +2916,7 @@ Conventional-Comment `label` is emitted (the orchestrator computes it from --- name: deploy-infra-config description: Specialist deploy & infra config lens — reviews deployment, infra-as-code, and config/flag changes for rollout-safety defects; returns structured findings as JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: high — launch default (specialist lens). --- You are the **deploy & infra config** specialist lens. You review only deployment @@ -2948,7 +2987,7 @@ Conventional-Comment `label` is emitted (the orchestrator computes it from --- name: money-payments description: Specialist money & payments lens — reviews monetary and payment code for precision, idempotency, and currency defects; returns structured findings as JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: high — launch default (specialist lens). --- You are the **money & payments** specialist lens. You review only monetary computation and @@ -3017,7 +3056,7 @@ Conventional-Comment `label` is emitted (the orchestrator computes it from --- name: content-i18n description: Specialist content & i18n lens — reviews user-facing content for localization and internationalization defects; returns structured findings as JSON. -model: claude-opus-4-8 +model: claude-opus-5 # effort: high — launch default (specialist lens). --- You are the **content & i18n** specialist lens. You review only user-facing content for