Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
80 commits
Select commit Hold shift + click to select a range
2d746ac
[jwbron/live-eval-corpus] review: live-enabled corpus format and ten …
jwbron Jul 9, 2026
6727eb6
[jwbron/live-eval-producer-staging] review: live-producer prompt extr…
jwbron Jul 9, 2026
5c00cf3
[jwbron/live-eval-producer-staging] review: the live producer and SDK…
jwbron Jul 9, 2026
52073ef
[jwbron/live-eval-ab-runner] review: the live A/B runner (phase 3)
jwbron Jul 9, 2026
93770e6
[jwbron/live-eval-ab-ci] review: per-PR live A/B workflow (phase 4)
jwbron Jul 9, 2026
30e7527
[jwbron/live-eval-corpus] review: exclude eval-corpus trees from lint…
jwbron Jul 9, 2026
19f06cb
[jwbron/live-eval-producer-staging] Merge branch 'jwbron/live-eval-co…
jwbron Jul 9, 2026
d1b5288
[jwbron/live-eval-ab-runner] Merge branch 'jwbron/live-eval-producer-…
jwbron Jul 9, 2026
254dfc4
[jwbron/live-eval-ab-ci] Merge branch 'jwbron/live-eval-ab-runner' in…
jwbron Jul 9, 2026
917cc11
[jwbron/live-eval-producer-staging] review: namespace live finding id…
jwbron Jul 9, 2026
0ee295a
[jwbron/live-eval-ab-runner] Merge branch 'jwbron/live-eval-producer-…
jwbron Jul 9, 2026
58cd4ed
[jwbron/live-eval-ab-runner] review: judge failures degrade the A/B r…
jwbron Jul 9, 2026
b19b08d
[jwbron/live-eval-ab-ci] Merge branch 'jwbron/live-eval-ab-runner' in…
jwbron Jul 9, 2026
bced291
[jwbron/review-trial-skill] review: add the review-trial skill (live …
jwbron Jul 9, 2026
da1115c
[tmp-refresh] Merge remote-tracking branch 'origin/main' into tmp-ref…
jwbron Jul 9, 2026
a6883be
[tmp-refresh] Merge remote-tracking branch 'origin/jwbron/live-eval-c…
jwbron Jul 9, 2026
0d02672
[tmp-refresh] Merge remote-tracking branch 'origin/jwbron/live-eval-p…
jwbron Jul 9, 2026
93d8dec
[tmp-refresh] Merge remote-tracking branch 'origin/jwbron/live-eval-a…
jwbron Jul 9, 2026
e3b34eb
[tmp-refresh] Merge remote-tracking branch 'origin/jwbron/live-eval-a…
jwbron Jul 9, 2026
491a983
[jwbron/review-rereview-accountability] review: re-review accountabil…
jwbron Jul 9, 2026
7b5318c
[jwbron/review-out-of-lane] review: hand off out-of-lane observations…
jwbron Jul 9, 2026
5dd182b
[jwbron/review-out-artifact-upload] review: fix the out/ artifact upl…
jwbron Jul 9, 2026
92bffa2
[jwbron/review-rereview-accountability] review: prettier-format the r…
jwbron Jul 9, 2026
2be8ede
[jwbron/review-out-of-lane] Merge branch 'jwbron/review-rereview-acco…
jwbron Jul 9, 2026
3a8fc5c
[jwbron/review-out-of-lane] review: prettier-format finding-schema
jwbron Jul 9, 2026
813767c
[jwbron/review-dispatch-tax] review: dedupe lens discipline snippets …
jwbron Jul 9, 2026
2812679
[jwbron/live-eval-corpus] review: route the specialist lens on each l…
jwbron Jul 9, 2026
2ce35a0
[jwbron/live-eval-producer-staging] Merge branch 'jwbron/live-eval-co…
jwbron Jul 9, 2026
c0fece2
[jwbron/live-eval-ab-runner] Merge branch 'jwbron/live-eval-producer-…
jwbron Jul 9, 2026
e02ac40
[jwbron/live-eval-ab-runner] review: carry agent-failure reasons into…
jwbron Jul 9, 2026
d225bd4
[jwbron/live-eval-ab-ci] Merge branch 'jwbron/live-eval-ab-runner' in…
jwbron Jul 9, 2026
de239ee
[jwbron/rereview-mode-dial] review: the re-review mode dial: ROUTING …
jwbron Jul 9, 2026
1329297
[jwbron/rereview-mode-dial] review: expose keptBlockingCount from the…
jwbron Jul 9, 2026
45c6f6e
[jwbron/rereview-mode-dial] review: wire the re-review mode dial into…
jwbron Jul 9, 2026
976b925
[jwbron/rereview-mode-dial] review: prettier-format the mode-dial fil…
jwbron Jul 9, 2026
cbc838d
[jwbron/live-eval-ab-runner] review: close every A/B report with a pe…
jwbron Jul 9, 2026
63097f1
[jwbron/live-eval-corpus] Merge remote-tracking branch 'origin/jwbron…
jwbron Jul 9, 2026
9012508
[jwbron/live-eval-producer-staging] Merge remote-tracking branch 'ori…
jwbron Jul 9, 2026
25133b4
[jwbron/live-eval-producer-staging] Merge branch 'jwbron/live-eval-co…
jwbron Jul 9, 2026
0a3d212
[jwbron/live-eval-ab-runner] Merge remote-tracking branch 'origin/jwb…
jwbron Jul 9, 2026
a547972
[jwbron/live-eval-ab-runner] Merge branch 'jwbron/live-eval-producer-…
jwbron Jul 9, 2026
391151b
[jwbron/live-eval-ab-ci] Merge remote-tracking branch 'origin/jwbron/…
jwbron Jul 9, 2026
d2c4c70
[jwbron/live-eval-ab-ci] Merge branch 'jwbron/live-eval-ab-runner' in…
jwbron Jul 9, 2026
082f580
[jwbron/rereview-live-corpus] review: re-review coverage for the eval…
jwbron Jul 9, 2026
48cdc39
[jwbron/rereview-live-corpus] review: the re-review mode sweep and th…
jwbron Jul 9, 2026
284fa40
[jwbron/rereview-live-corpus] review: dispatchable CI home for the re…
jwbron Jul 9, 2026
7f9ae36
[jwbron/rereview-live-corpus] review: manual trigger surface for ever…
jwbron Jul 9, 2026
996766f
[jwbron/live-eval-ab-runner] review: identity short-circuit, gate-fli…
jwbron Jul 10, 2026
fb81be8
[jwbron/live-eval-ab-runner] review: judge economics (Haiku pin, retr…
jwbron Jul 10, 2026
a659be8
[jwbron/live-eval-ab-ci] review: document why the baseline is the bas…
jwbron Jul 10, 2026
b7c3786
[jwbron/live-eval-ab-ci] Merge branch 'jwbron/live-eval-ab-runner' in…
jwbron Jul 10, 2026
541e413
[jwbron/review-trial-skill] Merge branch 'jwbron/live-eval-ab-ci' int…
jwbron Jul 10, 2026
fd42efd
[jwbron/review-out-artifact-upload] Merge branch 'jwbron/review-trial…
jwbron Jul 10, 2026
6d3459a
[jwbron/review-rereview-accountability] Merge branch 'jwbron/review-o…
jwbron Jul 10, 2026
1438a67
[jwbron/review-out-of-lane] Merge branch 'jwbron/review-rereview-acco…
jwbron Jul 10, 2026
98157c2
[jwbron/review-dispatch-tax] Merge branch 'jwbron/review-out-of-lane'…
jwbron Jul 10, 2026
bbb624c
[jwbron/rereview-mode-dial] Merge branch 'jwbron/review-dispatch-tax'…
jwbron Jul 10, 2026
4d1ac62
[jwbron/rereview-live-corpus] Merge branch 'jwbron/rereview-mode-dial…
jwbron Jul 10, 2026
193ae69
[jwbron/live-eval-ab-runner] review: type the judge response via the …
jwbron Jul 10, 2026
7a2065c
[jwbron/live-eval-ab-ci] Merge branch 'jwbron/live-eval-ab-runner' in…
jwbron Jul 10, 2026
00ce9d4
[jwbron/review-trial-skill] Merge branch 'jwbron/live-eval-ab-ci' int…
jwbron Jul 10, 2026
4bd445c
[jwbron/review-out-artifact-upload] Merge branch 'jwbron/review-trial…
jwbron Jul 10, 2026
da1e1db
[jwbron/review-rereview-accountability] Merge branch 'jwbron/review-o…
jwbron Jul 10, 2026
1bc9e08
[jwbron/review-out-of-lane] Merge branch 'jwbron/review-rereview-acco…
jwbron Jul 10, 2026
0123a92
[jwbron/review-dispatch-tax] Merge branch 'jwbron/review-out-of-lane'…
jwbron Jul 10, 2026
64f8115
[jwbron/rereview-mode-dial] Merge branch 'jwbron/review-dispatch-tax'…
jwbron Jul 10, 2026
5631477
[jwbron/rereview-live-corpus] Merge branch 'jwbron/rereview-mode-dial…
jwbron Jul 10, 2026
80da666
Merge branch 'main' into jwbron/live-eval-corpus
jwbron Jul 10, 2026
82bc5ed
Merge branch 'jwbron/live-eval-corpus' into jwbron/live-eval-producer…
jwbron Jul 10, 2026
7f2fc3a
Merge branch 'jwbron/live-eval-producer-staging' into jwbron/live-eva…
jwbron Jul 10, 2026
9cc1a89
Merge branch 'jwbron/live-eval-ab-runner' into jwbron/live-eval-ab-ci
jwbron Jul 10, 2026
da93344
Merge branch 'jwbron/live-eval-ab-ci' into jwbron/review-trial-skill
jwbron Jul 10, 2026
aa5a246
Merge branch 'jwbron/review-trial-skill' into jwbron/review-out-artif…
jwbron Jul 10, 2026
248cdf8
Merge branch 'jwbron/review-out-artifact-upload' into jwbron/review-r…
jwbron Jul 10, 2026
51691de
Merge branch 'jwbron/review-rereview-accountability' into jwbron/revi…
jwbron Jul 10, 2026
a752c5c
Merge branch 'jwbron/review-out-of-lane' into jwbron/review-dispatch-tax
jwbron Jul 10, 2026
73b55f4
Merge branch 'jwbron/review-dispatch-tax' into jwbron/rereview-mode-dial
jwbron Jul 10, 2026
9d83ddf
Merge branch 'jwbron/rereview-mode-dial' into jwbron/rereview-live-co…
jwbron Jul 10, 2026
68c7f83
[landtmp] Merge remote-tracking branch 'origin/main' into landtmp
jwbron Jul 10, 2026
2e7c3ac
[jwbron/rereview-live-corpus] Merge remote-tracking branch 'origin/ma…
jwbron Jul 13, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions .changeset/review-rereview-live-corpus.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
---
"review": minor
---

Re-review coverage for the eval corpus: a live-enabled case may now carry a `live.rereview` block that makes it an open-PR snapshot instead of a first review, staging the prior review's unresolved threads (author replies included), a stamped prior review derived from `priorDiff` (the same hidden-comment mechanics production reads), and the mode dial's depth plan. The live producer dispatches the `thread-reconciler` on such cases at every depth and sizes the finder roster by the plan (`scoped` keeps the roster over the scoped diff, `flip-gated` keeps the correctness pass, `fast` reconciles only); `eval/rereview-match.ts` scores thread-resolution accuracy per ground truth, the flip-gate input, and duplicate comments on kept threads, and the A/B runner prices a mode via `--re-review-mode` on the candidate arm. The deterministic replay learns the same rules: `computeVerdict` gains `keptBlockingCount` (the flip rule as code: an approval is floored at REQUEST_CHANGES while a prior blocking thread stays open) and re-review cases render the accountability section into the planned body. Ships with the `golden-retention-lifecycle-1/2/3` chain, a sanitized port of a measured seeded live trial: planted bugs, a bad partial fix with added bugs (kept threads plus fresh defects), then the full fix that must flip to APPROVE. A mode sweep (`eval/rereview-sweep.ts`) prices every dial setting in one command: it runs the working tree's reviewer over the rereview cases at each mode and reports recall, thread resolution, flip-gate correctness, duplicates, and dollars per mode, with the executed depth per case (the tripwire may override the dial). `golden-retention-fix-push` is the under-threshold pricing case: a one-hunk fix push (unreviewed share 0.2) whose fix resolves the prior blocking thread but plants a fresh defect inside the new hunk, so scoped and flip-gated must catch it and fast definitionally cannot. The sweep is dispatchable in CI (`review-rereview-sweep.yml`, manual only since it spends model dollars); the table lands in the job summary and the JSON report in the run artifact. Every model-spending eval now has the same manual trigger surface: PR labels (`rereview-sweep`, `live-judge`, `full-eval` now firing on the labeling event itself) plus `workflow_dispatch`; and the A/B short-circuits before spending when `review.md` is byte-identical in both arms, posting a "no reviewable delta" verdict and running nothing (`--force-arms` keeps deliberate wobble controls possible).
9 changes: 7 additions & 2 deletions .github/workflows/review-eval-ab.yml
Original file line number Diff line number Diff line change
Expand Up @@ -30,7 +30,10 @@ name: Review Eval A/B

on:
pull_request:
types: [opened, synchronize, reopened, ready_for_review]
# `labeled` is here so adding `full-eval` triggers the lifted run
# immediately instead of waiting for the next push; the job-level guard
# below ignores every other label.
types: [opened, synchronize, reopened, ready_for_review, labeled]
paths:
- "workflows/review/**"
workflow_dispatch:
Expand Down Expand Up @@ -67,7 +70,9 @@ jobs:
github.event_name == 'workflow_dispatch' ||
(github.event.pull_request.draft == false &&
github.head_ref != 'changeset-release/main' &&
!contains(github.event.pull_request.labels.*.name, 'skip-live-eval'))
!contains(github.event.pull_request.labels.*.name, 'skip-live-eval') &&
(github.event.action != 'labeled' ||
github.event.label.name == 'full-eval'))
steps:
- uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5
with:
Expand Down
10 changes: 10 additions & 0 deletions .github/workflows/review-eval-full.yml
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,11 @@
# the committed script `workflows/review/eval/live-judge.ts` (lint/typecheck/
# test-covered like any other source file); it appends the metrics/gates/judge
# report to the job summary so results are visible without opening the log.
#
# Manual triggers: workflow_dispatch against any branch, or the `live-judge`
# label on a PR (runs the judged pass against the PR head; useful when a PR
# changes the corpus or the judge itself). Remove and re-add the label to
# re-run after a push.

name: Review Eval Live Judge

Expand All @@ -20,11 +25,16 @@ on:
# Weekly, Monday 09:00 UTC.
- cron: "0 9 * * 1"
workflow_dispatch: {}
pull_request:
types: [labeled]

jobs:
live-judge:
name: Live judge over the full corpus
runs-on: ubuntu-latest
if: >-
github.event_name != 'pull_request' ||
github.event.label.name == 'live-judge'
steps:
- uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5
- uses: ./actions/shared-node-cache
Expand Down
105 changes: 105 additions & 0 deletions .github/workflows/review-rereview-sweep.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,105 @@
# Review eval — the re-review mode sweep, on demand.
#
# Runs the branch's reviewer over the corpus's open-PR (rereview) cases at
# each requested dial setting and prices the dial in one table: recall,
# thread resolution, flip-gate correctness, duplicates, and dollars per mode,
# with the EXECUTED depth per case (the tripwire may override the dial).
#
# Sweeps dispatch real model calls (ballpark: rereview cases x modes x a few
# agents, low tens of dollars at the default budget), so this never runs
# automatically: trigger it by adding the `rereview-sweep` label to a PR (the
# table lands in a sticky PR comment, the job summary, and the run artifact),
# or dispatch it manually against any branch. Remove and re-add the label to
# re-run after a push.

name: Review Re-review Mode Sweep

on:
pull_request:
types: [labeled]
workflow_dispatch:
inputs:
modes:
description: "Comma-separated dial settings to sweep"
required: false
default: "full,scoped,flip-gated,fast"
cases:
description: "Comma-separated rereview case ids (default: all)"
required: false
default: ""
max_usd:
description: "Hard budget across the sweep"
required: false
default: "30"

permissions:
contents: read
pull-requests: write

concurrency:
group: review-rereview-sweep-${{ github.event.pull_request.number || github.run_id }}
cancel-in-progress: true

jobs:
rereview-sweep:
name: Sweep the re-review dial over the rereview corpus
runs-on: ubuntu-latest
if: >-
github.event_name == 'workflow_dispatch' ||
github.event.label.name == 'rereview-sweep'
steps:
- uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5
- uses: ./actions/shared-node-cache
- name: Run the sweep
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
MODES: ${{ inputs.modes || 'full,scoped,flip-gated,fast' }}
CASES: ${{ inputs.cases || '' }}
MAX_USD: ${{ inputs.max_usd || '30' }}
run: |
if [ -z "$ANTHROPIC_API_KEY" ]; then
echo "ANTHROPIC_API_KEY secret not configured; skipping the sweep." >&2
exit 0
fi
ARGS="--modes $MODES --max-usd $MAX_USD --out out/rereview-sweep.json"
if [ -n "$CASES" ]; then
ARGS="$ARGS --cases $CASES"
fi
pnpm dlx tsx workflows/review/eval/rereview-sweep.ts $ARGS \
| tee sweep-report.md | tee -a "$GITHUB_STEP_SUMMARY"
- name: Upload the JSON report
if: always()
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: rereview-sweep-report
path: out/rereview-sweep.json
if-no-files-found: ignore
- name: Post the sticky PR comment
if: always() && github.event_name == 'pull_request'
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
with:
script: |
const fs = require("fs");
const path = "sweep-report.md";
if (!fs.existsSync(path)) {
core.info("no sweep report; nothing to post");
return;
}
const marker = "<!-- review-rereview-sweep -->";
const body = `${marker}\n${fs.readFileSync(path, "utf8")}`;
const {owner, repo} = context.repo;
const issue_number = context.payload.pull_request.number;
const comments = await github.paginate(
github.rest.issues.listComments,
{owner, repo, issue_number, per_page: 100},
);
const prior = comments.find((c) => c.body?.startsWith(marker));
if (prior) {
await github.rest.issues.updateComment({
owner, repo, comment_id: prior.id, body,
});
} else {
await github.rest.issues.createComment({
owner, repo, issue_number, body,
});
}
45 changes: 45 additions & 0 deletions workflows/review/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -235,6 +235,51 @@ counters aggregate `costByRereviewDepth`). Lifecycle-class changes like this
one are trialed with the seeded-defect skill
(`.claude/skills/review-trial/SKILL.md`).

Re-review behavior is evaluated at three layers, cheapest first:

- **Depth decisions** (`eval/lifecycle/`, deterministic): push sequences
replayed through the tripwire and depth logic alone; the adversarial
rewrite-after-approval and sparse-PR-then-payload cases live here.
- **Open-PR corpus cases** (`live.rereview` in a live-enabled case): a case
that is a mid-review snapshot, not a first review. It stages the prior
review's threads (with author replies), a stamped prior review derived from
`priorDiff`, and the depth plan; the live producer then dispatches the
reconciler (at every depth) plus the depth-sized finder roster, and
`eval/rereview-match.ts` scores thread-resolution accuracy against
per-thread ground truth, the flip-gate input (kept blocking count), and
duplicate comments on kept threads. The A/B runner prices a mode with
`--re-review-mode` (candidate arm only; baseline stays `full`), and the
deterministic replay of the same cases exercises the kept-blocking verdict
floor and the accountability section. The
`golden-retention-lifecycle-1/2/3` chain is the template: planted bugs, a
bad partial fix with added bugs, then the full fix.
- **Live trials** (the review-trial skill): the same lifecycle against the
real workflow on real PRs, reserved for architecture-class changes.

To price every dial setting in one command, `eval/rereview-sweep.ts` runs the
working tree's reviewer over the rereview cases at each mode
(`--modes full,scoped,flip-gated,fast` by default) and reports recall, thread
resolution, flip-gate correctness, duplicates, and dollars per mode. It
dispatches real model calls, so it never runs automatically: add the
`rereview-sweep` label to a PR (the table lands in a sticky PR comment, the
job summary, and the run artifact), dispatch the `Review Re-review Mode
Sweep` workflow against any branch, or run the CLI locally with
`ANTHROPIC_API_KEY` set.

Every model-spending eval has the same manual trigger surface: a PR label
(`full-eval` lifts the A/B to the whole live corpus and now triggers on the
labeling itself, `rereview-sweep` runs the dial sweep, `live-judge` runs the
judged corpus pass, `skip-live-eval` opts a PR out) plus `workflow_dispatch`
for off-PR runs. The A/B also short-circuits before spending: byte-identical
`review.md` in both arms posts a "no reviewable delta" verdict and runs
nothing (`--force-arms` bypasses this for deliberate wobble controls). The mode
is a run parameter, so no special case format exists; three realities the
sweep reports instead: the tripwire can override the dial (each row shows the
EXECUTED depth), pricing the cheap paths needs at least one under-threshold
case (`golden-retention-fix-push`, a one-hunk fix push whose fix plants a
fresh defect inside the new hunk), and `fast` has definitionally zero
fresh-defect recall (its cost, shown as recall against dollars).

### Models and effort per role

Each sub-agent pins its model in its own definition inside `review.md` (with a
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,93 @@
{
"id": "golden-retention-fix-push",
"tags": [
"golden",
"rereview-lifecycle",
"live"
],
"category": "golden",
"description": "The under-tripwire re-review pricing case: an open PR touched five regions of quota.ts and the prior full review blocked one (an off-by-one in the remaining-quota math). The fix push rewrites only that region (unreviewed share 0.2, below the tripwire threshold), so scoped/flip-gated/fast actually execute their reduced paths. The fix resolves the thread but plants a fresh defect in the same hunk: the remaining count is cached under a shared key with no user id, leaking quota across users. scoped and flip-gated stage exactly that hunk and must catch it; fast dispatches no finder and cannot, which is the recall cost the mode sweep prices.",
"changedFiles": [
{
"path": "src/notes/quota.ts",
"status": "modified"
}
],
"findings": [
{
"source": "correctness",
"finding": {
"schema_version": 2,
"id": "quota-cache-shared-key",
"lens": "correctness",
"anchor": {
"type": "line",
"path": "src/notes/quota.ts",
"line": 26,
"side": "RIGHT"
},
"severity": "blocking",
"confidence": 0.9,
"evidence_trace": [
"src/notes/quota.ts: cache.set(\"quota-remaining\", remaining) keys on a constant",
"src/notes/cache.ts: the cache is process-wide, not per user"
],
"failure_scenario": "Two users hit remainingQuota in the same process; the second write overwrites the first under the shared \"quota-remaining\" key, so any reader of the cached value sees another user's remaining quota.",
"producing_hunt": "correctness:shared-cache-key",
"model_authored_prose": "The cached remaining count is stored under a constant key with no user id, so concurrent users overwrite each other and any cached read leaks one user's quota to another. Key the entry by userId (or drop the cache)."
}
}
],
"validation": [
{
"id": "quota-cache-shared-key",
"verification": "confirmed"
}
],
"expected": {
"verdict": "REQUEST_CHANGES",
"postedCommentCount": 1,
"mustCatch": [
"quota-cache-shared-key"
]
},
"diff": "diff --git a/src/notes/quota.ts b/src/notes/quota.ts\n--- a/src/notes/quota.ts\n+++ b/src/notes/quota.ts\n@@ -1,15 +1,20 @@\n import type {Db} from \"./db\";\n import {cache} from \"./cache\";\n \n-export const QUOTA_LIMIT = 500;\n+/** Raised for the notes launch. */\n+export const QUOTA_LIMIT = 800;\n \n export const usedQuota = async (db: Db, userId: string): Promise<number> => {\n const notes = await db.query(\"Note\", {userId, pageSize: \"all\"});\n- return notes.length;\n+ // Drafts do not count against quota.\n+ return notes.filter((note) => !note.content.startsWith(\"draft:\")).length;\n };\n \n export const quotaHeaders = (remaining: number): Record<string, string> => {\n- return {\"x-quota-remaining\": String(remaining)};\n+ return {\n+ \"x-quota-remaining\": String(remaining),\n+ \"x-quota-limit\": String(QUOTA_LIMIT),\n+ };\n };\n \n export const remainingQuota = async (\n@@ -17,14 +22,17 @@\n userId: string,\n ): Promise<number> => {\n const used = await usedQuota(db, userId);\n- return QUOTA_LIMIT - used;\n+ const remaining = Math.max(0, QUOTA_LIMIT - used);\n+ cache.set(\"quota-remaining\", remaining);\n+ return remaining;\n };\n \n export const quotaExceeded = async (\n db: Db,\n userId: string,\n ): Promise<boolean> => {\n- return (await remainingQuota(db, userId)) <= 0;\n+ // Exceeded only when nothing remains; equality still allows a save.\n+ return (await remainingQuota(db, userId)) < 0;\n };\n \n export const quotaSummary = async (\n@@ -32,5 +40,5 @@\n userId: string,\n ): Promise<string> => {\n const remaining = await remainingQuota(db, userId);\n- return `quota: ${remaining} of ${QUOTA_LIMIT} remaining`;\n+ return `quota: ${remaining} of ${QUOTA_LIMIT} remaining (drafts excluded)`;\n };\n",
"live": {
"prContext": {
"title": "notes: raise the quota limit and exclude drafts from it",
"description": "Raises the per-user quota and excludes drafts from the count. Fix push: addresses the off-by-one from review.",
"author": "dev-notes",
"baseBranch": "main"
},
"mustCatchSpecs": [
{
"key": "quota-cache-shared-key",
"path": "src/notes/quota.ts",
"mechanism": [
"shared (cache )?key|constant key|no user ?id",
"quota-remaining",
"another user|cross.user|leak"
]
}
],
"rereview": {
"priorDiff": "diff --git a/src/notes/quota.ts b/src/notes/quota.ts\n--- a/src/notes/quota.ts\n+++ b/src/notes/quota.ts\n@@ -1,15 +1,20 @@\n import type {Db} from \"./db\";\n import {cache} from \"./cache\";\n \n-export const QUOTA_LIMIT = 500;\n+/** Raised for the notes launch. */\n+export const QUOTA_LIMIT = 800;\n \n export const usedQuota = async (db: Db, userId: string): Promise<number> => {\n const notes = await db.query(\"Note\", {userId, pageSize: \"all\"});\n- return notes.length;\n+ // Drafts do not count against quota.\n+ return notes.filter((note) => !note.content.startsWith(\"draft:\")).length;\n };\n \n export const quotaHeaders = (remaining: number): Record<string, string> => {\n- return {\"x-quota-remaining\": String(remaining)};\n+ return {\n+ \"x-quota-remaining\": String(remaining),\n+ \"x-quota-limit\": String(QUOTA_LIMIT),\n+ };\n };\n \n export const remainingQuota = async (\n@@ -17,14 +22,15 @@\n userId: string,\n ): Promise<number> => {\n const used = await usedQuota(db, userId);\n- return QUOTA_LIMIT - used;\n+ return QUOTA_LIMIT - used - 1;\n };\n \n export const quotaExceeded = async (\n db: Db,\n userId: string,\n ): Promise<boolean> => {\n- return (await remainingQuota(db, userId)) <= 0;\n+ // Exceeded only when nothing remains; equality still allows a save.\n+ return (await remainingQuota(db, userId)) < 0;\n };\n \n export const quotaSummary = async (\n@@ -32,5 +38,5 @@\n userId: string,\n ): Promise<string> => {\n const remaining = await remainingQuota(db, userId);\n- return `quota: ${remaining} of ${QUOTA_LIMIT} remaining`;\n+ return `quota: ${remaining} of ${QUOTA_LIMIT} remaining (drafts excluded)`;\n };\n",
"priorVerdict": "REQUEST_CHANGES",
"priorDepth": "full",
"priorThreads": [
{
"key": "quota-off-by-one",
"path": "src/notes/quota.ts",
"line": 25,
"body": "**issue (blocking):** Off-by-one: `QUOTA_LIMIT - used - 1` reports one fewer remaining slot than the user has, so the last legitimate save is rejected at the boundary.",
"authorReply": "Fixed: remaining is now Math.max(0, QUOTA_LIMIT - used).",
"expect": "resolve",
"mechanism": [
"off.by.one",
"QUOTA_LIMIT - used"
]
}
]
}
}
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,9 @@
/** Tiny process-local cache shared by the notes service. */
const store = new Map<string, unknown>();

export const cache = {
set: (key: string, value: unknown): void => {
store.set(key, value);
},
get: (key: string): unknown => store.get(key),
};
Loading
Loading