diff --git a/.github/scripts/__tests__/keepalive-reporter-applicability.test.js b/.github/scripts/__tests__/keepalive-reporter-applicability.test.js index dc3b0f734..476802b43 100644 --- a/.github/scripts/__tests__/keepalive-reporter-applicability.test.js +++ b/.github/scripts/__tests__/keepalive-reporter-applicability.test.js @@ -393,6 +393,32 @@ test('a later replay wake recovers a dropped middle reporter from the current re assert.equal(result.results[0].status, 'released'); }); +test('replay is a no-op when an ordinary PR has no authority ledger', async () => { + const result = await replayReporterAuthority({ + github: { request: async () => { throw new Error('no run lookup expected'); } }, + context: { repo: { owner: 'stranske', repo: 'repo' } }, + prNumber: 42, + readAuthority: async (_request, repository, number, options) => { + assert.equal(repository, 'stranske/repo'); + assert.equal(number, 42); + assert.deepEqual(options, { allowMissing: true }); + return null; + }, + makeRequest: () => 'request', + }); + assert.deepEqual(result, { prNumber: 42, results: [] }); +}); + +test('replay still fails closed when an authority ledger read fails', async () => { + await assert.rejects(replayReporterAuthority({ + github: {}, + context: { repo: { owner: 'stranske', repo: 'repo' } }, + prNumber: 42, + readAuthority: async () => { throw new Error('ledger unavailable'); }, + makeRequest: () => 'request', + }), /ledger unavailable/); +}); + test('replay fails closed when exact-attempt worker evidence is unknown', async () => { const ownerAttempt = 'stranske/repo:222:3'; await assert.rejects(replayReporterAuthority({ diff --git a/.github/scripts/keepalive_reporter_applicability.js b/.github/scripts/keepalive_reporter_applicability.js index cc99c7746..73909455b 100644 --- a/.github/scripts/keepalive_reporter_applicability.js +++ b/.github/scripts/keepalive_reporter_applicability.js @@ -168,7 +168,11 @@ async function replayReporterAuthority({ const results = []; const seen = new Set(); for (let pass = 0; pass < maxPasses; pass += 1) { - const { state } = await readAuthority(request, repository, number); + const authority = await readAuthority(request, repository, number, { allowMissing: true }); + // Most keepalive PRs never enter the challenge path and have no ledger. + // Only a confirmed 404 is an empty replay; all other read failures remain fatal. + if (authority === null) break; + const { state } = authority; const attempts = [state.receipt, state.released_receipt, state.recovered_receipt] .map((receipt) => parseOwnerAttempt(repository, receipt?.owner_attempt)) .filter(Boolean) diff --git a/.github/sync-manifest.yml b/.github/sync-manifest.yml index d9b59785f..070ccde30 100644 --- a/.github/sync-manifest.yml +++ b/.github/sync-manifest.yml @@ -452,7 +452,7 @@ scripts: description: "Attempt-bound worker job evidence from the originating registry revision for keepalive authority recovery" - source: .github/scripts/keepalive_reporter_applicability.js - description: "Fail-closed exact-run classification for unassociated keepalive dispatches" + description: "Fail-closed exact-run classification and no-ledger-safe reporter replay for keepalive dispatches" - source: .github/scripts/token_load_balancer.js description: "Dynamic token load balancer for API rate limit management" diff --git a/docs/keepalive/GoalsAndPlumbing.md b/docs/keepalive/GoalsAndPlumbing.md index ea9dd0a0f..d34b99b43 100644 --- a/docs/keepalive/GoalsAndPlumbing.md +++ b/docs/keepalive/GoalsAndPlumbing.md @@ -125,7 +125,7 @@ PR head and labels live outside the authority ledger, so their reads and ledger After a failed or cancelled originating run, the workflow-run reporter reads that run's exact job attempt and the agent registry from that run's exact head commit. A worker step that actually entered is treated as started; a fully terminal skipped worker path is positively not-started; incomplete jobs, API errors, missing job evidence, or unavailable originating registry are unknown. Unknown evidence fails the reporter before summary or fingerprint persistence so a retry remains possible. Only the not-started case may reconcile the matching receipt and owner attempt independently of summary `running` state or a fresh authority-failure classification. A prepared receipt can then be released and an unconfirmed consumed receipt can be reopened after a same-head PR read; confirmed receipts remain spent. A retry of an already reopened receipt revalidates the immutable attempt index, exact PR head, soft attention label, and absence of `needs-human` before returning success. If its 24-hour window expired, the retry conditionally rotates the generation and retains the original plus recent recovered generations as bounded lineage; summary projection accepts only that receipt-bound recovered lineage. A fresh preparation clears the lineage so a different attempt cannot inherit it. The reporter's consumed-receipt primitive requires non-start proof, so an ordinary summary or confirmation caller cannot bypass this boundary. Expired initialization is a separate preparation-only recovery: because `prepared` has not granted execution, exact claim/index and PR-state validation plus the conditional SHA are sufficient, while any consumed or conflicting state stays fail-closed. The reporter then projects the settled generation, due/expiry times, and non-running status to a matching trusted summary; retrying a lost projection response is idempotent. A running summary is bound to its exact repository, run ID, and run attempt, so a delayed reporter cannot clear a newer run that happens to share the same generation. Immediately before projection, the reporter re-reads the authoritative ledger and treats any changed or non-available settled receipt as superseded without falling through to the ordinary summary writer. A sweep-triggered run may have no webhook PR association; the reporter reads the exact attempt locator directly and verifies the receipt ID and bindings in the authoritative PR ledger without scanning historical PRs. Missing, conflicting or uncertain lookup fails explicitly for owner reconciliation rather than asserting no worker executed. Associated runs use their known PR directly. Consumer fingerprinting includes run ID and attempt so a rerun cannot be skipped as identical to its predecessor. A missing or inconclusive job read never refunds consumed authority. -Reporter concurrency is only a mutual-exclusion boundary; GitHub may replace an older pending job when several runs share one group. Every reporter wake therefore reconciles the PR ledger's current `receipt`, `released_receipt`, and `recovered_receipt` obligations before handling its triggering event. Each owner attempt is verified through its immutable index, exact run attempt, originating-head registry, and exact worker evidence. The hourly sweep independently dispatches a PR-bound reporter replay before ordinary keepalive evaluation, so a cancelled last reporter or a crash before projection does not require another producer completion. Replay is bounded to three ledger rereads, never scans historical indexes, and treats missing runs, unknown worker evidence, changed heads, hard human holds, or inaccessible storage as retryable uncertainty rather than absence. Consumer fingerprint debounce cannot suppress this manual replay path. The guarantee is eventual reconciliation of current ledger obligations, not an audit acknowledgment for every superseded historical completion. +Reporter concurrency is only a mutual-exclusion boundary; GitHub may replace an older pending job when several runs share one group. Every reporter wake therefore reconciles the PR ledger's current `receipt`, `released_receipt`, and `recovered_receipt` obligations before handling its triggering event. Each owner attempt is verified through its immutable index, exact run attempt, originating-head registry, and exact worker evidence. The hourly sweep independently dispatches a PR-bound reporter replay before ordinary keepalive evaluation, so a cancelled last reporter or a crash before projection does not require another producer completion. Replay is bounded to three ledger rereads, never scans historical indexes, and treats a confirmed missing ledger for an ordinary PR as a successful no-op. Missing runs, unknown worker evidence, changed heads, hard human holds, and inaccessible storage remain retryable uncertainty rather than absence. Consumer fingerprint debounce cannot suppress this manual replay path. The guarantee is eventual reconciliation of current ledger obligations, not an audit acknowledgment for every superseded historical completion. Reporter mutation is serialized by the resolved PR, including for unassociated `workflow_dispatch` runs. A read-only resolver job classifies the originating diff --git a/templates/consumer-repo/.github/scripts/keepalive_reporter_applicability.js b/templates/consumer-repo/.github/scripts/keepalive_reporter_applicability.js index cc99c7746..73909455b 100644 --- a/templates/consumer-repo/.github/scripts/keepalive_reporter_applicability.js +++ b/templates/consumer-repo/.github/scripts/keepalive_reporter_applicability.js @@ -168,7 +168,11 @@ async function replayReporterAuthority({ const results = []; const seen = new Set(); for (let pass = 0; pass < maxPasses; pass += 1) { - const { state } = await readAuthority(request, repository, number); + const authority = await readAuthority(request, repository, number, { allowMissing: true }); + // Most keepalive PRs never enter the challenge path and have no ledger. + // Only a confirmed 404 is an empty replay; all other read failures remain fatal. + if (authority === null) break; + const { state } = authority; const attempts = [state.receipt, state.released_receipt, state.recovered_receipt] .map((receipt) => parseOwnerAttempt(repository, receipt?.owner_attempt)) .filter(Boolean)