diff --git a/open-sse/config/providerErrorRules.ts b/open-sse/config/providerErrorRules.ts index d9eb14535b2..38cb0d86cc3 100644 --- a/open-sse/config/providerErrorRules.ts +++ b/open-sse/config/providerErrorRules.ts @@ -104,6 +104,35 @@ function buildOpencodeRules(): ProviderErrorRule[] { ]; } +// ─── Cloudflare Workers AI ──────────────────────────────────────────────── +// Free tier = 10,000 Neurons/day, shared across the WHOLE account +// (docs/reference/FREE_TIERS.md; official: developers.cloudflare.com/ +// workers-ai/platform/errors/). The exhaustion body doesn't match any +// QUOTA_PATTERNS keyword (src/shared/utils/classify429.ts) so it falls +// through to rate_limit and gets retried every ~60s against a budget that +// only resets at UTC midnight. +// +// Scope note: `scope: "connection"` (not "provider") for the same reason as +// Opencode above — the neuron budget is per-account, and a single OmniRoute +// connection maps to one Cloudflare account. Multiple connections under the +// same provider name would mean multiple accounts, each with its own budget. +function buildCloudflareAiRules(): ProviderErrorRule[] { + return [ + { + id: "cloudflare-ai-daily-neuron-allocation", + match: ({ status, body }) => { + if (status !== 429) return null; + const text = JSON.stringify(body ?? "").toLowerCase(); + if (!text.includes("daily free allocation")) return null; + // No cooldownMs: recordModelLockoutFailure already sets + // quota_exhausted without one to "next UTC midnight" — exactly this + // budget's real reset semantics. + return { reason: "quota_exhausted", scope: "connection" }; + }, + }, + ]; +} + // ─── Minimax ──────────────────────────────────────────────────────────────── // Minimax returns per-model quota info via custom headers. The body is generic // "rate limit exceeded" so we MUST read the headers. Other models on the same @@ -141,6 +170,7 @@ export const providerRuleRegistry = new Map([ ["opencode-cli", buildOpencodeRules()], ["minimax", buildMinimaxRules()], ["minimax-passthrough", buildMinimaxRules()], + ["cloudflare-ai", buildCloudflareAiRules()], ]); /** @@ -194,7 +224,9 @@ export function getProviderErrorRuleMatch( */ export function parseResetCountdownMs(text: string): number | null { if (typeof text !== "string" || text.length === 0) return null; - const match = text.match(/resets?\s+in\s+(\d+)\s+(day|days|hour|hours|minute|minutes|second|seconds)\b/); + const match = text.match( + /resets?\s+in\s+(\d+)\s+(day|days|hour|hours|minute|minutes|second|seconds)\b/ + ); if (!match) return null; const n = Number(match[1]); if (!Number.isFinite(n) || n <= 0) return null; diff --git a/src/shared/utils/classify429.ts b/src/shared/utils/classify429.ts index ff6f21ab312..f33abe1f0fa 100644 --- a/src/shared/utils/classify429.ts +++ b/src/shared/utils/classify429.ts @@ -53,6 +53,15 @@ const QUOTA_PATTERNS: ReadonlyArray = [ /individual quota reached/i, /enable overages/i, /INSUFFICIENT_G1_CREDITS_BALANCE/i, + + // Cloudflare Workers AI daily neuron budget exhaustion ("you have used up + // your daily free allocation of 10,000 neurons, please upgrade to + // Cloudflare's Workers Paid plan..."). No "quota"/"limit"/"exceed"/"credit" + // substring, so none of the patterns above match it. This is the primary + // provider-specific rule in providerErrorRules.ts (scope: "connection"); + // this entry is defense-in-depth for classify429FromError callers that + // bypass provider rule matching. + /daily free allocation/i, ]; /** diff --git a/tests/unit/provider-error-rules.test.ts b/tests/unit/provider-error-rules.test.ts index 163d08ea014..9dd1af4c509 100644 --- a/tests/unit/provider-error-rules.test.ts +++ b/tests/unit/provider-error-rules.test.ts @@ -83,6 +83,50 @@ test("S2b: provider error rules match canonical-cased plain header records", asy assert.equal(minimaxMatch.scope, "model"); }); +test("S2c: Cloudflare Workers AI daily neuron exhaustion → QUOTA_EXHAUSTED with connection scope", async () => { + // Cloudflare's free tier is a 10,000-Neurons/day budget shared across the + // WHOLE account. The exhaustion body doesn't contain "quota"/"limit"/ + // "exceed"/"credit" so it would otherwise fall through to the default + // RATE_LIMIT_EXCEEDED and get retried every ~60s against a budget that + // only resets at UTC midnight. + const { getProviderErrorRuleMatch } = await import("../../open-sse/config/providerErrorRules.ts"); + + const match = getProviderErrorRuleMatch( + "cloudflare-ai", + 429, + {}, + { + errors: [ + { + message: + "AiError: AiError: you have used up your daily free allocation of 10,000 neurons, please upgrade to Cloudflare's Workers Paid plan if you would like to continue usage.", + code: 4006, + }, + ], + success: false, + } + ); + assert.ok(match, "cloudflare-ai must have a rule matching the daily neuron allocation body"); + assert.equal(match.reason, "quota_exhausted"); + assert.equal( + match.scope, + "connection", + "Cloudflare's neuron budget is account-wide, so the lock must scope to the connection, not a single model" + ); + + // A 429 without the exhaustion wording (a real transient rate-limit) must + // NOT match — this rule is specific to the daily-allocation body. + const noMatch = getProviderErrorRuleMatch( + "cloudflare-ai", + 429, + {}, + { + errors: [{ message: "Too many requests, please retry shortly.", code: 3040 }], + } + ); + assert.equal(noMatch, null, "a generic 429 without the exhaustion wording must not match"); +}); + test("S3: Regression — provider with no rules falls back to global ERROR_RULES unchanged", () => { // A provider not in the registry (e.g. "unknown-vendor") must NOT cause // classifyError to crash or return a different result. It must behave