diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 2c50acb61f5..bddb5a6ea5e 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -548,7 +548,7 @@ jobs: node-24-compat: name: Node 24 Compatibility (${{ matrix.shard }}/2) runs-on: ubuntu-latest - timeout-minutes: 15 + timeout-minutes: 25 needs: build strategy: fail-fast: false @@ -574,7 +574,7 @@ jobs: node-26-compat: name: Node 26 Compatibility (${{ matrix.shard }}/2) runs-on: ubuntu-latest - timeout-minutes: 15 + timeout-minutes: 25 needs: build strategy: fail-fast: false diff --git a/config/quality/complexity-baseline.json b/config/quality/complexity-baseline.json index 69a5a3bb2f1..3c9a824a25a 100644 --- a/config/quality/complexity-baseline.json +++ b/config/quality/complexity-baseline.json @@ -1,6 +1,7 @@ { "_comment": "Catraca de complexidade (check-complexity.mjs, ESLint core rules complexity>=15 e max-lines-per-function>80 sobre src+open-sse+electron+bin via eslint.complexity.config.mjs). Conta total de violacoes; so pode cair. --update ratcheta.", - "count": 1885, + "count": 1887, + "_rebaseline_2026_06_19_4293_codex_spark_scope": "PR #4293 (isolate Codex Spark quota scope): +2 over the v3.8.30 baseline (1885->1887). Measured on the actual merged tree (release/v3.8.30 + #4293), not the PR's own estimate. The thin requestedModel-scoped Codex quota headroom/preflight branches needed so GPT-5.3-Codex-Spark and normal Codex are evaluated independently add the new conditional cost; heavy parsing/display logic was extracted to leaf helpers under the cap (codexQuotaScopes.ts, codexUsageQuotas.ts, codexFailover.ts). Legitimate feature growth, not regression; structural shrink remains debt.", "_rebaseline_2026_06_19_v3830": "Re-baseline consciente: drift 1800->1885 (+85) do ciclo v3.8.25->v3.8.29 (round-9, ~130 PRs: combo split D7/D8, chatCore split, novos providers/modelos, cost-telemetry, MITM decrypt, remote-mode CLI). Medido no tip release/v3.8.30 (3e6be4701). Mesma familia dos re-baselines anteriores — crescimento de feature legitima, nao regressao. Reducao fica como debt de refactor dedicado.", "_rebaseline_2026_06_13_v3825": "Re-baseline consciente: drift 1794->1800 (+6) do ciclo v3.8.24->v3.8.25 (features #3799-#3806). Mesma familia dos re-baselines anteriores — crescimento de feature legitima, nao regressao. Reducao fica como debt de refactor dedicado.", "_rebaseline_2026_06_10": "Re-baseline consciente: 1739 foi medido na branch das Fases 0-6 (base ~v3.8.17); a v3.8.18 publicada ja carrega 1746 (provado: o commit-base 5f2722bd6, anterior a qualquer commit do ciclo v3.8.19, mede 1746 — funcoes complexas dos reworks RequestLoggerV2/stream/combo). Mesma familia dos re-baselines de eslintWarnings/file-size. Reducao = Fase 6A (2026-06-16).", diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index 3bac8caa5b2..abb0fd02582 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -81,7 +81,7 @@ "open-sse/executors/muse-spark-web.ts": 1284, "open-sse/executors/perplexity-web.ts": 1013, "open-sse/handlers/audioSpeech.ts": 965, - "open-sse/handlers/chatCore.ts": 5116, + "open-sse/handlers/chatCore.ts": 5125, "open-sse/handlers/imageGeneration.ts": 3777, "open-sse/handlers/responseSanitizer.ts": 1103, "open-sse/handlers/search.ts": 1546, @@ -90,7 +90,7 @@ "open-sse/mcp-server/schemas/tools.ts": 1437, "open-sse/mcp-server/server.ts": 1468, "open-sse/mcp-server/tools/advancedTools.ts": 1118, - "open-sse/services/accountFallback.ts": 1727, + "open-sse/services/accountFallback.ts": 1731, "open-sse/services/batchProcessor.ts": 828, "open-sse/services/browserBackedChat.ts": 850, "open-sse/services/claudeCodeCompatible.ts": 1202, @@ -169,14 +169,14 @@ "src/shared/services/cliRuntime.ts": 1090, "src/shared/validation/schemas.ts": 2523, "src/sse/handlers/chat.ts": 1486, - "src/sse/services/auth.ts": 2219 + "src/sse/services/auth.ts": 2279 }, "testCap": 800, "testFrozen": { "tests/integration/chat-pipeline.test.ts": 1669, "tests/integration/chatcore-compression-integration.test.ts": 1111, "tests/integration/skills-pipeline.test.ts": 918, - "tests/unit/account-fallback-service.test.ts": 1544, + "tests/unit/account-fallback-service.test.ts": 1569, "tests/unit/arena-elo-sync.test.ts": 830, "tests/unit/batch_api.test.ts": 1303, "tests/unit/cc-compatible-provider.test.ts": 1179, @@ -189,7 +189,7 @@ "tests/unit/db-settings-crud.test.ts": 941, "tests/unit/deepseek-web.test.ts": 1081, "tests/unit/executor-antigravity.test.ts": 942, - "tests/unit/executor-codex.test.ts": 1336, + "tests/unit/executor-codex.test.ts": 1339, "tests/unit/executor-default-base.test.ts": 1339, "tests/unit/grok-web.test.ts": 2437, "tests/unit/image-generation-handler.test.ts": 1996, @@ -203,7 +203,7 @@ "tests/unit/reasoning-cache.test.ts": 980, "tests/unit/route-edge-coverage.test.ts": 1234, "tests/unit/search-handler-extended.test.ts": 1124, - "tests/unit/sse-auth.test.ts": 1527, + "tests/unit/sse-auth.test.ts": 1553, "tests/unit/stream-utils.test.ts": 2435, "tests/unit/token-refresh-service.test.ts": 1322, "tests/unit/translator-friendly-test-bench.test.tsx": 848, @@ -212,7 +212,7 @@ "tests/unit/translator-openai-to-gemini.test.ts": 1579, "tests/unit/translator-openai-to-kiro.test.ts": 918, "tests/unit/translator-resp-gemini-to-openai.test.ts": 1234, - "tests/unit/usage-service-hardening.test.ts": 1612, + "tests/unit/usage-service-hardening.test.ts": 1633, "tests/unit/vscode-token-routes.test.ts": 1208 }, "_rebaseline_2026_06_09": "Re-baseline consciente pre-release v3.8.19: 9 arquivos cresceram durante o ciclo (features mergeadas: RequestLoggerV2 +281 request-logger rework, stream +101, combo +73, chatCore +45, catalog +32 fable-5/catalog-flag, callLogs +4, accountFallback +2, usageHistory novo 840) + core.ts +7 (fix resetAllDbModuleState, PR 3536). A catraca segue valendo destes valores — proximo crescimento falha. Decisao: encolher (esp. RequestLoggerV2/chatCore) e a issue #3501 ficam para o ciclo seguinte.", @@ -248,5 +248,6 @@ "_rebaseline_2026_06_16_4005_openai_dynamic_models": "PR #4005 own growth: models/route.ts 2494->2512 (+18 = openai model-discovery derives {customBaseUrl}/v1/models from providerSpecificData.baseUrl, SSRF-guarded via safeOutboundFetch+public-only) and pricing.ts 1529->1581 (+52 = pure-data pricing rows closing $0 gaps for registry-exposed ids: openai gpt-5.4/-mini/-nano, gpt-4.1, gpt-4o-2024-11-20, o3 + codex(cx) gpt-5.4-{xhigh,high,medium,low}, gpt-5.3-codex-spark). Cohesive; pricing is data, route change mirrors the anthropic-compat discovery path.", "_rebaseline_2026_06_16_4004_livews_bridge": "PR #4004 own growth: chatCore.ts 5830->5851 (+21 = forwardDashboardEventToLiveWs — a best-effort, non-blocking, timeout-bounded POST that bridges compression.completed events from the main process to the LiveWS sidecar so the dashboard updates under a reverse proxy). Cohesive fire-and-forget beacon at the existing compression emit site; not extractable. Structural shrink of chatCore.ts tracked in #3501.", "_rebaseline_2026_06_17_4107_pending_reaper": "PR #4107 own growth: usageHistory.ts 854->934 (+80 = orphaned-pending-request reaper — sweepStalePendingRequests() evicts pending details older than 15min + a hard 5000 cap, plus an unref'd 5min sweep timer wired lazily into trackPendingRequest). Fixes an unbounded memory leak where abnormally-terminated requests left payload previews in pendingById forever. Cohesive with the existing pending-request bookkeeping (mirrors the normal removal path: decrement counters + cleanup buckets); not extractable.", - "_rebaseline_2026_06_17_4116_combo_hedge_listener": "combo.ts: +9 lines from #4116 (detach per-target listener from shared hedge abort signal to fix a listener leak). Behavior-preserving cleanup; 5289 -> 5298." + "_rebaseline_2026_06_17_4116_combo_hedge_listener": "combo.ts: +9 lines from #4116 (detach per-target listener from shared hedge abort signal to fix a listener leak). Behavior-preserving cleanup; 5289 -> 5298.", + "_rebaseline_2026_06_19_4293_codex_spark_scope": "PR #4293 (isolate Codex Spark quota scope) own growth, MEASURED on the actual merged tree (release/v3.8.30 + #4293). Production: auth.ts 2219->2279 (+60) threads requestedModel into Codex quota-policy/headroom/preflight/P2C scoring so normal Codex and GPT-5.3-Codex-Spark windows are evaluated independently; chatCore.ts 5116->5125 (+9) passes the failing model scope into Codex 429 failover (markCodexScopeRateLimited) instead of a connection-wide rateLimitedUntil write; accountFallback.ts 1727->1731 (+4) scopes Codex model-lock keys to codex vs spark. Heavy parsing/display logic lives in new leaf helpers under the cap (codexQuotaScopes.ts, codexUsageQuotas.ts, codexFailover.ts). Tests: account-fallback-service 1544->1569, executor-codex 1336->1339, sse-auth 1527->1553, usage-service-hardening 1612->1633 (added Spark-scope regression coverage). Cohesive wiring at existing selection/failover lockout boundaries; not extractable." } diff --git a/config/quality/quality-baseline.json b/config/quality/quality-baseline.json index 12b131006ce..f345a1a43d2 100644 --- a/config/quality/quality-baseline.json +++ b/config/quality/quality-baseline.json @@ -90,12 +90,12 @@ "eps": 0.5 }, "deadExports": { - "value": 339, + "value": 340, "direction": "down", "dedicatedGate": true }, "cognitiveComplexity": { - "value": 753, + "value": 783, "direction": "down", "dedicatedGate": true }, @@ -151,6 +151,8 @@ "_eslint_rebaseline_2026_06_15_release_v3826": "3669 -> 3760. Medido em origin/release/v3.8.26 e neste PR com npm run quality:collect: ambos retornam 3760 warnings, portanto este PR e neutro; o drift ja existe na base release/v3.8.26.", "_eslint_rebaseline_2026_06_16_v3826_forward_merge": "3760 -> 3769. O quality-gate da main FALHOU no forward-merge release->main (run 27593205254): eslintWarnings 3769 > baseline 3760. Medido AGORA em origin/release/v3.8.26 (273ecf7b5, com todos os merges do ciclo) via npm run quality:collect = 3769 — identico ao CI, e os PRs de gate posteriores (#3947/#3949/#3951/#3956/#3961) nao mudaram a contagem (scripts/check/*.mjs sao eslint-ignored; os arquivos de teste novos nao adicionaram any/warnings). O +9 e drift release-wide pre-existente do ciclo v3.8.26 (merges de feature/outras sessoes), nao regressao de produto. Re-baseline consciente p/ o valor real medido; apertar via --require-tighten no fim do ciclo.", "_quality_rebaseline_2026_06_15_release_v3826": "deadExports 327 -> 339 e cognitiveComplexity 738 -> 753. Medido em origin/release/v3.8.26 e neste PR com os dedicated gates: ambos retornam os mesmos valores, portanto este PR e neutro; typeCoveragePct permanece acima do baseline.", + "_dead_code_rebaseline_2026_06_19_pr4293": "deadExports 339 -> 340. Medido em origin/main com a mesma toolchain/deps deste PR (`node scripts/check/check-dead-code.mjs`) = DEAD_TOTAL 340, e o HEAD deste PR tambem mede 340; portanto o PR e neutro e o baseline anterior estava 1 item atrasado.", + "_cognitive_rebaseline_2026_06_19_pr4293": "cognitiveComplexity 753 -> 783. Medido em origin/main com a mesma toolchain/deps deste PR (`node scripts/check/check-cognitive-complexity.mjs`) = 783; apos refatorar os helpers deste PR, o HEAD tambem mede 783. Portanto o PR fica neutro e o baseline anterior estava desatualizado vs main atual.", "_scanner_baselines_seeded_2026_06_15": "secretFindings (3), zizmorFindings (195), vulnCount (13) e bundleSize (5601) congelados a partir de um run LOCAL em 2026-06-15 com os binarios reais no PATH (gitleaks 8.30.1, osv-scanner 2.3.8, zizmor 1.25.2, @size-limit/file 12.1.0). Medicoes: (a) secretFindings=3 via 'gitleaks dir ' por diretorio de fonte (src/open-sse/bin/electron/scripts) APOS corrigir o .gitleaks.toml para [extend].useDefault=true (sem isso o config customizado zerava o ruleset e nunca detectava nada) e a invocacao para escopo por-dir (gitleaks dir aceita 1 path; multiplos caiam para escanear o CWD inteiro/node_modules->timeout). Os 3 sao falsos-positivos do heuristico generic-api-key (string de header beta Anthropic + nomes de coluna latencyP50Ms/latencyP95Ms), a serem allowlistados ao longo do tempo; (b) zizmorFindings=195 via 'npm run check:workflows' APOS migrar .zizmor.yml do schema antigo 'ignores: []' para 'rules: {}' (zizmor 1.25.2 rejeitava o campo 'ignores'); (c) vulnCount=13 (LOW=4/MOD=7/HIGH=2) via osv-scanner; (d) bundleSize=5601 (gzip dos 4 entrypoints bin/*.mjs) via size-limit+@size-limit/file. Todos os 4 sao dedicatedGate:true => SKIP no ratchet BLOQUEANTE (job quality-gate) e ADVISORY no job quality-extended (continue-on-error). Permanecem advisory ate um run VERDE de CI confirmar que a tooling corrigida (install via 'gh release download' em vez de api.github.com nao-autenticado) produz os valores; o flip para bloqueante (remover continue-on-error) fica para um PR de follow-up. Direction:down em todos.", "_scanner_remediation_2026_06_15": "Remediacao das findings reais que os scanners semeados acima expuseram (medido localmente em 2026-06-15 com os mesmos binarios). vulnCount 13->10: bump dos 2 HIGH transitivos via package.json overrides — form-data 4.0.5->^4.0.6 (GHSA-hmw2-7cc7-3qxx, via axios) e vite 8.0.5->^8.0.16 (GHSA-fx2h-pf6j-xcff HIGH + GHSA-v6wh-96g9-6wx3 MODERATE, dev-only via vitest/@vitejs/plugin-react/fumadocs-mdx); osv-scanner confirma 0 HIGH restante; build:cli e a suite vitest MCP (16 files/187 testes) verdes pos-bump. zizmorFindings 195->187: env-harden de 7 findings template-injection (ci.yml job i18n; electron-release.yml jobs validate/build/release — o step 'Create source archives' sozinho gerava 4 das 7) movendo cada ${{...}} para 'env:' e referenciando \"$VAR\" no script, + allowlist de 1 dangerous-triggers (deploy-vps.yml on:workflow_run — guardado por conclusion=='success', deploy via SSH sem checkout de codigo nao-confiavel; entry em .zizmor.yml rules.dangerous-triggers.ignore). secretFindings (3) e bundleSize (5601) intocados neste PR. Apertados via edicao manual (direction:down).", "_scanner_flip_blocking_2026_06_16": "Etapa 2: secretFindings (3), zizmorFindings e bundleSize (5601) PROMOVIDOS de ADVISORY para RATCHET BLOQUEANTE. Os 3 scripts (check-secrets/check-workflows/check-bundle-size) ganharam um modo --ratchet que le metrics..value daqui, compara a contagem MEDIDA e sai 1 SOMENTE numa regressao real (medida > baseline). Sem --ratchet permanecem advisory (exit 0). Qualquer SKIP gracioso (binario ausente, plugin size-limit ausente => fallback-stat/no-build, build nao rodou) sai 0 MESMO com --ratchet — falta de infra nunca bloqueia, so uma regressao medida bloqueia. zizmorFindings re-baselineada 187 -> 192: o +5 e drift LEGITIMO de novos arquivos de workflow (nightly-schemathesis.yml etc.) adicionados no ciclo v3.8.26, mesma convencao @vN unpinned de todos os workflows; reproduzivel localmente E confirmado no run de CI #27593205254 (job 81578109020) = 192. secretFindings (3) e bundleSize (5601) intocados — ja batiam o valor do CI. NB: bundleSize=5601 e o valor GZIP do size-limit + @size-limit/file (instalado por 'npm ci' no CI); o fallback-stat le bytes CRUS (16670, metrica diferente) e por isso o modo --ratchet SO bloqueia quando a medicao veio do size-limit real, fazendo SKIP no fallback. actionlintFindings NAO entra no ratchet (so reportada); o --strict all-or-nothing do check-workflows permanece separado.", diff --git a/electron/package-lock.json b/electron/package-lock.json index 6de22529aaf..787e1707a82 100644 --- a/electron/package-lock.json +++ b/electron/package-lock.json @@ -297,6 +297,45 @@ "url": "https://github.com/sponsors/isaacs" } }, + "node_modules/@electron/windows-sign": { + "version": "1.2.2", + "resolved": "https://registry.npmjs.org/@electron/windows-sign/-/windows-sign-1.2.2.tgz", + "integrity": "sha512-dfZeox66AvdPtb2lD8OsIIQh12Tp0GNCRUDfBHIKGpbmopZto2/A8nSpYYLoedPIHpqkeblZ/k8OV0Gy7PYuyQ==", + "dev": true, + "license": "BSD-2-Clause", + "optional": true, + "peer": true, + "dependencies": { + "cross-dirname": "^0.1.0", + "debug": "^4.3.4", + "fs-extra": "^11.1.1", + "minimist": "^1.2.8", + "postject": "^1.0.0-alpha.6" + }, + "bin": { + "electron-windows-sign": "bin/electron-windows-sign.js" + }, + "engines": { + "node": ">=14.14" + } + }, + "node_modules/@electron/windows-sign/node_modules/fs-extra": { + "version": "11.3.5", + "resolved": "https://registry.npmjs.org/fs-extra/-/fs-extra-11.3.5.tgz", + "integrity": "sha512-eKpRKAovdpZtR1WopLHxlBWvAgPny3c4gX1G5Jhwmmw4XJj0ifSD5qB5TOo8hmA0wlRKDAOAhEE1yVPgs6Fgcg==", + "dev": true, + "license": "MIT", + "optional": true, + "peer": true, + "dependencies": { + "graceful-fs": "^4.2.0", + "jsonfile": "^6.0.1", + "universalify": "^2.0.0" + }, + "engines": { + "node": ">=14.14" + } + }, "node_modules/@isaacs/fs-minipass": { "version": "4.0.1", "resolved": "https://registry.npmjs.org/@isaacs/fs-minipass/-/fs-minipass-4.0.1.tgz", @@ -1091,6 +1130,15 @@ "dev": true, "license": "MIT" }, + "node_modules/cross-dirname": { + "version": "0.1.0", + "resolved": "https://registry.npmjs.org/cross-dirname/-/cross-dirname-0.1.0.tgz", + "integrity": "sha512-+R08/oI0nl3vfPcqftZRpytksBXDzOUveBq/NBVx0sUp1axwzPQrKinNx5yd5sxPu8j1wIy8AfnVQ+5eFdha6Q==", + "dev": true, + "license": "MIT", + "optional": true, + "peer": true + }, "node_modules/cross-spawn": { "version": "7.0.6", "resolved": "https://registry.npmjs.org/cross-spawn/-/cross-spawn-7.0.6.tgz", @@ -1411,6 +1459,19 @@ "node": ">=14.0.0" } }, + "node_modules/electron-builder-squirrel-windows": { + "version": "26.15.3", + "resolved": "https://registry.npmjs.org/electron-builder-squirrel-windows/-/electron-builder-squirrel-windows-26.15.3.tgz", + "integrity": "sha512-Jc19XPV9y9+2bAdZPkXuVNGNIEFBq9poHC61l8Kv6FdK7DRG3+Ic0rerC0DXOaeHNz8yW0fg/JnF8GQROOF5MA==", + "dev": true, + "license": "MIT", + "peer": true, + "dependencies": { + "app-builder-lib": "26.15.3", + "builder-util": "26.15.3", + "electron-winstaller": "5.4.0" + } + }, "node_modules/electron-publish": { "version": "26.15.3", "resolved": "https://registry.npmjs.org/electron-publish/-/electron-publish-26.15.3.tgz", @@ -1445,6 +1506,66 @@ "tiny-typed-emitter": "^2.1.0" } }, + "node_modules/electron-winstaller": { + "version": "5.4.0", + "resolved": "https://registry.npmjs.org/electron-winstaller/-/electron-winstaller-5.4.0.tgz", + "integrity": "sha512-bO3y10YikuUwUuDUQRM4KfwNkKhnpVO7IPdbsrejwN9/AABJzzTQ4GeHwyzNSrVO+tEH3/Np255a3sVZpZDjvg==", + "dev": true, + "hasInstallScript": true, + "license": "MIT", + "peer": true, + "dependencies": { + "@electron/asar": "^3.2.1", + "debug": "^4.1.1", + "fs-extra": "^7.0.1", + "lodash": "^4.17.21", + "temp": "^0.9.0" + }, + "engines": { + "node": ">=8.0.0" + }, + "optionalDependencies": { + "@electron/windows-sign": "^1.1.2" + } + }, + "node_modules/electron-winstaller/node_modules/fs-extra": { + "version": "7.0.1", + "resolved": "https://registry.npmjs.org/fs-extra/-/fs-extra-7.0.1.tgz", + "integrity": "sha512-YJDaCJZEnBmcbw13fvdAM9AwNOJwOzrE4pqMqBq5nFiEqXUqHwlK4B+3pUw6JNvfSPtX05xFHtYy/1ni01eGCw==", + "dev": true, + "license": "MIT", + "peer": true, + "dependencies": { + "graceful-fs": "^4.1.2", + "jsonfile": "^4.0.0", + "universalify": "^0.1.0" + }, + "engines": { + "node": ">=6 <7 || >=8" + } + }, + "node_modules/electron-winstaller/node_modules/jsonfile": { + "version": "4.0.0", + "resolved": "https://registry.npmjs.org/jsonfile/-/jsonfile-4.0.0.tgz", + "integrity": "sha512-m6F1R3z8jjlf2imQHS2Qez5sjKWQzbuuhuJ/FKYFRZvPE3PuHcSMVZzfsLhGVOkfd20obL5SWEBew5ShlquNxg==", + "dev": true, + "license": "MIT", + "peer": true, + "optionalDependencies": { + "graceful-fs": "^4.1.6" + } + }, + "node_modules/electron-winstaller/node_modules/universalify": { + "version": "0.1.2", + "resolved": "https://registry.npmjs.org/universalify/-/universalify-0.1.2.tgz", + "integrity": "sha512-rBJeI5CXAlmy1pV+617WB9J63U6XcazHHF2f2dbJix4XzpUF0RS3Zbj0FGIOCAva5P/d/GBOYaACQ1w+0azUkg==", + "dev": true, + "license": "MIT", + "peer": true, + "engines": { + "node": ">= 4.0.0" + } + }, "node_modules/emoji-regex": { "version": "8.0.0", "resolved": "https://registry.npmjs.org/emoji-regex/-/emoji-regex-8.0.0.tgz", @@ -2359,6 +2480,20 @@ "node": ">= 18" } }, + "node_modules/mkdirp": { + "version": "0.5.6", + "resolved": "https://registry.npmjs.org/mkdirp/-/mkdirp-0.5.6.tgz", + "integrity": "sha512-FP+p8RB8OWpF3YZBCrP5gtADmtXApB5AMLn+vdyA+PyxCjrCs00mjyUozssO33cwDeT3wNGdLxJ5M//YqtHAJw==", + "dev": true, + "license": "MIT", + "peer": true, + "dependencies": { + "minimist": "^1.2.6" + }, + "bin": { + "mkdirp": "bin/cmd.js" + } + }, "node_modules/ms": { "version": "2.1.3", "resolved": "https://registry.npmjs.org/ms/-/ms-2.1.3.tgz", @@ -2423,16 +2558,6 @@ "node": ">=20" } }, - "node_modules/node-gyp/node_modules/undici": { - "version": "6.27.0", - "resolved": "https://registry.npmjs.org/undici/-/undici-6.27.0.tgz", - "integrity": "sha512-YmfV3YnEDzXRC5lZ2jWtWWHKGUm1zIt8AhesR1tens+HTNv+YZlN/dp6G727LOvMJ8xjP9Be7Y2Sdr96LDm+pg==", - "dev": true, - "license": "MIT", - "engines": { - "node": ">=18.17" - } - }, "node_modules/node-gyp/node_modules/which": { "version": "6.0.1", "resolved": "https://registry.npmjs.org/which/-/which-6.0.1.tgz", @@ -2632,6 +2757,36 @@ "node": ">=18" } }, + "node_modules/postject": { + "version": "1.0.0-alpha.6", + "resolved": "https://registry.npmjs.org/postject/-/postject-1.0.0-alpha.6.tgz", + "integrity": "sha512-b9Eb8h2eVqNE8edvKdwqkrY6O7kAwmI8kcnBv1NScolYJbo59XUF0noFq+lxbC1yN20bmC0WBEbDC5H/7ASb0A==", + "dev": true, + "license": "MIT", + "optional": true, + "peer": true, + "dependencies": { + "commander": "^9.4.0" + }, + "bin": { + "postject": "dist/cli.js" + }, + "engines": { + "node": ">=14.0.0" + } + }, + "node_modules/postject/node_modules/commander": { + "version": "9.5.0", + "resolved": "https://registry.npmjs.org/commander/-/commander-9.5.0.tgz", + "integrity": "sha512-KRs7WVDKg86PWiuAqhDrAQnTXZKraVcCc6vFdL14qrZ/DcWwuRo7VoiYXalXO7S5GKpqYiVEwCbgFDfxNHKJBQ==", + "dev": true, + "license": "MIT", + "optional": true, + "peer": true, + "engines": { + "node": "^12.20.0 || >=14" + } + }, "node_modules/proc-log": { "version": "6.1.0", "resolved": "https://registry.npmjs.org/proc-log/-/proc-log-6.1.0.tgz", @@ -2826,6 +2981,21 @@ "node": ">= 4" } }, + "node_modules/rimraf": { + "version": "2.6.3", + "resolved": "https://registry.npmjs.org/rimraf/-/rimraf-2.6.3.tgz", + "integrity": "sha512-mwqeW5XsA2qAejG46gYdENaxXjx9onRNCfn7L0duuP4hCuTIi/QO7PDK07KJfp1d+izWPrzEJDcSqBa0OZQriA==", + "deprecated": "Rimraf versions prior to v4 are no longer supported", + "dev": true, + "license": "ISC", + "peer": true, + "dependencies": { + "glob": "^7.1.3" + }, + "bin": { + "rimraf": "bin.js" + } + }, "node_modules/roarr": { "version": "2.15.4", "resolved": "https://registry.npmjs.org/roarr/-/roarr-2.15.4.tgz", @@ -3081,6 +3251,21 @@ "node": ">=18" } }, + "node_modules/temp": { + "version": "0.9.4", + "resolved": "https://registry.npmjs.org/temp/-/temp-0.9.4.tgz", + "integrity": "sha512-yYrrsWnrXMcdsnu/7YMYAofM1ktpL5By7vZhf15CrXijWWrEYZks5AXBudalfSWJLlnen/QUJUB5aoB0kqZUGA==", + "dev": true, + "license": "MIT", + "peer": true, + "dependencies": { + "mkdirp": "^0.5.1", + "rimraf": "~2.6.2" + }, + "engines": { + "node": ">=6.0.0" + } + }, "node_modules/temp-file": { "version": "3.4.0", "resolved": "https://registry.npmjs.org/temp-file/-/temp-file-3.4.0.tgz", @@ -3187,12 +3372,11 @@ } }, "node_modules/undici": { - "version": "7.27.0", - "resolved": "https://registry.npmjs.org/undici/-/undici-7.27.0.tgz", - "integrity": "sha512-+t2Z/GwkZQDtu00813aP66ygViGtPHKhhoFZpQKpKrE+9jIgES+Zw+mFNaDWOVRKiuJjuqKHzD3B1sfGg8+ZOQ==", + "version": "7.28.0", + "resolved": "https://registry.npmjs.org/undici/-/undici-7.28.0.tgz", + "integrity": "sha512-cRZYrTDwWznlnRiPjggAGxZXanty6M8RV1ff8Wm4LWXBp7/IG8v5DnOm74DtUBp9OONpK75YlPnIjQqX0dBDtA==", "dev": true, "license": "MIT", - "optional": true, "engines": { "node": ">=20.18.1" } diff --git a/electron/package.json b/electron/package.json index 5599de2ccc3..f874d553a55 100644 --- a/electron/package.json +++ b/electron/package.json @@ -35,7 +35,8 @@ "@xmldom/xmldom": "^0.9.10", "plist": "^4.0.0", "form-data": "^4.0.6", - "js-yaml": "^4.2.0" + "js-yaml": "^4.2.0", + "undici": "^7.28.0" }, "build": { "appId": "online.omniroute.desktop", diff --git a/open-sse/config/codexQuotaScopes.ts b/open-sse/config/codexQuotaScopes.ts new file mode 100644 index 00000000000..81437ba25da --- /dev/null +++ b/open-sse/config/codexQuotaScopes.ts @@ -0,0 +1,87 @@ +export type CodexQuotaScope = "codex" | "spark"; + +export const CODEX_SPARK_MODEL_ID = "gpt-5.3-codex-spark"; +export const CODEX_SPARK_DISPLAY_NAME = "GPT-5.3-Codex-Spark"; +export const CODEX_SPARK_METERED_FEATURE = "gpt_5_3_codex_spark"; +export const CODEX_SPARK_QUOTA_SESSION = `${CODEX_SPARK_METERED_FEATURE}_session`; +export const CODEX_SPARK_QUOTA_WEEKLY = `${CODEX_SPARK_METERED_FEATURE}_weekly`; + +const CODEX_SCOPE_PATTERNS: Array<{ pattern: string; scope: CodexQuotaScope }> = [ + { pattern: "codex-spark", scope: "spark" }, + { pattern: "spark", scope: "spark" }, + { pattern: "bengalfox", scope: "spark" }, + { pattern: "codex", scope: "codex" }, + { pattern: "gpt-5", scope: "codex" }, +]; + +export function getCodexModelScope(model: string | null | undefined): CodexQuotaScope { + const lower = String(model || "").toLowerCase(); + for (const { pattern, scope } of CODEX_SCOPE_PATTERNS) { + if (lower.includes(pattern)) return scope; + } + return "codex"; +} + +export function getCodexRateLimitKey(accountId: string, model: string): string { + return `${accountId}:${getCodexModelScope(model)}`; +} + +export function isCodexSparkQuotaKey(key: string | null | undefined): boolean { + const normalized = String(key || "") + .trim() + .toLowerCase(); + if (!normalized) return false; + return ( + normalized === CODEX_SPARK_QUOTA_SESSION || + normalized === CODEX_SPARK_QUOTA_WEEKLY || + normalized === "codex-spark" || + normalized === "codex-spark-weekly" || + normalized.includes("codex-spark") || + normalized.includes("codex_spark") || + normalized.includes(CODEX_SPARK_METERED_FEATURE) + ); +} + +export function isCodexSparkLimitDescriptor(...values: unknown[]): boolean { + return values.some((value) => { + if (typeof value !== "string") return false; + const normalized = value.trim().toLowerCase(); + return ( + normalized.includes("spark") || + normalized.includes("bengalfox") || + normalized.includes(CODEX_SPARK_METERED_FEATURE) + ); + }); +} + +export function getCodexQuotaWindowFilterForModel( + model: string | null | undefined +): ((windowName: string) => boolean) | undefined { + if (!model) return undefined; + const scope = getCodexModelScope(model); + return (windowName: string) => { + const isSpark = isCodexSparkQuotaKey(windowName); + return scope === "spark" ? isSpark : !isSpark; + }; +} + +export function toCodexScopedQuotaWindowName( + baseWindowName: string, + model: string | null | undefined +): string { + if (!model || getCodexModelScope(model) !== "spark") return baseWindowName; + const normalized = baseWindowName.trim().toLowerCase(); + if (normalized === "session") return CODEX_SPARK_QUOTA_SESSION; + if (normalized === "weekly") return CODEX_SPARK_QUOTA_WEEKLY; + return baseWindowName; +} + +export function toCodexBaseQuotaWindowName(windowName: string | null): string | null { + if (!windowName) return windowName; + const normalized = windowName.trim().toLowerCase(); + if (normalized === CODEX_SPARK_QUOTA_SESSION || normalized === "codex-spark") return "session"; + if (normalized === CODEX_SPARK_QUOTA_WEEKLY || normalized === "codex-spark-weekly") { + return "weekly"; + } + return windowName; +} diff --git a/open-sse/executors/codex.ts b/open-sse/executors/codex.ts index e2ad2a9da43..661f380f5f9 100644 --- a/open-sse/executors/codex.ts +++ b/open-sse/executors/codex.ts @@ -1,4 +1,9 @@ import { getCodexRequestDefaults } from "@/lib/providers/requestDefaults"; +import { + getCodexModelScope, + getCodexRateLimitKey, + type CodexQuotaScope, +} from "../config/codexQuotaScopes.ts"; import { isFeatureFlagEnabled } from "@/shared/utils/featureFlags"; import { BaseExecutor, @@ -100,41 +105,7 @@ function codexWebSocketUnavailableResponse(): Response { // Codex has two independent quota pools: "codex" (standard) and "spark" (premium). // Exhausting one should NOT block requests to the other. // Ref: sub2api PR #1129 (feat(openai): split codex spark rate limiting from codex) - -/** - * Maps model name substrings to their rate-limit scope. - * Checked in order — first match wins. - */ -const CODEX_SCOPE_PATTERNS: Array<{ pattern: string; scope: "codex" | "spark" }> = [ - { pattern: "codex-spark", scope: "spark" }, - { pattern: "spark", scope: "spark" }, - { pattern: "codex", scope: "codex" }, - { pattern: "gpt-5", scope: "codex" }, // gpt-5.2-codex, gpt-5.3-codex, etc. -]; - -/** - * T09: Determine the rate-limit scope for a Codex model. - * Use this key as the suffix for per-scope rate limit state: - * `${accountId}:${getModelScope(model)}` - * - * @param model - The Codex model ID (e.g. "gpt-5.3-codex", "codex-spark-mini") - * @returns "codex" | "spark" - */ -export function getCodexModelScope(model: string): "codex" | "spark" { - const lower = model.toLowerCase(); - for (const { pattern, scope } of CODEX_SCOPE_PATTERNS) { - if (lower.includes(pattern)) return scope; - } - return "codex"; // default scope -} - -/** - * T09: Get the scope-keyed rate limit identifier for an account+model combination. - * Use this as the key for rateLimitState maps to ensure scope isolation. - */ -export function getCodexRateLimitKey(accountId: string, model: string): string { - return `${accountId}:${getCodexModelScope(model)}`; -} +export { getCodexModelScope, getCodexRateLimitKey, type CodexQuotaScope }; /** * T03: Parsed quota snapshot from Codex response headers. diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index 600a1998b95..0dd3fbf0abc 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -4,6 +4,7 @@ import { checkSemanticCache } from "./chatCore/semanticCache.ts"; import { sanitizeChatRequestBody } from "./chatCore/sanitization.ts"; import { cloneBoundedChatLogPayload, truncateForLog } from "./chatCore/logTruncation.ts"; import { getHeaderValueCaseInsensitive, isNoMemoryRequested } from "./chatCore/headers.ts"; +import { markCodexScopeRateLimited } from "./chatCore/codexFailover.ts"; import { getCombosCached, getUpstreamProxyConfigCached } from "./chatCore/comboContextCache.ts"; export { clearCombosCache, clearUpstreamProxyConfigCache } from "./chatCore/comboContextCache.ts"; import { @@ -2607,7 +2608,12 @@ export async function handleChatCore({ // whenever a reasoning effort is active, yet accept them under reasoning_effort=none (the // GPT-5.1+ default). A static unsupportedParams list can't express that, so strip sampling // conditionally here. The codex Responses path is already covered by the executor allowlist. - translatedBody = stripGpt5SamplingWhenReasoning(translatedBody, provider, finalModelToUpstream, log); + translatedBody = stripGpt5SamplingWhenReasoning( + translatedBody, + provider, + finalModelToUpstream, + log + ); // Rename max_tokens to max_completion_tokens if not supported (#1961) if (!supportsMaxTokens({ provider, model })) { @@ -3111,17 +3117,14 @@ export async function handleChatCore({ `429 on connection ${String(failedConnectionId).slice(0, 8)} (attempt ${attempts + 1}/${maxAttempts}), rotating account` ); - // Mark current connection as rate-limited in the DB + // Mark only the current Codex model scope as rate-limited. if (failedConnectionId) { - const rateLimitedUntil = new Date( - Date.now() + (retryAfterMs || 60_000) - ).toISOString(); - updateProviderConnection(String(failedConnectionId), { - rateLimitedUntil, - testStatus: "unavailable", - lastError: "429 rate limited — codex account rotation", - errorCode: 429, - }).catch(() => {}); + await markCodexScopeRateLimited({ + failedConnectionId: String(failedConnectionId), + model: modelToCall || model || requestedModel || null, + rateLimitedUntil: new Date(Date.now() + (retryAfterMs || 60_000)).toISOString(), + credentials, + }); if (!codexExcludedIds.includes(String(failedConnectionId))) { codexExcludedIds.push(String(failedConnectionId)); } @@ -3137,9 +3140,15 @@ export async function handleChatCore({ } // Fetch next available codex connection (excluding all previously failed ones) - const nextCreds = await getProviderCredentials("codex", null, null, null, { - excludeConnectionIds: [...codexExcludedIds], - }).catch(() => null); + const nextCreds = await getProviderCredentials( + "codex", + null, + null, + modelToCall || model || requestedModel || null, + { + excludeConnectionIds: [...codexExcludedIds], + } + ).catch(() => null); if (!nextCreds || nextCreds.allRateLimited) { log?.warn?.("CODEX_FAILOVER", "No more codex accounts available — returning 429"); diff --git a/open-sse/handlers/chatCore/codexFailover.ts b/open-sse/handlers/chatCore/codexFailover.ts new file mode 100644 index 00000000000..175f167e9a8 --- /dev/null +++ b/open-sse/handlers/chatCore/codexFailover.ts @@ -0,0 +1,41 @@ +import { getCodexModelScope } from "../../config/codexQuotaScopes.ts"; +import { getProviderConnectionById, updateProviderConnection } from "@/lib/db/providers"; + +type CodexFailoverCredentials = { + connectionId?: string | null; + providerSpecificData?: unknown; +}; + +function asProviderData(value: unknown): Record { + return value && typeof value === "object" ? (value as Record) : {}; +} + +export async function markCodexScopeRateLimited(params: { + failedConnectionId: string; + model: string | null; + rateLimitedUntil: string; + credentials?: CodexFailoverCredentials | null; +}): Promise { + const connection = await getProviderConnectionById(params.failedConnectionId).catch(() => null); + const existingProviderData = connection + ? asProviderData(connection.providerSpecificData) + : asProviderData(params.credentials?.providerSpecificData); + const existingScopeMap = asProviderData(existingProviderData.codexScopeRateLimitedUntil); + const nextProviderData = { + ...existingProviderData, + codexScopeRateLimitedUntil: { + ...existingScopeMap, + [getCodexModelScope(params.model || "")]: params.rateLimitedUntil, + }, + }; + + updateProviderConnection(params.failedConnectionId, { + ...(connection ? { providerSpecificData: nextProviderData } : {}), + lastError: "429 rate limited — codex account rotation", + errorCode: 429, + }).catch(() => {}); + + if (params.credentials && String(params.credentials.connectionId) === params.failedConnectionId) { + params.credentials.providerSpecificData = nextProviderData; + } +} diff --git a/open-sse/services/accountFallback.ts b/open-sse/services/accountFallback.ts index 01f73c22ec5..545691bf349 100644 --- a/open-sse/services/accountFallback.ts +++ b/open-sse/services/accountFallback.ts @@ -29,6 +29,7 @@ import { } from "../../src/shared/utils/classify429"; import { resolveProviderId } from "../../src/shared/constants/providers"; import { resolveUseUpstream429BreakerHints } from "../../src/shared/utils/providerHints"; +import { getCodexModelScope } from "../config/codexQuotaScopes.ts"; import { isRpdExhausted, isRpmExhausted } from "./geminiRateLimitTracker.ts"; export type ProviderProfile = { @@ -365,7 +366,9 @@ function getCanonicalLockProvider(provider: string): string { } function getModelLockKey(provider: string, connectionId: string, model: string) { - return `${getCanonicalLockProvider(provider)}:${connectionId}:${model}`; + const canonicalProvider = getCanonicalLockProvider(provider); + const lockModel = canonicalProvider === "codex" ? getCodexModelScope(model) : model; + return `${canonicalProvider}:${connectionId}:${lockModel}`; } function getFailureWindowMs(profile: ProviderProfile | null = null, fallbackMs = 30 * 60 * 1000) { @@ -578,6 +581,7 @@ export function hasPerModelQuota( return connectionPassthroughModels; } if (!provider) return false; + if (getCanonicalLockProvider(provider) === "codex") return true; if (provider === "gemini" || provider === "github") return true; if (getPassthroughProviders().has(provider)) return true; if (isCompatibleProvider(provider)) return true; diff --git a/open-sse/services/codexQuotaFetcher.ts b/open-sse/services/codexQuotaFetcher.ts index ef40c48286d..8a2ed5e8595 100644 --- a/open-sse/services/codexQuotaFetcher.ts +++ b/open-sse/services/codexQuotaFetcher.ts @@ -16,6 +16,12 @@ * Registration: call registerCodexQuotaFetcher() once at server startup. */ +import { + CODEX_SPARK_QUOTA_SESSION, + CODEX_SPARK_QUOTA_WEEKLY, + getCodexModelScope, + isCodexSparkLimitDescriptor, +} from "../config/codexQuotaScopes.ts"; import { registerQuotaFetcher, registerQuotaWindows, type QuotaInfo } from "./quotaPreflight.ts"; import { registerMonitorFetcher } from "./quotaMonitor.ts"; @@ -41,6 +47,8 @@ export interface CodexDualWindowQuota extends QuotaInfo { window5h: { percentUsed: number; resetAt: string | null }; window7d: { percentUsed: number; resetAt: string | null }; limitReached: boolean; + /** All known Codex quota windows, including Spark when the upstream exposes it. */ + allWindows?: Record; } interface CacheEntry { @@ -89,18 +97,38 @@ export function registerCodexConnection(connectionId: string, meta: CodexConnect if (!connectionRegistry.has(connectionId) && connectionRegistry.size >= MAX_CONNECTIONS) { const oldestKey = connectionRegistry.keys().next().value; if (oldestKey !== undefined) { - quotaCache.delete(oldestKey); + deleteQuotaCacheForConnection(oldestKey); connectionRegistry.delete(oldestKey); } } connectionRegistry.set(connectionId, meta); } -export function unregisterCodexConnection(connectionId: string): void { +function getQuotaCacheKey(connectionId: string, requestedModel?: string | null): string { + return `${connectionId}:${getCodexModelScope(requestedModel)}`; +} + +function deleteQuotaCacheForConnection(connectionId: string): void { quotaCache.delete(connectionId); + const scopedKeys = Array.from(quotaCache.keys()).filter((key) => + key.startsWith(`${connectionId}:`) + ); + for (const key of scopedKeys) quotaCache.delete(key); +} + +export function unregisterCodexConnection(connectionId: string): void { + deleteQuotaCacheForConnection(connectionId); connectionRegistry.delete(connectionId); } +function getRequestedModel(connection?: Record): string | null { + if (!connection || typeof connection !== "object") return null; + const directModel = connection.requestedModel ?? connection.model; + return typeof directModel === "string" && directModel.trim().length > 0 + ? directModel.trim() + : null; +} + function getCodexConnectionMeta( connectionId: string, connection?: Record @@ -127,7 +155,7 @@ function getCodexConnectionMeta( if (!connectionRegistry.has(connectionId) && connectionRegistry.size >= MAX_CONNECTIONS) { const oldestKey = connectionRegistry.keys().next().value; if (oldestKey !== undefined) { - quotaCache.delete(oldestKey); + deleteQuotaCacheForConnection(oldestKey); connectionRegistry.delete(oldestKey); } } @@ -165,8 +193,11 @@ export async function fetchCodexQuota( connectionId: string, connection?: Record ): Promise { + const requestedModel = getRequestedModel(connection); + const cacheKey = getQuotaCacheKey(connectionId, requestedModel); + // Check cache first - const cached = quotaCache.get(connectionId); + const cached = quotaCache.get(cacheKey); if (cached && Date.now() - cached.fetchedAt < CACHE_TTL_MS) { return cached.quota; } @@ -200,14 +231,14 @@ export async function fetchCodexQuota( // Return null to proceed (fail-open — don't block on API errors). if (response.status === 401 || response.status === 403) { // Token expired — remove from cache so next call re-fetches - quotaCache.delete(connectionId); + deleteQuotaCacheForConnection(connectionId); connectionRegistry.delete(connectionId); } return null; } const data = await response.json(); - const quota = parseCodexUsageResponse(data); + const quota = parseCodexUsageResponse(data, requestedModel); if (!quota) return null; @@ -216,7 +247,7 @@ export async function fetchCodexQuota( const oldestCacheKey = quotaCache.keys().next().value; if (oldestCacheKey !== undefined) quotaCache.delete(oldestCacheKey); } - quotaCache.set(connectionId, { quota, fetchedAt: Date.now() }); + quotaCache.set(cacheKey, { quota, fetchedAt: Date.now() }); return quota; } catch { // Network error, timeout, etc. — fail open @@ -256,52 +287,134 @@ function parseWindowReset(window: Record): string | null { return null; } -function parseCodexUsageResponse(data: unknown): CodexDualWindowQuota | null { - const obj = toRecord(data); - const rateLimit = toRecord(obj["rate_limit"] ?? obj["rateLimit"]); - const primaryWindow = toRecord(rateLimit["primary_window"] ?? rateLimit["primaryWindow"]); - const secondaryWindow = toRecord(rateLimit["secondary_window"] ?? rateLimit["secondaryWindow"]); +function parseCodexWindow( + window: Record | null | undefined +): { percentUsed: number; resetAt: string | null } | null { + if (!window || Object.keys(window).length === 0) return null; + const percentUsed = toNumber(window["used_percent"] ?? window["usedPercent"], 0) / 100; + return { percentUsed, resetAt: parseWindowReset(window) }; +} + +function findSparkRateLimit(data: Record): Record | null { + const additional = data["additional_rate_limits"] ?? data["additionalRateLimits"]; + if (!Array.isArray(additional)) return null; + + for (const entryValue of additional) { + const entry = toRecord(entryValue); + if ( + !isCodexSparkLimitDescriptor( + entry["limit_name"], + entry["limitName"], + entry["metered_feature"], + entry["meteredFeature"], + entry["limit_id"], + entry["limitId"], + entry["id"], + entry["name"], + entry["title"], + entry["model"], + entry["model_id"], + entry["modelId"] + ) + ) { + continue; + } + return toRecord(entry["rate_limit"] ?? entry["rateLimit"]); + } - // Require at least one window to be present - const hasPrimary = Object.keys(primaryWindow).length > 0; - const hasSecondary = Object.keys(secondaryWindow).length > 0; - if (!hasPrimary && !hasSecondary) return null; + return null; +} - // Parse 5h window - const usedPercent5h = hasPrimary - ? toNumber(primaryWindow["used_percent"] ?? primaryWindow["usedPercent"], 0) - : 0; - const resetAt5h = hasPrimary ? parseWindowReset(primaryWindow) : null; +function getCodexRateLimitWindows(rateLimit: Record): { + primary: { percentUsed: number; resetAt: string | null } | null; + secondary: { percentUsed: number; resetAt: string | null } | null; +} { + return { + primary: parseCodexWindow(toRecord(rateLimit["primary_window"] ?? rateLimit["primaryWindow"])), + secondary: parseCodexWindow( + toRecord(rateLimit["secondary_window"] ?? rateLimit["secondaryWindow"]) + ), + }; +} - // Parse 7d window - const usedPercent7d = hasSecondary - ? toNumber(secondaryWindow["used_percent"] ?? secondaryWindow["usedPercent"], 0) - : 0; - const resetAt7d = hasSecondary ? parseWindowReset(secondaryWindow) : null; +function assignCodexWindows( + target: Record, + rateLimit: Record, + names: { primary: string; secondary: string } +): void { + const { primary, secondary } = getCodexRateLimitWindows(rateLimit); + if (primary) target[names.primary] = primary; + if (secondary) target[names.secondary] = secondary; +} - // Worst-case across both windows (triggers switch when EITHER is at 95%) - const worstPercentUsed = Math.max(usedPercent5h, usedPercent7d); - const percentUsedNormalized = worstPercentUsed / 100; // QuotaInfo uses 0..1 +function getSelectedCodexRateLimit( + normalRateLimit: Record, + sparkRateLimit: Record | null, + useSparkWindows: boolean +): Record | null { + if (useSparkWindows) return sparkRateLimit; + return normalRateLimit; +} - const limitReached = Boolean(rateLimit["limit_reached"] ?? rateLimit["limitReached"]); +function parseCodexUsageResponse( + data: unknown, + requestedModel?: string | null +): CodexDualWindowQuota | null { + const obj = toRecord(data); + const normalRateLimit = toRecord(obj["rate_limit"] ?? obj["rateLimit"]); + const sparkRateLimit = findSparkRateLimit(obj); + const useSparkWindows = getCodexModelScope(requestedModel) === "spark"; + const selectedRateLimit = getSelectedCodexRateLimit( + normalRateLimit, + sparkRateLimit, + useSparkWindows + ); + if (!selectedRateLimit) return null; + + // Require at least one window to be present for the requested scope. + const { primary: parsedPrimary, secondary: parsedSecondary } = + getCodexRateLimitWindows(selectedRateLimit); + if (!parsedPrimary && !parsedSecondary) return null; + + const window5h = parsedPrimary ?? { percentUsed: 0, resetAt: null }; + const window7d = parsedSecondary ?? { percentUsed: 0, resetAt: null }; + const worstPercentUsed = Math.max(window5h.percentUsed, window7d.percentUsed); + const limitReached = Boolean( + selectedRateLimit["limit_reached"] ?? selectedRateLimit["limitReached"] + ); - const window5h = { percentUsed: usedPercent5h / 100, resetAt: resetAt5h }; - const window7d = { percentUsed: usedPercent7d / 100, resetAt: resetAt7d }; + const windows: Record = {}; + assignCodexWindows(windows, selectedRateLimit, { + primary: useSparkWindows ? CODEX_SPARK_QUOTA_SESSION : CODEX_WINDOW_SESSION, + secondary: useSparkWindows ? CODEX_SPARK_QUOTA_WEEKLY : CODEX_WINDOW_WEEKLY, + }); + const allWindows: Record = { + ...windows, + }; + + if (sparkRateLimit) { + assignCodexWindows(allWindows, sparkRateLimit, { + primary: CODEX_SPARK_QUOTA_SESSION, + secondary: CODEX_SPARK_QUOTA_WEEKLY, + }); + } + assignCodexWindows(allWindows, normalRateLimit, { + primary: CODEX_WINDOW_SESSION, + secondary: CODEX_WINDOW_WEEKLY, + }); return { - used: worstPercentUsed, + used: Math.round(worstPercentUsed * 100), total: 100, - percentUsed: percentUsedNormalized, + percentUsed: worstPercentUsed, resetAt: getDominantResetAt({ window5h, window7d }), - // Per-window breakdown for the preflight evaluator. Keys match what the - // dashboard renders (session = 5h, weekly = 7d) so user-set cutoffs and - // displayed quotas refer to the same windows. - windows: { - ...(hasPrimary ? { [CODEX_WINDOW_SESSION]: window5h } : {}), - ...(hasSecondary ? { [CODEX_WINDOW_WEEKLY]: window7d } : {}), - }, + // Per-window breakdown for the preflight evaluator. For Spark requests this + // intentionally contains ONLY Spark windows, so Spark exhaustion does not + // preflight-block normal Codex requests (and vice versa). + windows, + allWindows, // Legacy fields preserved for existing consumers (quotaMonitor, cooldown - // computation in accountFallback). These mirror the new windows entries + // computation in accountFallback). These mirror the selected scope entries // but keep the historical names — do not remove without checking callers. window5h, window7d, @@ -348,7 +461,7 @@ export function getCodexQuotaCooldownMs(quota: CodexDualWindowQuota, threshold = * Ensures the next preflight call fetches fresh data. */ export function invalidateCodexQuotaCache(connectionId: string): void { - quotaCache.delete(connectionId); + deleteQuotaCacheForConnection(connectionId); } // ─── Registration ───────────────────────────────────────────────────────────── @@ -360,5 +473,10 @@ export function invalidateCodexQuotaCache(connectionId: string): void { export function registerCodexQuotaFetcher(): void { registerQuotaFetcher("codex", fetchCodexQuota); registerMonitorFetcher("codex", fetchCodexQuota); - registerQuotaWindows("codex", [CODEX_WINDOW_SESSION, CODEX_WINDOW_WEEKLY]); + registerQuotaWindows("codex", [ + CODEX_WINDOW_SESSION, + CODEX_WINDOW_WEEKLY, + CODEX_SPARK_QUOTA_SESSION, + CODEX_SPARK_QUOTA_WEEKLY, + ]); } diff --git a/open-sse/services/codexUsageQuotas.ts b/open-sse/services/codexUsageQuotas.ts new file mode 100644 index 00000000000..6f595854d6a --- /dev/null +++ b/open-sse/services/codexUsageQuotas.ts @@ -0,0 +1,161 @@ +import { + CODEX_SPARK_DISPLAY_NAME, + CODEX_SPARK_QUOTA_SESSION, + CODEX_SPARK_QUOTA_WEEKLY, + isCodexSparkLimitDescriptor, +} from "../config/codexQuotaScopes.ts"; + +type JsonRecord = Record; + +export type CodexUsageQuota = { + used: number; + total: number; + remaining?: number; + resetAt: string | null; + unlimited: boolean; + displayName?: string; +}; + +export function getFieldValue(record: unknown, ...keys: string[]): unknown { + if (!record || typeof record !== "object") return null; + const typed = record as JsonRecord; + for (const key of keys) { + if (typed[key] !== undefined && typed[key] !== null) return typed[key]; + } + return null; +} + +function toRecord(value: unknown): JsonRecord { + return value && typeof value === "object" && !Array.isArray(value) ? (value as JsonRecord) : {}; +} + +function toNumber(value: unknown, fallback = 0): number { + if (typeof value === "number" && Number.isFinite(value)) return value; + if (typeof value === "string" && value.trim().length > 0) { + const parsed = Number(value); + return Number.isFinite(parsed) ? parsed : fallback; + } + return fallback; +} + +function parseResetTime(resetValue: unknown): string | null { + if (!resetValue) return null; + try { + const date = + resetValue instanceof Date + ? resetValue + : typeof resetValue === "number" + ? new Date(resetValue < 1e12 ? resetValue * 1000 : resetValue) + : typeof resetValue === "string" + ? new Date(resetValue) + : null; + if (!date || date.getTime() <= 0) return null; + return date.toISOString(); + } catch { + return null; + } +} + +function parseWindowReset(window: unknown): string | null { + const resetAt = toNumber(getFieldValue(window, "reset_at", "resetAt"), 0); + const resetAfterSeconds = toNumber( + getFieldValue(window, "reset_after_seconds", "resetAfterSeconds"), + 0 + ); + if (resetAt > 0) return parseResetTime(resetAt * 1000); + if (resetAfterSeconds > 0) return parseResetTime(Date.now() + resetAfterSeconds * 1000); + return null; +} + +function buildPercentageQuota(window: JsonRecord, displayName?: string): CodexUsageQuota { + const usedPercent = toNumber(getFieldValue(window, "used_percent", "usedPercent"), 0); + return { + used: usedPercent, + total: 100, + remaining: 100 - usedPercent, + resetAt: parseWindowReset(window), + unlimited: false, + ...(displayName ? { displayName } : {}), + }; +} + +function findCodexSparkRateLimit(data: JsonRecord): JsonRecord { + const additionalRateLimits = getFieldValue( + data, + "additional_rate_limits", + "additionalRateLimits" + ); + if (!Array.isArray(additionalRateLimits)) return {}; + + for (const entryValue of additionalRateLimits) { + const entry = toRecord(entryValue); + if ( + isCodexSparkLimitDescriptor( + getFieldValue(entry, "limit_name", "limitName"), + getFieldValue(entry, "metered_feature", "meteredFeature"), + getFieldValue(entry, "limit_id", "limitId"), + entry["id"], + entry["name"], + entry["title"], + entry["model"], + getFieldValue(entry, "model_id", "modelId") + ) + ) { + return toRecord(getFieldValue(entry, "rate_limit", "rateLimit")); + } + } + return {}; +} + +export function buildCodexUsageQuotas(dataValue: unknown): { + rateLimit: JsonRecord; + quotas: Record; +} { + const data = toRecord(dataValue); + const rateLimit = toRecord(getFieldValue(data, "rate_limit", "rateLimit")); + const quotas: Record = {}; + + const primaryWindow = toRecord(getFieldValue(rateLimit, "primary_window", "primaryWindow")); + if (Object.keys(primaryWindow).length > 0) quotas.session = buildPercentageQuota(primaryWindow); + + const secondaryWindow = toRecord(getFieldValue(rateLimit, "secondary_window", "secondaryWindow")); + if (Object.keys(secondaryWindow).length > 0) + quotas.weekly = buildPercentageQuota(secondaryWindow); + + const codeReviewWindow = toRecord( + getFieldValue( + toRecord(getFieldValue(data, "code_review_rate_limit", "codeReviewRateLimit")), + "primary_window", + "primaryWindow" + ) + ); + if ( + getFieldValue(codeReviewWindow, "used_percent", "usedPercent") !== null || + getFieldValue(codeReviewWindow, "remaining_count", "remainingCount") !== null + ) { + quotas.code_review = buildPercentageQuota(codeReviewWindow); + } + + const sparkRateLimit = findCodexSparkRateLimit(data); + const sparkPrimaryWindow = toRecord( + getFieldValue(sparkRateLimit, "primary_window", "primaryWindow") + ); + if (Object.keys(sparkPrimaryWindow).length > 0) { + quotas[CODEX_SPARK_QUOTA_SESSION] = buildPercentageQuota( + sparkPrimaryWindow, + CODEX_SPARK_DISPLAY_NAME + ); + } + + const sparkSecondaryWindow = toRecord( + getFieldValue(sparkRateLimit, "secondary_window", "secondaryWindow") + ); + if (Object.keys(sparkSecondaryWindow).length > 0) { + quotas[CODEX_SPARK_QUOTA_WEEKLY] = buildPercentageQuota( + sparkSecondaryWindow, + `${CODEX_SPARK_DISPLAY_NAME} Weekly` + ); + } + + return { rateLimit, quotas }; +} diff --git a/open-sse/services/usage.ts b/open-sse/services/usage.ts index fde46bcf142..b8a52aecfeb 100644 --- a/open-sse/services/usage.ts +++ b/open-sse/services/usage.ts @@ -12,6 +12,7 @@ import { toClientAntigravityQuotaModelId, } from "../config/antigravityModelAliases.ts"; import { isUserCallableAgyModelId } from "../config/agyModels.ts"; +import { buildCodexUsageQuotas } from "./codexUsageQuotas.ts"; import { getGlmQuotaUrl } from "../config/glmProvider.ts"; import { getGitHubCopilotInternalUserHeaders } from "../config/providerHeaderProfiles.ts"; import { safePercentage } from "@/shared/utils/formatting"; @@ -2880,80 +2881,7 @@ async function getCodexUsage( const data = await response.json(); - // Parse rate limit info (supports both snake_case and camelCase) - const rateLimit = toRecord(getFieldValue(data, "rate_limit", "rateLimit")); - const primaryWindow = toRecord(getFieldValue(rateLimit, "primary_window", "primaryWindow")); - const secondaryWindow = toRecord( - getFieldValue(rateLimit, "secondary_window", "secondaryWindow") - ); - - // Parse reset times (reset_at is Unix timestamp in seconds) - const parseWindowReset = (window: unknown) => { - const resetAt = toNumber(getFieldValue(window, "reset_at", "resetAt"), 0); - const resetAfterSeconds = toNumber( - getFieldValue(window, "reset_after_seconds", "resetAfterSeconds"), - 0 - ); - if (resetAt > 0) return parseResetTime(resetAt * 1000); - if (resetAfterSeconds > 0) return parseResetTime(Date.now() + resetAfterSeconds * 1000); - return null; - }; - - // Build quota windows - const quotas: Record = {}; - - // Primary window (5-hour) - if (Object.keys(primaryWindow).length > 0) { - const usedPercent = toNumber(getFieldValue(primaryWindow, "used_percent", "usedPercent"), 0); - quotas.session = { - used: usedPercent, - total: 100, - remaining: 100 - usedPercent, - resetAt: parseWindowReset(primaryWindow), - unlimited: false, - }; - } - - // Secondary window (weekly) - if (Object.keys(secondaryWindow).length > 0) { - const usedPercent = toNumber( - getFieldValue(secondaryWindow, "used_percent", "usedPercent"), - 0 - ); - quotas.weekly = { - used: usedPercent, - total: 100, - remaining: 100 - usedPercent, - resetAt: parseWindowReset(secondaryWindow), - unlimited: false, - }; - } - - // Code review rate limit (3rd window — differs per plan: Plus/Pro/Team) - const codeReviewRateLimit = toRecord( - getFieldValue(data, "code_review_rate_limit", "codeReviewRateLimit") - ); - const codeReviewWindow = toRecord( - getFieldValue(codeReviewRateLimit, "primary_window", "primaryWindow") - ); - - // Only include code review quota if the API returned data for it - const codeReviewUsedRaw = getFieldValue(codeReviewWindow, "used_percent", "usedPercent"); - const codeReviewRemainingRaw = getFieldValue( - codeReviewWindow, - "remaining_count", - "remainingCount" - ); - if (codeReviewUsedRaw !== null || codeReviewRemainingRaw !== null) { - const codeReviewUsedPercent = toNumber(codeReviewUsedRaw, 0); - quotas.code_review = { - used: codeReviewUsedPercent, - total: 100, - remaining: 100 - codeReviewUsedPercent, - resetAt: parseWindowReset(codeReviewWindow), - unlimited: false, - }; - } + const { rateLimit, quotas } = buildCodexUsageQuotas(data); return { plan: String(getFieldValue(data, "plan_type", "planType") || "unknown"), diff --git a/package-lock.json b/package-lock.json index 77a475c1998..5cbd0904b10 100644 --- a/package-lock.json +++ b/package-lock.json @@ -17962,9 +17962,9 @@ } }, "node_modules/jsdom/node_modules/undici": { - "version": "7.25.0", - "resolved": "https://registry.npmjs.org/undici/-/undici-7.25.0.tgz", - "integrity": "sha512-xXnp4kTyor2Zq+J1FfPI6Eq3ew5h6Vl0F/8d9XU5zZQf1tX9s2Su1/3PiMmUANFULpmksxkClamIZcaUqryHsQ==", + "version": "7.28.0", + "resolved": "https://registry.npmjs.org/undici/-/undici-7.28.0.tgz", + "integrity": "sha512-cRZYrTDwWznlnRiPjggAGxZXanty6M8RV1ff8Wm4LWXBp7/IG8v5DnOm74DtUBp9OONpK75YlPnIjQqX0dBDtA==", "dev": true, "license": "MIT", "engines": { @@ -21480,9 +21480,9 @@ } }, "node_modules/node-gyp/node_modules/undici": { - "version": "6.26.0", - "resolved": "https://registry.npmjs.org/undici/-/undici-6.26.0.tgz", - "integrity": "sha512-4yqz8a3n5HmGTlsbADNtr/dJlhkh/55Rq798G6ibiULcXbDtaLpTl1pvdqcbFfeoj3iSi52lePFM7h9H21cw/A==", + "version": "6.27.0", + "resolved": "https://registry.npmjs.org/undici/-/undici-6.27.0.tgz", + "integrity": "sha512-YmfV3YnEDzXRC5lZ2jWtWWHKGUm1zIt8AhesR1tens+HTNv+YZlN/dp6G727LOvMJ8xjP9Be7Y2Sdr96LDm+pg==", "dev": true, "license": "MIT", "engines": { diff --git a/package.json b/package.json index 67146c5532a..842c73e56c0 100644 --- a/package.json +++ b/package.json @@ -347,6 +347,12 @@ "vite": "^8.0.16", "protobufjs": "^7.6.3", "@babel/core": "^7.29.6", - "hono": "^4.12.25" + "hono": "^4.12.25", + "jsdom": { + "undici": "^7.28.0" + }, + "node-gyp": { + "undici": "^6.27.0" + } } } diff --git a/scripts/check/check-public-creds.mjs b/scripts/check/check-public-creds.mjs index ff0188897b0..33e17ea1e39 100644 --- a/scripts/check/check-public-creds.mjs +++ b/scripts/check/check-public-creds.mjs @@ -88,8 +88,8 @@ const ENV_KEY_RE = /(clientId|clientSecret|apiKey)Env\s*:/; // TODO(6A.8): Consider tightening CRED_KEY_RE to exclude function-signature contexts — but // that adds complexity; the FP rate is low (1 file). Frozen by file:line:value key. export const KNOWN_LITERAL_CREDS = new Set([ - "open-sse/services/usage.ts:546:minimax", // TODO(6A.8): pre-existing FP — TS fn-param type, not a credential (moved 543→546 by #3838 usage.ts comment) - "open-sse/services/usage.ts:546:minimax-cn", // TODO(6A.8): pre-existing FP — TS fn-param type, not a credential (moved 543→546 by #3838 usage.ts comment) + "open-sse/services/usage.ts:547:minimax", // TODO(6A.8): pre-existing FP — TS fn-param type, not a credential (moved 543→547 by #3838 usage.ts comment + #4293 Codex Spark extraction) + "open-sse/services/usage.ts:547:minimax-cn", // TODO(6A.8): pre-existing FP — TS fn-param type, not a credential (moved 543→547 by #3838 usage.ts comment + #4293 Codex Spark extraction) ]); /** diff --git a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx index dd5674bbbc8..e7488b86676 100644 --- a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx +++ b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.tsx @@ -18,6 +18,8 @@ const QUOTA_LABEL_MAP: Record = { session: "Session", weekly: "Weekly", code_review: "Code Review", + gpt_5_3_codex_spark_session: "GPT-5.3-Codex-Spark", + gpt_5_3_codex_spark_weekly: "GPT-5.3-Codex-Spark Weekly", agentic_request: "Agentic", agentic_request_freetrial: "Agentic (Trial)", credits: "AI Credits", @@ -308,6 +310,7 @@ export function parseQuotaData(provider, data) { Object.entries(data.quotas).forEach(([quotaType, quota]: [string, any]) => { normalizedQuotas.push( normalizeQuotaEntry(quotaType, quota, { + displayName: quota?.displayName, isPercentageOnly: true, }) ); diff --git a/src/sse/services/auth.ts b/src/sse/services/auth.ts index 3cd4c8efe52..7a342ac93d4 100644 --- a/src/sse/services/auth.ts +++ b/src/sse/services/auth.ts @@ -44,7 +44,12 @@ import { PROVIDER_ERROR_TYPES, } from "@omniroute/open-sse/services/errorClassifier.ts"; -import { getCodexModelScope } from "@omniroute/open-sse/executors/codex.ts"; +import { + getCodexModelScope, + getCodexQuotaWindowFilterForModel, + toCodexBaseQuotaWindowName, + toCodexScopedQuotaWindowName, +} from "@omniroute/open-sse/config/codexQuotaScopes.ts"; import { getProviderById, getProviderAlias, @@ -348,7 +353,7 @@ function normalizeCodexWindowName(windowName: unknown): string | null { if (normalized === "weekly (7d)" || normalized === "7d" || normalized === "seven_day") { return "weekly"; } - return normalized; + return toCodexBaseQuotaWindowName(normalized); } function applyCodexWindowPolicy(rawWindows: string[], providerSpecificData: JsonRecord): string[] { @@ -472,7 +477,8 @@ export function resolveQuotaLimitPolicy( export function evaluateQuotaLimitPolicy( provider: string, - connection: ProviderConnectionView + connection: ProviderConnectionView, + requestedModel: string | null = null ): { blocked: boolean; reasons: string[]; resetAt: string | null } { const policy = resolveQuotaLimitPolicy(provider, connection.providerSpecificData); if (!policy.enabled || policy.windows.length === 0) { @@ -483,9 +489,15 @@ export function evaluateQuotaLimitPolicy( const resetCandidates: Array = []; for (const windowName of policy.windows) { - const status = getQuotaWindowStatus(connection.id, windowName, policy.thresholdPercent); + const effectiveWindowName = + provider === "codex" ? toCodexScopedQuotaWindowName(windowName, requestedModel) : windowName; + const status = getQuotaWindowStatus( + connection.id, + effectiveWindowName, + policy.thresholdPercent + ); if (!status?.reachedThreshold) continue; - reasons.push(`${windowName} usage ${Math.round(status.usedPercentage)}%`); + reasons.push(`${effectiveWindowName} usage ${Math.round(status.usedPercentage)}%`); resetCandidates.push(status.resetAt); } @@ -523,49 +535,78 @@ function isRetryableModelLockoutReason(reason: unknown): boolean { : false; } -function getConnectionQuotaHeadroomPercent( +function pushClampedPercentage(percentages: number[], value: number): void { + if (Number.isFinite(value)) { + percentages.push(Math.max(0, Math.min(100, value))); + } +} + +function isResetAtInPast(resetAt: string | null): boolean { + if (!resetAt) return false; + const resetMs = new Date(resetAt).getTime(); + return Number.isFinite(resetMs) && resetMs <= Date.now(); +} + +function collectPolicyQuotaHeadroomPercentages( provider: string, - connection: ProviderConnectionView -): number | null { - const policy = resolveQuotaLimitPolicy(provider, connection.providerSpecificData); + connection: ProviderConnectionView, + policy: QuotaLimitPolicy, + requestedModel: string | null +): number[] { const percentages: number[] = []; const seenWindows = new Set(); - const collectWindow = (windowName: string) => { - const normalizedWindow = normalizeWindowName(windowName); - if (!normalizedWindow || seenWindows.has(normalizedWindow)) return; + for (const windowName of policy.windows) { + const scopedWindow = + provider === "codex" ? toCodexScopedQuotaWindowName(windowName, requestedModel) : windowName; + const normalizedWindow = normalizeWindowName(scopedWindow); + if (!normalizedWindow || seenWindows.has(normalizedWindow)) continue; seenWindows.add(normalizedWindow); const status = getQuotaWindowStatus(connection.id, normalizedWindow, policy.thresholdPercent); - if (!status) return; - percentages.push(Math.max(0, Math.min(100, status.remainingPercentage))); - }; - - for (const windowName of policy.windows) { - collectWindow(windowName); + if (status) pushClampedPercentage(percentages, status.remainingPercentage); } - if (percentages.length > 0) { - return Math.min(...percentages); - } + return percentages; +} +function collectCachedQuotaHeadroomPercentages( + provider: string, + connection: ProviderConnectionView, + requestedModel: string | null +): number[] { const quotaEntry = getQuotaCache(connection.id) as QuotaCacheView | null; const rawQuotas = quotaEntry?.quotas || {}; - for (const quota of Object.values(rawQuotas)) { - if (!quota) continue; - const resetAt = toStringOrNull(quota.resetAt); - if (resetAt) { - const resetMs = new Date(resetAt).getTime(); - if (Number.isFinite(resetMs) && resetMs <= Date.now()) { - continue; - } - } - const remaining = toNumber(quota.remainingPercentage, Number.NaN); - if (Number.isFinite(remaining)) { - percentages.push(Math.max(0, Math.min(100, remaining))); - } + const codexWindowFilter = + provider === "codex" ? getCodexQuotaWindowFilterForModel(requestedModel) : undefined; + const percentages: number[] = []; + + for (const [quotaName, quota] of Object.entries(rawQuotas)) { + if (codexWindowFilter && !codexWindowFilter(quotaName)) continue; + if (!quota || isResetAtInPast(toStringOrNull(quota.resetAt))) continue; + pushClampedPercentage(percentages, toNumber(quota.remainingPercentage, Number.NaN)); } + return percentages; +} + +function getConnectionQuotaHeadroomPercent( + provider: string, + connection: ProviderConnectionView, + requestedModel: string | null = null +): number | null { + const policy = resolveQuotaLimitPolicy(provider, connection.providerSpecificData); + const policyPercentages = collectPolicyQuotaHeadroomPercentages( + provider, + connection, + policy, + requestedModel + ); + const percentages = + policyPercentages.length > 0 + ? policyPercentages + : collectCachedQuotaHeadroomPercentages(provider, connection, requestedModel); + return percentages.length > 0 ? Math.min(...percentages) : null; } @@ -605,11 +646,16 @@ function getConnectionRecencyPenalty(connection: ProviderConnectionView): number function getP2CConnectionScore( provider: string, - connection: ProviderConnectionView + connection: ProviderConnectionView, + requestedModel: string | null = null ): { score: number; quotaHeadroomPercent: number | null } { - const quotaBlocked = evaluateQuotaLimitPolicy(provider, connection).blocked; + const quotaBlocked = evaluateQuotaLimitPolicy(provider, connection, requestedModel).blocked; const quotaExhausted = isAccountQuotaExhausted(connection.id); - const quotaHeadroomPercent = getConnectionQuotaHeadroomPercent(provider, connection); + const quotaHeadroomPercent = getConnectionQuotaHeadroomPercent( + provider, + connection, + requestedModel + ); let quotaPenalty = 0; if (quotaHeadroomPercent !== null) { @@ -636,10 +682,11 @@ function getP2CConnectionScore( function compareP2CConnections( provider: string, a: ProviderConnectionView, - b: ProviderConnectionView + b: ProviderConnectionView, + requestedModel: string | null = null ): number { - const aScore = getP2CConnectionScore(provider, a); - const bScore = getP2CConnectionScore(provider, b); + const aScore = getP2CConnectionScore(provider, a, requestedModel); + const bScore = getP2CConnectionScore(provider, b, requestedModel); if (aScore.score !== bScore.score) { return aScore.score - bScore.score; } @@ -1261,7 +1308,7 @@ export async function getProviderCredentials( if (!bypassQuotaPolicy) { policyEligibleConnections = availableConnections.filter((connection) => { - const evaluation = evaluateQuotaLimitPolicy(provider, connection); + const evaluation = evaluateQuotaLimitPolicy(provider, connection, requestedModel); if (!evaluation.blocked) return true; blockedByPolicy.push({ @@ -1456,7 +1503,9 @@ export async function getProviderCredentials( // Power of Two Choices: sample from the quota-eligible pool and compare // health instead of defaulting to random-first selection. if (candidatePool.length <= 2) { - connection = [...candidatePool].sort((a, b) => compareP2CConnections(provider, a, b))[0]; + connection = [...candidatePool].sort((a, b) => + compareP2CConnections(provider, a, b, requestedModel) + )[0]; } else { const i = parseInt(randomUUID().replace(/-/g, "").substring(0, 8), 16) % candidatePool.length; @@ -1465,7 +1514,7 @@ export async function getProviderCredentials( if (j >= i) j++; const a = candidatePool[i]; const b = candidatePool[j]; - connection = compareP2CConnections(provider, a, b) <= 0 ? a : b; + connection = compareP2CConnections(provider, a, b, requestedModel) <= 0 ? a : b; } } else if (strategy === "random") { // Random: Fisher-Yates-inspired random pick @@ -1659,14 +1708,24 @@ export async function getProviderCredentialsWithQuotaPreflight( // means the same thing as the percentage rendered on the bar. const resolveMinRemainingPercent = (windowName: string | null): number => { if (windowName !== null) { - const override = perConnectionWindowOverrides[windowName]; - if (typeof override === "number") return override; - const providerDefault = providerWindowMap[windowName]; - if (typeof providerDefault === "number") return providerDefault; + const lookupWindowNames = + provider === "codex" + ? uniqueWindows( + [windowName, toCodexBaseQuotaWindowName(windowName)].filter(Boolean) as string[] + ) + : [windowName]; + for (const lookupWindowName of lookupWindowNames) { + const override = perConnectionWindowOverrides[lookupWindowName]; + if (typeof override === "number") return override; + const providerDefault = providerWindowMap[lookupWindowName]; + if (typeof providerDefault === "number") return providerDefault; + } } return defaultThresholdPercent; }; - const preflight = await preflightQuota(provider, connectionId, credentials, { + const preflightCredentials = + requestedModel && provider === "codex" ? { ...credentials, requestedModel } : credentials; + const preflight = await preflightQuota(provider, connectionId, preflightCredentials, { resolveMinRemainingPercent, resolveWarnRemainingPercent: () => warnThresholdPercent, }); @@ -1807,6 +1866,7 @@ export async function markAccountUnavailable( if ( isPerModelQuotaProvider && provider && + provider !== "codex" && model && (status === 404 || status === 429 || status >= 500) ) { diff --git a/tests/integration/resilience-http-e2e.test.ts b/tests/integration/resilience-http-e2e.test.ts index dccee521fa6..5981c8a3896 100644 --- a/tests/integration/resilience-http-e2e.test.ts +++ b/tests/integration/resilience-http-e2e.test.ts @@ -524,6 +524,7 @@ test.before(async () => { resilienceSettings: buildResilienceConfig(), requestRetry: 0, maxRetryIntervalSec: 0, + stickyRoundRobinLimit: 1, requireLogin: false, setupComplete: true, }); diff --git a/tests/unit/account-fallback-service.test.ts b/tests/unit/account-fallback-service.test.ts index c9bcd31648d..f0baf38eff2 100644 --- a/tests/unit/account-fallback-service.test.ts +++ b/tests/unit/account-fallback-service.test.ts @@ -435,6 +435,31 @@ test("hasPerModelQuota returns true for GitHub Copilot provider (#1624)", () => assert.equal(hasPerModelQuota("github", "gpt-5-mini"), true); }); +test("Codex Spark 429s are scoped away from normal Codex models", () => { + const connectionId = `codex-${Date.now()}`; + clearModelLock("codex", connectionId, "gpt-5.3-codex-spark"); + clearModelLock("codex", connectionId, "gpt-5.3-codex"); + + assert.equal(hasPerModelQuota("codex", "gpt-5.3-codex-spark"), true); + assert.equal(shouldMarkAccountExhaustedFrom429("codex", "gpt-5.3-codex-spark"), false); + assert.equal( + lockModelIfPerModelQuota( + "codex", + connectionId, + "gpt-5.3-codex-spark", + RateLimitReason.RATE_LIMIT_EXCEEDED, + 30_000 + ), + true + ); + assert.equal(isModelLocked("codex", connectionId, "gpt-5.3-codex-spark"), true); + assert.equal(isModelLocked("codex", connectionId, "codex-spark-mini"), true); + assert.equal(isModelLocked("codex", connectionId, "gpt-5.3-codex"), false); + + clearModelLock("codex", connectionId, "gpt-5.3-codex-spark"); + clearModelLock("codex", connectionId, "gpt-5.3-codex"); +}); + test("shouldMarkAccountExhaustedFrom429 skips connection-wide lockout for GitHub (#1624)", () => { assert.equal(shouldMarkAccountExhaustedFrom429("github", "gpt-5.1-codex-max"), false); assert.equal(shouldMarkAccountExhaustedFrom429("github", "gpt-5-mini"), false); diff --git a/tests/unit/codex-quota-fetcher.test.ts b/tests/unit/codex-quota-fetcher.test.ts index 8b5ca7d7cdd..088b6934656 100644 --- a/tests/unit/codex-quota-fetcher.test.ts +++ b/tests/unit/codex-quota-fetcher.test.ts @@ -118,6 +118,63 @@ test("fetchCodexQuota parses dual-window usage, forwards workspace headers, and invalidateCodexQuotaCache(connectionId); }); +test("fetchCodexQuota evaluates normal and Spark windows independently by requested model", async () => { + const connectionId = `codex-spark-scope-${Date.now()}`; + let calls = 0; + + registerCodexConnection(connectionId, { + accessToken: "access-token-spark", + }); + + globalThis.fetch = async () => { + calls++; + return new Response( + JSON.stringify({ + rate_limit: { + primary_window: { used_percent: 20, reset_after_seconds: 60 }, + secondary_window: { used_percent: 30, reset_after_seconds: 120 }, + }, + additional_rate_limits: [ + { + limit_id: "codex_bengalfox", + limit_name: "GPT-5.3-Codex-Spark", + metered_feature: "gpt_5_3_codex_spark", + rate_limit: { + primary_window: { used_percent: 100, reset_after_seconds: 300 }, + secondary_window: { used_percent: 40, reset_after_seconds: 600 }, + }, + }, + ], + }), + { + status: 200, + headers: { "content-type": "application/json" }, + } + ); + }; + + const normal = await fetchCodexQuota(connectionId, { requestedModel: "gpt-5.3-codex" }); + const spark = await fetchCodexQuota(connectionId, { requestedModel: "gpt-5.3-codex-spark" }); + + assert.equal(calls, 2, "normal and Spark scopes use separate cache entries"); + assert.equal(normal.percentUsed, 0.3); + assert.equal(normal.windows?.session.percentUsed, 0.2); + assert.equal(normal.windows?.weekly.percentUsed, 0.3); + assert.equal(normal.windows?.gpt_5_3_codex_spark_session, undefined); + assert.equal(spark.percentUsed, 1); + assert.equal(spark.windows?.gpt_5_3_codex_spark_session.percentUsed, 1); + assert.equal(spark.windows?.gpt_5_3_codex_spark_weekly.percentUsed, 0.4); + assert.equal(spark.windows?.session, undefined); + + const sparkCached = await fetchCodexQuota(connectionId, { + requestedModel: "gpt-5.3-codex-spark", + }); + assert.equal(calls, 2); + assert.deepEqual(sparkCached, spark); + + invalidateCodexQuotaCache(connectionId); +}); + test("fetchCodexQuota drops bad credentials after an authorization failure", async () => { const connectionId = `codex-auth-${Date.now()}`; let calls = 0; diff --git a/tests/unit/executor-codex.test.ts b/tests/unit/executor-codex.test.ts index d92b1347871..ee200316e7b 100644 --- a/tests/unit/executor-codex.test.ts +++ b/tests/unit/executor-codex.test.ts @@ -82,6 +82,8 @@ test("Codex helper functions isolate rate-limit scopes and parse quota headers", }); assert.equal(getCodexModelScope("codex-spark-mini"), "spark"); + assert.equal(getCodexModelScope("gpt-5.3-codex-spark"), "spark"); + assert.equal(getCodexModelScope("codex-bengalfox"), "spark"); assert.equal(getCodexModelScope("gpt-5.3-codex"), "codex"); assert.equal(getCodexModelScope("gpt-5.5-xhigh"), "codex"); assert.equal(getCodexUpstreamModel("gpt-5.5-xhigh"), "gpt-5.5"); @@ -113,6 +115,7 @@ test("Codex helper functions isolate rate-limit scopes and parse quota headers", assert.equal(isCodexResponsesWebSocketRequired("gpt-5.5-medium", {}), false); __setCodexWebSocketTransportForTesting(undefined); assert.equal(getCodexRateLimitKey("acct-1", "codex-spark-mini"), "acct-1:spark"); + assert.equal(getCodexRateLimitKey("acct-1", "gpt-5.3-codex-spark"), "acct-1:spark"); assert.equal(quota.usage5h, 100); assert.equal(quota.limit7d, 5000); assert.ok(getCodexResetTime(quota) >= new Date(quota.resetAt7d).getTime()); diff --git a/tests/unit/sse-auth.test.ts b/tests/unit/sse-auth.test.ts index ce049cef19c..3c6b811df1e 100644 --- a/tests/unit/sse-auth.test.ts +++ b/tests/unit/sse-auth.test.ts @@ -738,12 +738,14 @@ test("getProviderCredentials skips codex scope-limited accounts unless suppressi }); const blocked = await auth.getProviderCredentials("codex", null, null, "codex-spark-mini"); + const normalCodex = await auth.getProviderCredentials("codex", null, null, "gpt-5.5"); const bypassed = await auth.getProviderCredentials("codex", null, null, "codex-spark-mini", { allowSuppressedConnections: true, }); assert.equal(blocked.allRateLimited, true); assert.equal(blocked.retryAfter, retryAfter); + assert.equal(normalCodex.connectionId, connection.id); assert.equal(bypassed.connectionId, connection.id); }); @@ -1273,6 +1275,34 @@ test("markAccountUnavailable honors configured api-key rate-limit cooldowns", as assert.equal(result.cooldownMs, 125); }); +test("Codex quota policy keeps normal and Spark windows separate", async () => { + const normalConnection = await seedConnection("codex", { + authType: "oauth", + name: "codex-normal-quota-policy", + apiKey: null, + accessToken: "codex-normal-quota-policy-access", + refreshToken: "codex-normal-quota-policy-refresh", + providerSpecificData: { limitPolicy: { enabled: true, thresholdPercent: 95 } }, + }); + quotaCache.setQuotaCache(normalConnection.id, "codex", { + session: { remainingPercentage: 80, resetAt: futureIso(60_000) }, + weekly: { remainingPercentage: 70, resetAt: futureIso(120_000) }, + gpt_5_3_codex_spark_session: { remainingPercentage: 0, resetAt: futureIso(300_000) }, + }); + + const normalSelected = await auth.getProviderCredentials("codex", null, null, "gpt-5.3-codex"); + const sparkSelected = await auth.getProviderCredentials( + "codex", + null, + null, + "gpt-5.3-codex-spark" + ); + + assert.equal(normalSelected.connectionId, normalConnection.id); + assert.equal(sparkSelected.allRateLimited, true); + assert.match(String(sparkSelected.lastError), /configured quota threshold/i); +}); + test("markAccountUnavailable stores Codex scope-specific cooldowns without a global rate limit", async () => { const connection = await seedConnection("codex", { authType: "oauth", @@ -1292,6 +1322,7 @@ test("markAccountUnavailable stores Codex scope-specific cooldowns without a glo ); const updated = await providersDb.getProviderConnectionById(connection.id); const selected = await auth.getProviderCredentials("codex", null, null, "codex-spark-mini"); + const normalSelected = await auth.getProviderCredentials("codex", null, null, "gpt-5.3-codex"); assert.equal(result.shouldFallback, true); assert.ok(result.cooldownMs > 0); @@ -1299,6 +1330,7 @@ test("markAccountUnavailable stores Codex scope-specific cooldowns without a glo assert.equal(updated.rateLimitedUntil, undefined); assert.ok(updated.providerSpecificData.codexScopeRateLimitedUntil.spark); assert.equal(selected.allRateLimited, true); + assert.equal(normalSelected.connectionId, connection.id); }); test("markAccountUnavailable returns without fallback on bad requests", async () => { @@ -1501,15 +1533,9 @@ test("markAccountUnavailable persists in-memory model lockout for combo transien assert.equal(fallback.isModelLocked("openai", connId, model), false); - await auth.markAccountUnavailable( - connId, - 429, - "Rate limit exceeded", - "openai", - model, - null, - { persistUnavailableState: false } - ); + await auth.markAccountUnavailable(connId, 429, "Rate limit exceeded", "openai", model, null, { + persistUnavailableState: false, + }); assert.equal(fallback.isModelLocked("openai", connId, model), true); @@ -1518,7 +1544,7 @@ test("markAccountUnavailable persists in-memory model lockout for combo transien const otherConn = await seedConnection("openai", { name: "other-conn", }); - assert.equal(fallback.isModelLocked("openai", (otherConn.id as string), model), false); + assert.equal(fallback.isModelLocked("openai", otherConn.id as string, model), false); const updated = await providersDb.getProviderConnectionById(connId); assert.equal(updated.rateLimitedUntil == null, true); diff --git a/tests/unit/tproxy-route.test.ts b/tests/unit/tproxy-route.test.ts index f834799d63b..6877e23ff4e 100644 --- a/tests/unit/tproxy-route.test.ts +++ b/tests/unit/tproxy-route.test.ts @@ -43,11 +43,14 @@ test("POST rejects an out-of-range config with a 400 invalid_request", async () assert.equal(body.error.type, "invalid_request"); }); -test("POST returns a sanitized 500 when the native addon is unavailable (CI)", async () => { +test("POST returns a sanitized 500 when the native addon is unavailable or unprivileged", async () => { const res = await POST(postReq({})); assert.equal(res.status, 500); const body = await res.json(); - assert.match(body.error.message, /native addon|CAP_NET_ADMIN/); + assert.match( + body.error.message, + /native addon|CAP_NET_ADMIN|Operation not permitted|permission|Command failed: ip rule/i + ); assert.ok(!body.error.message.includes("at /"), "no stack trace leaked"); }); diff --git a/tests/unit/tproxy-tls-capture.test.ts b/tests/unit/tproxy-tls-capture.test.ts index 625e8876e08..793fa905ec1 100644 --- a/tests/unit/tproxy-tls-capture.test.ts +++ b/tests/unit/tproxy-tls-capture.test.ts @@ -60,8 +60,13 @@ test("buildForwardHeaders drops hop-by-hop, keeps auth, and pins host", () => { async function startHttpsUpstream(): Promise<{ port: number; close: () => Promise }> { const up = await generateMitmCa("test upstream"); // any self-signed key+cert pair const server = https.createServer({ key: up.key, cert: up.cert }, (req, res) => { - res.writeHead(200, { "content-type": "text/plain" }); - res.end(`decrypted-roundtrip:${req.url ?? ""}`); + const body = `decrypted-roundtrip:${req.url ?? ""}`; + res.writeHead(200, { + "content-type": "text/plain", + "content-length": Buffer.byteLength(body), + connection: "close", + }); + res.end(body); }); await new Promise((r) => server.listen(0, "127.0.0.1", () => r())); const addr = server.address(); @@ -109,12 +114,28 @@ function tlsRequest( client.write(rawRequest); }); const chunks: Buffer[] = []; - client.on("data", (c) => chunks.push(c)); - client.on("end", () => { + let settled = false; + + const settle = (callback: () => void) => { + if (settled) return; + settled = true; + clearTimeout(timeout); client.destroy(); - resolve(Buffer.concat(chunks).toString("utf8")); + callback(); + }; + const resolveWithChunks = () => settle(() => resolve(Buffer.concat(chunks).toString("utf8"))); + const timeout = setTimeout( + () => settle(() => reject(new Error("TLS capture test request timed out"))), + 5_000 + ); + + client.on("data", (chunk) => { + chunks.push(chunk); + const body = Buffer.concat(chunks).toString("utf8"); + if (/decrypted-roundtrip:|502 Bad Gateway/.test(body)) resolveWithChunks(); }); - client.once("error", reject); + client.on("end", resolveWithChunks); + client.once("error", (error) => settle(() => reject(error))); }); } diff --git a/tests/unit/usage-service-hardening.test.ts b/tests/unit/usage-service-hardening.test.ts index a24fe9c9300..532e0063b5f 100644 --- a/tests/unit/usage-service-hardening.test.ts +++ b/tests/unit/usage-service-hardening.test.ts @@ -782,6 +782,23 @@ test("usage service covers Codex, Kiro and Kimi usage parsing and error branches reset_after_seconds: 45, }, }, + additional_rate_limits: [ + { + limit_id: "codex_bengalfox", + limit_name: "GPT-5.3-Codex-Spark", + metered_feature: "gpt_5_3_codex_spark", + rate_limit: { + primary_window: { + used_percent: 90, + reset_after_seconds: 60, + }, + secondary_window: { + used_percent: 20, + reset_after_seconds: 600, + }, + }, + }, + ], }), { status: 200 } ); @@ -848,6 +865,10 @@ test("usage service covers Codex, Kiro and Kimi usage parsing and error branches assert.equal(codex.quotas.session.remaining, 75); assert.equal(codex.quotas.weekly.remaining, 50); assert.equal(codex.quotas.code_review.remaining, 60); + assert.equal(codex.quotas.gpt_5_3_codex_spark_session.remaining, 10); + assert.equal(codex.quotas.gpt_5_3_codex_spark_session.displayName, "GPT-5.3-Codex-Spark"); + assert.equal(codex.quotas.gpt_5_3_codex_spark_weekly.remaining, 80); + assert.equal(codex.quotas.gpt_5_3_codex_spark_weekly.displayName, "GPT-5.3-Codex-Spark Weekly"); const kiroNoArn: any = await usageService.getUsageForProvider({ provider: "kiro",