diff --git a/.env.example b/.env.example index df5703dd701..f98c656048b 100644 --- a/.env.example +++ b/.env.example @@ -1071,6 +1071,16 @@ APP_LOG_TO_FILE=true # Comma-separated data sources. Default: litellm # PRICING_SYNC_SOURCES=litellm +# ═══════════════════════════════════════════════════════════════════════════════ +# 18b. ARENA ELO SYNC +# ═══════════════════════════════════════════════════════════════════════════════ +# Enable auto-updating model intelligence from Arena AI leaderboard ELO scores. +# Used by: src/lib/arenaEloSync.ts +# ARENA_ELO_SYNC_ENABLED=false + +# Sync interval in seconds. Default: 86400 (24 hours). +# ARENA_ELO_SYNC_INTERVAL=86400 + # ═══════════════════════════════════════════════════════════════════════════════ # 19. MODEL SYNC (Dev) # ═══════════════════════════════════════════════════════════════════════════════ diff --git a/.github/FUNDING.yml b/.github/FUNDING.yml new file mode 100644 index 00000000000..1bfb96a016f --- /dev/null +++ b/.github/FUNDING.yml @@ -0,0 +1,5 @@ +# Funding links for OmniRoute — rendered as the "Sponsor" button on GitHub. +# Docs: https://docs.github.com/repositories/managing-your-repositorys-settings-and-features/customizing-your-repository/displaying-a-sponsor-button-in-your-repository +github: diegosouzapw +# Additional platforms (uncomment and fill in before enabling): +# custom: ["https://omniroute.online/donate"] diff --git a/@omniroute/opencode-plugin/README.md b/@omniroute/opencode-plugin/README.md index f5d52c511d2..55329dec8aa 100644 --- a/@omniroute/opencode-plugin/README.md +++ b/@omniroute/opencode-plugin/README.md @@ -18,23 +18,50 @@ This plugin solves that by: ## Install -Once published to npm: +The plugin ships **pre-built inside the `omniroute` npm package** since v3.8.23. +If you have OmniRoute installed, the plugin is already on disk: ```sh -npm install @omniroute/opencode-plugin +# 1. One command — copy the plugin into OpenCode and update opencode.json +omniroute setup opencode --auth + +# 2. Follow the interactive prompt to enter your OmniRoute API key +# 3. Restart OpenCode — /models lists the full live catalog ``` -Until then (or for local development), reference the built artifact directly. Either extract the package into your OpenCode plugins dir and point at the extracted `dist/index.js`: +The `--auth` flag runs `opencode auth login --provider omniroute` automatically. +Use `--base-url` to point at a non-default OmniRoute address: + +```sh +omniroute setup opencode --base-url https://or.example.com --auth +``` + +### What it does + +1. Locates the bundled plugin inside the omniroute installation +2. Copies `dist/` + `package.json` to `~/.config/opencode/plugins/omniroute/` +3. Writes/updates `opencode.json` with the plugin entry (idempotent, replaces legacy entries) +4. (With `--auth`) runs `opencode auth login` so the API key is stored + +Re-run any time to update the plugin or change the base URL. Older entries for +`@omniroute/opencode-provider` or the legacy `opencode-omniroute-auth` package are +automatically cleaned up. + +### Manual install (without omniroute CLI) + +If you cannot run `omniroute setup opencode` (local dev, CI, air-gapped), reference +the built artifact directly: ```sh -# from inside the OmniRoute repo cd @omniroute/opencode-plugin && npm run build && npm pack # then extract into ~/.config/opencode/plugins/omniroute-opencode-plugin/ ``` +And add the entry to `opencode.json` manually (see Quick Start below). + Peer dep: `@opencode-ai/plugin` (managed by your OpenCode install). -## Quick start (single instance) +## Quick start (single instance, manual) ```jsonc // opencode.json @@ -42,7 +69,7 @@ Peer dep: `@opencode-ai/plugin` (managed by your OpenCode install). "$schema": "https://opencode.ai/config.json", "plugin": [ [ - "@omniroute/opencode-plugin", + "./plugins/omniroute-opencode-plugin/dist/index.js", { "providerId": "omniroute", "baseURL": "https://or.example.com", diff --git a/@omniroute/opencode-plugin/src/index.ts b/@omniroute/opencode-plugin/src/index.ts index 8ad06edfe69..6c8c92540b5 100644 --- a/@omniroute/opencode-plugin/src/index.ts +++ b/@omniroute/opencode-plugin/src/index.ts @@ -2552,17 +2552,31 @@ export function createOmniRouteProviderHook( const apiKey = (auth as { key: string }).key; // baseURL resolution: plugin opts first, then credential-attached - // baseURL (auth backends sometimes stash it next to the key). No - // silent default to localhost: a misconfigured plugin should surface - // a clear error, not phantom /v1/models calls. Cast through unknown - // because the Auth union (OAuth | ApiAuth | WellKnownAuth) doesn't - // declare baseURL on any branch — we duck-type it as a defensive - // extension point. + // baseURL (auth backends sometimes stash it next to the key), then the + // provider config itself — a baseURL set via opencode.json provider + // options (or a config hook) lands on `provider.options` and is not + // visible through either of the first two links. No silent default to + // localhost: a misconfigured plugin should surface a clear warning, + // not phantom /v1/models calls. Cast through unknown because the Auth + // union (OAuth | ApiAuth | WellKnownAuth) doesn't declare baseURL on + // any branch — we duck-type it as a defensive extension point. const authBaseURL = (auth as unknown as { baseURL?: unknown }).baseURL; + const providerBaseURL = ( + _provider as { options?: { baseURL?: unknown } } | undefined + )?.options?.baseURL; const baseURL = resolved.baseURL ?? - (typeof authBaseURL === "string" ? authBaseURL : ""); + (typeof authBaseURL === "string" && authBaseURL.length > 0 ? authBaseURL : undefined) ?? + (typeof providerBaseURL === "string" && providerBaseURL.length > 0 + ? providerBaseURL + : undefined) ?? + ""; if (!baseURL) { + console.warn( + `[omniroute-plugin] provider.models(${resolved.providerId}): ` + + `no baseURL resolvable — checked plugin opts, auth.json, and provider config. ` + + `Set baseURL in opencode.json plugin options or run \`opencode connect ${resolved.providerId}\` with a baseURL.` + ); return {}; } diff --git a/CHANGELOG.md b/CHANGELOG.md index 942ca20da9f..a1d7cc206dd 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,74 +2,6 @@ ## [Unreleased] ---- - -## [3.8.22] — 2026-06-11 - -### ✨ Added - -- **MiMoCode free-tier provider** ([#3659] — thanks @pizzav-xyz): new no-auth provider `mimocode` (alias `mcode`) exposing Xiaomi's `mimo-auto` model (1M context) via device-fingerprint bootstrap-JWT auth (`/api/free-ai/bootstrap` → Bearer JWT → `/api/free-ai/openai/chat`). Supports multiple accounts (N fingerprints → round-robin with exponential cooldown), re-bootstrap on 401/403, and cooldown on 429. Reuses a new generic `NoAuthAccountCard` dashboard component (also wired for `opencode`). 22 unit tests; upstream validated live during review. (Maintainer follow-up: added the required `authHeader: "none"` field to the registry entry.) Co-authored with @pizzav-xyz. -- **Prefer Claude Code for unprefixed `claude-*` model IDs** ([#3540] — thanks @Witroch4): opt-in setting (default off) that routes bare `claude-*` model IDs from Claude Code clients through the Claude Code OAuth account instead of requiring a provider prefix. Configurable via the `OMNIROUTE_PREFER_CLAUDE_CODE_FOR_UNPREFIXED_CLAUDE_MODELS` env flag or a dashboard toggle on the Claude provider page; explicit provider prefixes still win. Full layer coverage (resolver + DB setting + zod schemas + types + UI) with 6 tests. Co-authored with @Witroch4. -- **Codex Responses-WebSocket call history** ([#3616] — thanks @kkkayye): Codex `/v1/responses` WebSocket calls are now persisted to request history — success completions plus prepare-failures, upstream WS errors and premature closes — with `sanitizeErrorMessage` applied to the stored error. Two proxy-side integration tests cover the success and failure paths. -- **Obsidian/WebDAV**: add the `/api/v1/webdav` file server (PROPFIND/GET/PUT/DELETE/MKCOL/MOVE, Basic-Auth, path-traversal hardened) so Obsidian mobile can sync the vault (#3485, part 2). Implemented in the custom server layer (`scripts/dev/webdav-handler.mjs`) — intercepted before Next.js to support non-standard HTTP methods (`PROPFIND`, `MKCOL`, `MOVE`, `LOCK`). Reads vault path and credentials (with enc:v1: AES-256-GCM decryption) directly from the SQLite `key_value` table; credentials configured via PR1's `/api/settings/obsidian/webdav` endpoint. 36 TDD unit tests covering traversal guard, constant-time auth, decrypt round-trip, XML generation, and full CRUD cycle. -- **Quota overview**: deactivate/activate an account directly from the quota card header (toggle button) so users can park a near-zero-quota account without navigating to the provider detail page. ([#3675](https://github.com/diegosouzapw/OmniRoute/pull/3675) — thanks @leninejunior) - -### ♻️ Code Quality - -- **providers/[id]**: extract `useProviderConnections`, `useProviderSettings`, `useProviderModels` hooks from the god-component — #3501 Phase 1f. `ProviderDetailPageClient.tsx`: 4,948 → 4,063 LOC (−885 lines). New hooks in `hooks/`: `useProviderConnections.ts` (954 LOC — all connection management, batch ops, proxy/CLIProxyAPI state, batch-test runner with MAX_BULK_IDS chunking), `useProviderSettings.ts` (264 LOC — Codex global service mode + Claude routing preference), `useProviderModels.ts` (155 LOC — model metadata, aliases). Frozen baselines updated. 10 Phase-1f smoke tests; typecheck/cycles/lint green. Co-authored with @oyi77. -- **providers/[id]**: extract `useModelCompatState` hook + model sections (`ModelRow`, `PassthroughModelRow`, `PassthroughModelsSection`, `CustomModelsSection`, `CompatibleModelsSection`) from the god-component — #3501 Phase 1e. `ProviderDetailPageClient.tsx`: 6,838 → 4,922 LOC (−1,916 lines). New leaf `hooks/useModelCompatState.ts` (101 LOC); compat helpers moved to `providerPageHelpers.ts`. Frozen baselines: `providerPageHelpers.ts: 822`. 12 Phase-1e smoke tests; typecheck/cycles/lint green; #3610 auto-hide fix preserved. -- **providers/[id]**: extract `ConnectionRow` (+ `CooldownTimer`/`inferErrorType`/`getStatusPresentation`), `ModelCompatPopover` (+ `recordToHeaderRows`), and `SiliconFlowEndpointModal` from the god-component into `components/` — #3501 Phase 1d. `ProviderDetailPageClient.tsx`: 8,092 → 6,838 LOC (−1,254 lines). Frozen baselines: `ConnectionRow.tsx: 941`. 7 new Phase-1d smoke tests; typecheck/cycles/lint green. -- **providers/[id]: extract AddApiKeyModal + EditConnectionModal (+ WebSessionCredentialGuide) from the god-component into components/** ([#3501] Phase 1c): extracted the two heaviest inline modals — `AddApiKeyModal` (~787-LOC body) and `EditConnectionModal` (~1091-LOC body) — plus shared `WebSessionCredentialGuide` (~103 LOC) into standalone files under `providers/[id]/components/modals/` and `providers/[id]/components/` respectively. Added `ERROR_TYPE_LABELS` and `formatTimeAgo` to `providerPageHelpers.ts` (leaf) so `EditConnectionModal` and `ConnectionRow` share them without cycles. Pruned 14 now-unused imports from the god-component. `ProviderDetailPageClient.tsx`: 9,981 → 8,092 LOC (−1,889 lines). Frozen baselines: `AddApiKeyModal.tsx: 842`, `EditConnectionModal.tsx: 1170`. 6 new Phase-1c smoke tests; all 21 vitest modal tests pass; typecheck/cycles/lint green. -- **refactor: small db/utils cleanup** ([#3523] — thanks @androw): table-driven `compression_analytics` column migration (replaces 17 repeated `ALTER TABLE` calls), a single merged `serializeJsonField` helper in `db/providers.ts` (folded two byte-identical serializers), and removal of the dead no-op `syncProviderDataToCloud`/`getProvidersNeedingRefresh` stubs from `shared/utils/machine.ts` (no remaining callers). Pure refactor; behavior unchanged. -- **Provider-detail god-component decomposition — Phase 2b (remaining shared helpers→leaf)** ([#3501]): extended `providers/[id]/providerPageHelpers.ts` with all remaining pure helpers needed by the heavy modals (`AddApiKeyModal`/`EditConnectionModal`) before they can be extracted. Moved 22 symbols: web-session credential label/hint/check/title helpers; upstream-headers helpers (`upstreamHeadersRecordsEqual`, `headerRowsToRecord`, `effectiveUpstreamHeadersForProtocol`, `anyUpstreamHeadersBadge`, `getProtoSlice`) plus their `HeaderDraftRow`/`CompatModelRow`/`CompatModelMap`/`CompatByProtocolMap` types; Codex consts and helpers (`CODEX_REASONING_STRENGTH_OPTIONS`, `CODEX_ACCOUNT_SERVICE_TIER_VALUES`, `CODEX_GLOBAL_SERVICE_MODE_VALUES`, `getCodexServiceTierLabel`, `normalizeCodexLimitPolicy`, `getCodexRequestDefaults`, `getClaudeCodeCompatibleRequestDefaults`); misc helpers (`compatProtocolLabelKey`, `extractCommandCodeCredentialInput`, `normalizeAndValidateHttpBaseUrl`, `SILICONFLOW_ENDPOINTS`, `CommandCodeAuthFlowState`). New transitive imports wired into the leaf: `MODEL_COMPAT_PROTOCOL_KEYS` (`@/shared/constants/modelCompat`), `CodexServiceTier`/`getCodexRequestDefaults`/`getClaudeCodeCompatibleRequestDefaults` (`@/lib/providers/requestDefaults`), `CodexGlobalServiceMode` (`@/lib/providers/codexFastTier`), `WebSessionCredentialRequirement` (`./webSessionCredentials`). `ProviderDetailPageClient.tsx`: 10,288 → 9,980 LOC. Leaf module: 589 LOC (acyclic). 25-assertion unit test suite passes; smoke test 3/3; no import cycles. Co-authored with @oyi77. -- **Provider-detail god-component decomposition — Phase 2 (helpers→lib)** ([#3501]): extracted the pure shared helpers — `ProviderMessageTranslator`/`LocalProviderMetadata` types, `providerText`/`providerCountText`/`readBooleanToggle`, and the provider base-URL + routing-tag/excluded-model parse/format block — into a new leaf `providers/[id]/providerPageHelpers.ts` (imports only `@/shared`, so the client and modals share them with no import cycle). `ProviderDetailPageClient.tsx`: 10,435 → 10,288 LOC. Unblocks extracting the heavier `AddApiKeyModal`/`EditConnectionModal` (which depend on these helpers) without cycling. The Phase 0 smoke test caught a missing transitive import (`isSelfHostedChatProvider`) at mount — now wired + locked by a new helpers unit test (12 assertions). Co-authored with @oyi77. - -- **#3500 fully resolved** — Hard Rule #5 (no raw SQL in route handlers): all 13 internal offenders migrated to `src/lib/db/` modules across slices (call*logs, usage_history/daily_usage_summary, community_servers, usage_logs, semantic_cache, proxy_logs, skills UPDATE, db-backups). The gate's `KNOWN_RAW_SQL` set is renamed to `EXTERNAL_DB_ALLOWED` (with a back-compat alias) and now holds only the **2 external-DB reads** (`oauth/cursor/auto-import`, `oauth/kiro/auto-import`) — these open \_another app's* SQLite to import credentials, so by design they cannot live in OmniRoute's `db/` domain. The gate still blocks any NEW raw SQL against OmniRoute's DB. -- **chore(db-gate):** reclassify `KNOWN_UNEXPORTED` → `INTENTIONALLY_INTERNAL` in `scripts/check/check-db-rules.mjs` ([#3499]): a full audit of all 25 db modules confirmed each is consumed via direct/dynamic import per Hard Rule #2 ("Never barrel-import from localDb.ts"). The old framing labelled them as "debt", which was misleading — they are the correct pattern. The gate's blocking behaviour is unchanged (a NEW unexported module still fails); only the name, comments, and per-module justifications were updated to reflect audited truth. Four modules flagged `DEAD?` (`compressionScheduler`, `discovery`, `pluginMetrics`, `prompts`) have zero production importers and are documented as schema-reserved. A new regression-guard test (`tests/unit/check-db-rules-classification.test.ts`) asserts every non-dead module in the set has ≥1 real importer, so a future consumer removal surfaces as a test failure requiring explicit reclassification. -- **refactor(db): move `call_logs` aggregations into `callLogStats` db module** ([#3500]): extracted raw SQL from three route handlers (`/api/provider-metrics`, `/api/search/stats`, `/api/v1/search/analytics`) into a new `src/lib/db/callLogStats.ts` domain module (`getProviderMetrics`, `getSearchProviderStats`, `getRecentSearchLogs`, `getSearchAggregateStats`, `getSearchProviderCounts`). First slice of #3500 (call_logs cluster). Behavior unchanged; the three routes are removed from `KNOWN_RAW_SQL` in the gate. Validated with TDD unit tests (6 assertions seeding an in-memory SQLite fixture). -- **refactor(db): move `usage_history`/`daily_usage_summary` SQL into `usageAnalytics` db module** ([#3500]): extracted all inline `db.prepare(...)` calls from two route handlers (`/api/usage/analytics`, `/api/settings/export-json`) into a new `src/lib/db/usageAnalytics.ts` module and extended `src/lib/db/callLogStats.ts` with `getFallbackStats`. New exports: `buildUnifiedSource`, `buildPresetUnifiedSource` (UNION CTE builders), plus 12 typed query functions covering summary, daily, daily-cost, heatmap, model, provider, account, api-key, service-tier, weekly-pattern, and preset-cost aggregations, plus `getAllUsageHistory`/`getAllDomainCostHistory`/`getAllDomainBudgets` for backup export. Second slice of #3500. `KNOWN_RAW_SQL` drops from 12 → 10. Validated with 21 TDD unit tests (`tests/unit/db-usage-analytics-3500.test.ts`) seeding a temp SQLite fixture. - -- **refactor(db): move `community_servers` auth look-up into `gamification` db module** ([#3500]): extracted raw SQL from two federation route handlers (`/api/gamification/federation/leaderboard`, `/api/gamification/federation/score`) into a new `getConnectedServerByKeyHash(apiKeyHash)` function in `src/lib/db/gamification.ts`. Third slice of #3500 (gamification federation cluster). Behavior unchanged; the two routes are removed from `KNOWN_RAW_SQL` in the gate. Validated with TDD unit tests (3 assertions seeding a temp SQLite fixture). - -- **refactor(db): move `skills UPDATE` + `db-backups` SQL into db modules** ([#3500]): fifth slice of #3500. Extracted the dynamic `UPDATE skills SET …` from `src/app/api/skills/[id]/route.ts` into a new `src/lib/db/skills.ts` module (`updateSkill(id, patch)`). The dynamic SET clause is injection-safe: column names are validated against a hard-coded allowlist of known writable columns before being interpolated; unknown keys are silently ignored. Extended `src/lib/db/backup.ts` with three new functions: `exportAllSummaryRows()` (multi-table SELECT for key_value / combos / provider_connections / api_keys, used by exportAll), `getTableNamesFromAdapter()` (sqlite_master introspection via an adapter arg, used by import validation), and `countImportedRows()` (post-import COUNT(\*) per table). The backup domain module is the correct home for sqlite_master introspection — it is not "raw SQL in a route" once moved there. `KNOWN_RAW_SQL` drops by 3 (from 8 → 5). Validated with 11 TDD unit tests (`tests/unit/db-backups-skills-3500.test.ts`). -- **refactor(db): move `usage_logs`/`semantic_cache`/`proxy_logs` SQL into db modules** ([#3500]): extracted raw `db.prepare(...)` SQL from three route handlers (`/api/analytics/auto-routing` → `usageLogs.ts`; `/api/cache/entries` → `semanticCache.ts`; `/api/logs/export` → `proxyLogs.ts`) into new `src/lib/db/` domain modules. New exports: `getAutoRoutingTotalCount`, `getAutoRoutingVariantBreakdown`, `getAutoRoutingTopProviders` (usage_logs), `listSemanticCacheEntries`, `deleteSemanticCacheBySignature`, `deleteSemanticCacheByModel` (semantic_cache), and `exportProxyLogsSince` (proxy_logs). Fourth slice of #3500. `KNOWN_RAW_SQL` drops from 8 → 5. Validated with 13 TDD unit tests (`tests/unit/db-logs-cache-3500.test.ts`) seeding temp SQLite fixtures. - -- **Provider-detail god-component decomposition — Phase 0** ([#3501]): introduced `ProviderDetailPageClient.tsx` and reduced `providers/[id]/page.tsx` to a thin 9-line route wrapper (was 12,882 LOC), following the repo's `*PageClient` convention. Added the first-ever smoke render test for the page (Hard Rule #8) as the safety net every later extraction phase is diffed against. Behavior unchanged; the `check-file-size` ratchet now tracks the extracted client. Foundation for Phases 1–6 (strangler-fig). Thanks @oyi77 for the parallel modularization effort in #3627. -- **Provider-detail god-component decomposition — Phase 1a** ([#3501]): extracted the three self-contained auth-import modal clusters (Codex/Claude/Gemini `Import*AuthModal` + `Apply*AuthModal` + their co-located helpers, ~2,160 LOC) into `providers/[id]/components/modals/`. `ProviderDetailPageClient.tsx` drops 12,882 → 10,719 LOC. Behavior unchanged (smoke test green; clusters had clean `{ onClose, onSuccess }` / inline-prop interfaces). Co-authored with @oyi77. -- **Provider-detail god-component decomposition — Phase 1b** ([#3501]): extracted `EditCompatibleNodeModal` (+ its node/props types) into `providers/[id]/components/modals/`, and moved the shared `CC_COMPATIBLE_DEFAULT_CHAT_PATH` constant into a leaf `providerDetailConstants.ts` so the page client and the modal can both import it without a circular dependency. Also removed two dangling section comments left by Phase 1a. `ProviderDetailPageClient.tsx`: 10,719 → 10,435 LOC. Behavior unchanged (smoke test + a new standalone modal render test green; `check:cycles` clean). Co-authored with @oyi77. - -### 🔧 Bug Fixes - -- **Combos / Auto-Combo: premature context compaction ("agent keeps forgetting things")** ([#3680]): two related context-window bugs fixed. (1) `GET /api/combos/auto` now advertises `context_length` / `max_output_tokens` (MAX across the candidate pool — safe because the auto-combo context pre-filter routes oversized requests to large-window candidates), and the opencode plugin consumes them instead of hardcoding `limit: { context: 0 }` — a zero context silently disables opencode's smart auto-compaction, letting sessions grow until the gateway's destructive history purge kicks in. (2) chatCore's proactive compression for DB combos (incl. quota-shared pools) no longer compresses at `min(...allTargets)`: it now uses the EXECUTING target's own window (`resolveComboContextLimit`), keeping min-of-targets only as a defensive fallback when the current provider/model resolves no specific limit. TDD: 8 server tests (`tests/unit/auto-combo-context-advertising.test.ts`) + 3 plugin tests (`tests/auto-combo-context.test.ts`). -- **Obsidian/WebDAV**: add the `/api/settings/obsidian/webdav` config route (enable/disable vault sync), encrypt WebDAV credentials at rest, and remove the duplicate UI block (#3485, part 1). - -- **OpenCode Free / passthrough**: "Test all models" now respects "Auto-hide failed models" and switches the list to the visible filter so hidden models actually disappear (#3610). Three related bugs fixed: `autoHideFailed` is now threaded from the outer component into `PassthroughModelsSection` via a prop (single shared checkbox); the `/api/models/test-all` request body now includes `autoHideFailed: true` so the server persists the hide; and after the loop, `visibilityFilter` is switched to `"visible"` when ≥1 model was hidden. Two pure-function helpers (`buildPassthroughTestBody`, `shouldSwitchToVisibleFilter`) extracted to `providerPageHelpers.ts` with 7 unit tests. -- **Resilience**: clear stale transient connection cooldowns on startup so a prior unclean crash no longer makes every request time out at 120s after restart (#3625) - -- **fix(home topology): restore live in-flight request pulse** ([#3507]): the animated "pulse" edges in the home Provider Topology panel went dead after PR #3401 unified request visibility, because `activeRequests` was hardcoded to `[]`. Re-wired to `useLiveRequests()` (the existing WebSocket hook on port 20129) so that every pending/running request drives the animation in real time. A pure `selectActiveRequests` mapping helper was extracted to `home/topologyUtils.ts` with 5 unit tests. -- **Electron desktop**: launch the peer-stamping `server-ws.mjs` entrypoint so local-only routes (AgentBridge, MCP, services) no longer return 403 LOCAL_ONLY (#3386) -- **Provider Topology**: stop flagging healthy providers as errored based on stale historical failures; use current request status (#3619) -- **OpenCode Free**: fetch the live model catalog from the provider's `modelsUrl` for the no-auth model picker instead of serving a stale hardcoded list (#3611) -- **Hermes Agent**: honour the `HERMES_HOME` env var when writing/reading the agent config instead of always using `~/.hermes` (#3628). Introduced `getHermesHome()` / `getHermesConfigPath()` helpers (read at call-time) and routed all four hardcoded callsites through them so OmniRoute's config lands in the same directory that the Hermes PowerShell installer configures on Windows. -- **MITM/cert**: remove the duplicated "Command failed:" prefix in system-command error messages ([#3641](https://github.com/diegosouzapw/OmniRoute/issues/3641)): `execFileText` was prepending its own `"Command failed: "` prefix on top of Node's `execFile` error message, which already begins with `"Command failed: "` for non-zero exits. The error message now surfaces Node's message directly (no double prefix), with stderr appended only when non-empty. - -- **fix(reasoning): replay `reasoning_content` on plain DeepSeek turns** ([#3632] — thanks @adivekar-utexas): the reasoning-replay gate previously only fired when an assistant message already carried `reasoning_content`. Plain (non-tool-call) turns whose `reasoning_content` was stripped by the client (e.g. Cursor) were forwarded without it, so DeepSeek V4+ rejected the request with 400 "the reasoning_content in the thinking mode must be passed back". The gate now also covers missing/empty `reasoning_content` on DeepSeek replay targets, injecting the cached reasoning (or the non-Anthropic placeholder) so multi-turn text conversations no longer 400. Fixes #1682. 2 regression tests. -- **fix(kiro): route enterprise IAM Identity Center accounts to their regional endpoint** ([#3631] — thanks @artickc): Kiro/CodeWhisperer access tokens and Q Developer profile ARNs are region-bound, so enterprise IAM Identity Center accounts outside `us-east-1` (e.g. `eu-central-1`) were rejected by the default host. Adds `resolveKiroRegion` (stored region → profileArn region → `us-east-1`) and `kiroRuntimeHost` (regional `q.{region}.amazonaws.com`, legacy `codewhisperer.us-east-1` for the default), routes chat + usage to the regional endpoint, and discovers the region-matched `profileArn` via `ListAvailableProfiles` in a best-effort `postExchange` hook. 9 tests. -- **fix(combo): skip same-provider/connection targets on connection-level errors** ([#3637] — thanks @herjarsa): on connection-level upstream errors (408/500/502/503/504/524), remaining same-`provider:connection` targets in a combo request are now skipped to avoid hammering a known-bad connection, in both the priority and round-robin paths. Adjusted in review to **exclude OmniRoute circuit-breaker-open responses** (503 + `X-OmniRoute-Provider-Breaker` / `provider_circuit_open`) from this skip, preserving the invariant that a breaker-open is an ordinary target failure (the next same-provider target is still tried). Co-authored with @herjarsa. -- **/v1/responses**: detect stream readiness for tool-call-only and `object`-less chunks so Codex-shaped (reasoning + tools) requests no longer fail with "Stream ended before producing useful content" (#3612) -- **RTL locales (ar/fa/he/ur)**: use logical CSS direction utilities for the sidebar and key overlays so the layout mirrors correctly under `dir=rtl` (#3541, partial — core layout) -- **Kiro/AWS auto-import**: set a descriptive account name and dedupe by `profileArn` so imports no longer create nameless duplicate "OAuth Account" rows (#3615) -- **fix(guardrails):** the `/api/guardrails/test` route now validates its body through `validateBody()` (Zod) instead of parsing raw JSON directly, aligning it with the repo-wide input-validation pattern (Hard Rule #7). ([#3621](https://github.com/diegosouzapw/OmniRoute/pull/3621) — thanks @diegosouzapw) -- **fix(dashboard): bulk provider connection actions** — close audit, API, and UX gaps in the batch activate/deactivate flow: register the `provider.credentials.batch_updated` event in `HIGH_LEVEL_ACTIONS` and `ACTIVITY_ICONS` (was silently dropped from the Activity feed); fix `/api/providers` PATCH to return `warn` status when `notFound` is non-empty instead of always `success`; `/api/providers/test-batch` empty-result early-return now includes a `summary` so stale-ID mode reports to the user; bulk activate/deactivate chunks selection by 100 to avoid the Zod 400 cap on large provider accounts. ([#3673](https://github.com/diegosouzapw/OmniRoute/pull/3673) — thanks @leninejunior) - -### 📝 Maintenance - -- **fix(ci):** increase the `execFileSync` `maxBuffer` in `validate-pack-artifact` so the npm-pack inventory no longer overflows on large tarballs during release validation — follow-up to the v3.8.21 pack-artifact hotfix. ([#3622](https://github.com/diegosouzapw/OmniRoute/pull/3622) — thanks @diegosouzapw) - ---- - -## [3.8.21] — 2026-06-11 - ### ✨ Added - **feat(cli):** `omniroute autostart` now accepts the shorthand the headless / `omniroute serve` path was missing — `omniroute autostart on` / `... true` (aliases of `enable`), `... off` / `... false` (aliases of `disable`), a new `... toggle`, and a default `... status` (bare `omniroute autostart` is a safe read-only). Previously autostart could only be toggled from the tray (`serve --tray`) or the Electron Appearance tab, so a plain `omniroute serve` user had no way to enable it. (The cross-platform launchd/systemd/registry logic is unchanged — this only wires the ergonomic CLI surface.) ([#3331](https://github.com/diegosouzapw/OmniRoute/issues/3331) — thanks @uniQta) @@ -78,7 +10,6 @@ - **refactor(chatCore):** extract the chatCore request phases — idempotency check, semantic cache check, common request sanitization, and memory/skills injection — into dedicated `open-sse/handlers/chatCore/` modules (`idempotency.ts`, `semanticCache.ts`, `sanitization.ts`, `memorySkillsInjection.ts`), slimming the monolithic handler with no behavior change. (Maintainer follow-up: re-derive `idempotencyKey` at the Phase 9.2 save site after the check moved into the module, fixing a `ReferenceError` on successful non-cached responses.) ([#3598](https://github.com/diegosouzapw/OmniRoute/pull/3598) — thanks @oyi77) - **docs(opencode-provider):** soft-deprecate `@omniroute/opencode-provider` in favour of `@omniroute/opencode-plugin`. The provider package writes a **static** model list to `opencode.json` that drifts behind the live OmniRoute catalog, whereas the plugin fetches `/v1/models` at OpenCode startup. The package keeps working (no code/behavior change), but its npm description and README now carry a deprecation banner with the one-line migration, and a guard test pins the notice. ([#3419](https://github.com/diegosouzapw/OmniRoute/issues/3419) — thanks @herjarsa) -- **chore(review):** pre-release hardening from a multi-reviewer `/review-reviews` battery over the v3.8.21 diff (7 Opus reviewers; zero blocker/high). Resolved findings: npm tarball no longer ships co-located test files (`files[]` negations + reconciled `.npmignore`; the #3578 closure gate now asserts the real `npm pack` output in both directions); `getSanitizedCachedProviderLimitsMap` scopes its connection scan to antigravity/agy instead of decrypting every active connection on each dashboard poll; the Antigravity quota-tier remap (`toClientAntigravityQuotaModelId`) is centralized in `antigravityModelAliases.ts` (was an inline if-ladder in `usage.ts`); the chatCore idempotency check returns its resolved key so the save site reuses a single derivation; and new tests pin the chatCore extracted modules, the Antigravity `usage_history` fallback contract, the reasoning-wrapper prefix-preservation heuristic, the Antigravity SSE `markdown` branch, and the upstream-ca/test no-persist guarantee. (Live-verified that agy consumer tokens are accepted by the non-daily `cloudcode-pa` host used by `retrieveUserQuota`, so #3604 is not agy-host-limited.) ### 🔧 Bug Fixes @@ -97,8 +28,6 @@ - **fix(antigravity):** the Antigravity/agy Gemini 3.5 Flash catalog now exposes clean public tier IDs (`gemini-3.5-flash-low`/`-medium`/`-high`, matching Antigravity 2.0.4's Low/Medium/High selector) and maps them to the live upstream IDs at the executor boundary, instead of the old confusing `-preview`/`-agent` names. Antigravity model-id normalization moved out of the global model resolver into the executor so client-visible IDs are no longer rewritten before account/credential routing and logging. (Maintainer follow-up: kept `gemini-3.5-flash-preview` as a hidden backward-compat alias routing to the High tier so saved combos/configs keep working; live-validated the tier set via the `agy` CLI catalog.) ([#3603](https://github.com/diegosouzapw/OmniRoute/pull/3603) — thanks @dhaern) - **fix(usage):** Antigravity/agy Provider Limits now report accurate consumption — `retrieveUserQuota` (live usage) is preferred over the `fetchAvailableModels` catalog view (which keeps reporting full buckets after real usage), with a local `usage_history` fallback for buckets that are only catalog-visible; cached entries are sanitized so retired upstream IDs are not re-exposed, and a deduplicated post-usage refresh keeps the dashboard fresh after each request. (Maintainer follow-up: the post-usage refresh is decoupled through a lightweight `usageEvents` bus so `usageHistory` no longer imports `providerLimits`/the executors graph, keeping the `typecheck:core` surface stable.) ([#3604](https://github.com/diegosouzapw/OmniRoute/pull/3604) — thanks @dhaern) - **fix(gemini):** textual reasoning wrappers emitted as assistant prose (``/``/``/``, including malformed/open tags like ` `modelContextLengths` > static map) so passthrough models outside the legacy 8-model map no longer silently truncate to OpenCode's 128K default. ([#3298](https://github.com/diegosouzapw/OmniRoute/pull/3298) — thanks @herjarsa / @diegosouzapw) - **fix(dev):** auto-rebuild `better-sqlite3` on a Node ABI mismatch at `npm run dev` startup (nvm 22↔24) — dev-only, no-op on the healthy path, unrelated errors not swallowed. ([#3301](https://github.com/diegosouzapw/OmniRoute/pull/3301) — thanks @zhiru) @@ -417,18 +346,18 @@ Thanks to everyone whose work landed in v3.8.14: Thanks to everyone whose work landed in v3.8.13: -| Contributor | PRs / Issues | -| -------------------------------------------------------------- | ------------------------------------------------------------------------------------ | -| [@zhiru](https://github.com/zhiru) | #3300, #3306, #3307 / #3310, #3309, #3301, #3311, #3320, #3319, #3316 | -| [@tycronk20](https://github.com/tycronk20) | #3317, #3318 | -| [@Vinayrnani](https://github.com/Vinayrnani) | #3267 | -| [@oyi77](https://github.com/oyi77) | #3292 (closes #3070), #3322 | -| [@onizukashonan14-png](https://github.com/onizukashonan14-png) | #3296 | -| [@uniQta](https://github.com/uniQta) | #3290, #3295 | -| [@wilsonicdev](https://github.com/wilsonicdev) | #3297 | -| [@herjarsa](https://github.com/herjarsa) | #3298, #3304 | -| [@mikmaneggahommie](https://github.com/mikmaneggahommie) | reported the Completions.me rickroll (discussion #3293) | -| [@diegosouzapw](https://github.com/diegosouzapw) | maintainer — #3299, #3302, #3303; co-author on #3292 / #3306 / #3298 / #3304 / #3309 | +| Contributor | PRs / Issues | +| --- | --- | +| [@zhiru](https://github.com/zhiru) | #3300, #3306, #3307 / #3310, #3309, #3301, #3311, #3320, #3319, #3316 | +| [@tycronk20](https://github.com/tycronk20) | #3317, #3318 | +| [@Vinayrnani](https://github.com/Vinayrnani) | #3267 | +| [@oyi77](https://github.com/oyi77) | #3292 (closes #3070), #3322 | +| [@onizukashonan14-png](https://github.com/onizukashonan14-png) | #3296 | +| [@uniQta](https://github.com/uniQta) | #3290, #3295 | +| [@wilsonicdev](https://github.com/wilsonicdev) | #3297 | +| [@herjarsa](https://github.com/herjarsa) | #3298, #3304 | +| [@mikmaneggahommie](https://github.com/mikmaneggahommie) | reported the Completions.me rickroll (discussion #3293) | +| [@diegosouzapw](https://github.com/diegosouzapw) | maintainer — #3299, #3302, #3303; co-author on #3292 / #3306 / #3298 / #3304 / #3309 | --- @@ -470,14 +399,14 @@ Thanks to everyone whose work landed in v3.8.13: Thanks to everyone whose work landed in v3.8.12: -| Contributor | PRs / Issues | -| ------------------------------------------------ | --------------------------------------------------------------------------------- | -| [@oyi77](https://github.com/oyi77) | #3250, #3259, #3280, #3285, #3286 | -| [@wilsonicdev](https://github.com/wilsonicdev) | #3249, #3268, #3282 / #3283 (co-author, #3247 diagnosis), #3287 | -| [@strangersp](https://github.com/strangersp) | #3261 | -| [@MikeTuev](https://github.com/MikeTuev) | #3248 | -| [@leninejunior](https://github.com/leninejunior) | #3271 | -| [@zhiru](https://github.com/zhiru) | #3274 | +| Contributor | PRs / Issues | +| --- | --- | +| [@oyi77](https://github.com/oyi77) | #3250, #3259, #3280, #3285, #3286 | +| [@wilsonicdev](https://github.com/wilsonicdev) | #3249, #3268, #3282 / #3283 (co-author, #3247 diagnosis), #3287 | +| [@strangersp](https://github.com/strangersp) | #3261 | +| [@MikeTuev](https://github.com/MikeTuev) | #3248 | +| [@leninejunior](https://github.com/leninejunior) | #3271 | +| [@zhiru](https://github.com/zhiru) | #3274 | | [@diegosouzapw](https://github.com/diegosouzapw) | maintainer — #3256, #3263, #3270, #3275, #3277, #3278, #3279, #3281, #3284, #3289 | --- @@ -526,26 +455,26 @@ Thanks to everyone whose work landed in v3.8.12: Thanks to everyone whose work landed in v3.8.11: -| Contributor | PRs / Issues | -| -------------------------------------------------- | ----------------------------------------------- | -| [@wilsonicdev](https://github.com/wilsonicdev) | #3189, #3201, #3203, #3204, #3232, #3240, #3241 | -| [@pizzav-xyz](https://github.com/pizzav-xyz) | #3170, #3171, #3172 | -| [@zhiru](https://github.com/zhiru) | #3185, #3195 | -| [@oyi77](https://github.com/oyi77) | #3217 | -| [@miracuves](https://github.com/miracuves) | #3116, #3226 | -| [@ngocquynh85](https://github.com/ngocquynh85) | #3205, #3214, #3215 | -| [@xz-dev](https://github.com/xz-dev) | #3188 | -| [@bypanghu](https://github.com/bypanghu) | #3191 | -| [@juandisay](https://github.com/juandisay) | #3206 | -| [@tjengbudi](https://github.com/tjengbudi) | #3197, #3198, #3199 | -| [@naimo84](https://github.com/naimo84) | #3151 | -| [@yeardie](https://github.com/yeardie) | #3025 | -| [@pulyankote](https://github.com/pulyankote) | #3202 | -| [@YoursSweetDom](https://github.com/YoursSweetDom) | #3180 | -| [@Guru01100101](https://github.com/Guru01100101) | #3091 | -| [@androw](https://github.com/androw) | #3167 (co-author) | -| [@ibanunmangun](https://github.com/ibanunmangun) | #3193 (co-author) | -| [@diegosouzapw](https://github.com/diegosouzapw) | maintainer — #3187, #3200, issue-fix batches | +| Contributor | PRs / Issues | +| --- | --- | +| [@wilsonicdev](https://github.com/wilsonicdev) | #3189, #3201, #3203, #3204, #3232, #3240, #3241 | +| [@pizzav-xyz](https://github.com/pizzav-xyz) | #3170, #3171, #3172 | +| [@zhiru](https://github.com/zhiru) | #3185, #3195 | +| [@oyi77](https://github.com/oyi77) | #3217 | +| [@miracuves](https://github.com/miracuves) | #3116, #3226 | +| [@ngocquynh85](https://github.com/ngocquynh85) | #3205, #3214, #3215 | +| [@xz-dev](https://github.com/xz-dev) | #3188 | +| [@bypanghu](https://github.com/bypanghu) | #3191 | +| [@juandisay](https://github.com/juandisay) | #3206 | +| [@tjengbudi](https://github.com/tjengbudi) | #3197, #3198, #3199 | +| [@naimo84](https://github.com/naimo84) | #3151 | +| [@yeardie](https://github.com/yeardie) | #3025 | +| [@pulyankote](https://github.com/pulyankote) | #3202 | +| [@YoursSweetDom](https://github.com/YoursSweetDom) | #3180 | +| [@Guru01100101](https://github.com/Guru01100101) | #3091 | +| [@androw](https://github.com/androw) | #3167 (co-author) | +| [@ibanunmangun](https://github.com/ibanunmangun) | #3193 (co-author) | +| [@diegosouzapw](https://github.com/diegosouzapw) | maintainer — #3187, #3200, issue-fix batches | --- @@ -761,9 +690,9 @@ And thank you to the OmniRoute community for the bug reports, reproductions, and refreshed siblings concurrently, so Auth0 revoked the whole token family (`openai/codex#9648`) and every account but the last died with `[403] `. The quota path now skips proactive refresh for - rotating providers (`rotationGroupFor`) and reuses the current access*token, + rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two \_queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -826,7 +755,7 @@ And thank you to the OmniRoute community for the bug reports, reproductions, and via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed -back to the API` (its DeepSeek-thinking upstream is not detectable from the + back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -859,7 +788,7 @@ back to the API` (its DeepSeek-thinking upstream is not detectable from the instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' -must be a response to a preceding message with 'tool_calls'` when a Codex + must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -901,7 +830,7 @@ must be a response to a preceding message with 'tool_calls'` when a Codex - **sse/chatCore:** the heap-pressure guard now auto-calibrates its threshold to 85% of the live V8 heap ceiling (floor 400 MB) instead of a fixed 200 MB that sat below the app's ~260 MB baseline and returned `503 Service temporarily unavailable due to -resource pressure` for every request once the heap warmed up. It now tracks + resource pressure` for every request once the heap warmed up. It now tracks `--max-old-space-size` across 1 GB / 2 GB / large VPS; `HEAP_PRESSURE_THRESHOLD_MB` still overrides. (#3052) - **proxy:** fail closed for OAuth usage-account proxies (#3051 — thanks @terence71-glitch) @@ -916,33 +845,33 @@ resource pressure` for every request once the heap warmed up. It now tracks A special thanks to everyone who contributed to this release — 746 commits since `v3.8.7`: -| Contributor | PRs / Contribution | -| -------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------- | -| [@diegosouzapw](https://github.com/diegosouzapw) | maintainer — AgentBridge, Traffic Inspector, Quota Share Engine, Nav Restructure, Plugins integration, releases & upstream ports | -| [@oyi77](https://github.com/oyi77) | #2913, #2947, #2954, #2978, #3015, #3018, #3039, #3041, #3045, #3046, #3049 | -| [@terence71-glitch](https://github.com/terence71-glitch) | #2956, #2960, #2963, #2984, #3000, #3006, #3012, #3048, #3051 | -| [@soyelmismo](https://github.com/soyelmismo) | #2951, #2965, #2973 | -| [@branben](https://github.com/branben) | #2958, #2959 | -| [@makcimbx](https://github.com/makcimbx) | #2937, #2938 | -| [@guanbear](https://github.com/guanbear) | #2931, #3031 | -| [@Lion-killer](https://github.com/Lion-killer) | #2981, #2988 | -| [@JxnLexn](https://github.com/JxnLexn) | per-API-key stream default mode | -| [@androw](https://github.com/androw) | #3017 | -| [@xz-dev](https://github.com/xz-dev) | #2975, #3064 | -| [@S0yora](https://github.com/S0yora) | #2964 | -| [@NekoMonci12](https://github.com/NekoMonci12) | #3008 | -| [@Tentoxa](https://github.com/Tentoxa) | #3010 | -| [@ReqX](https://github.com/ReqX) | #2957 | -| [@NomenAK](https://github.com/NomenAK) | #2943 | -| [@charithharshana](https://github.com/charithharshana) | #2940 | -| [@dhaern](https://github.com/dhaern) | #2927 | -| [@dangeReis](https://github.com/dangeReis) | #3021 | -| [@bobbyunknown](https://github.com/bobbyunknown) | #3029 | -| [@CitrusIce](https://github.com/CitrusIce) | #3035, #3058 | -| [@wussh](https://github.com/wussh) | #3036 | -| [@Chewji9875](https://github.com/Chewji9875) | #3037 | -| [@herjarsa](https://github.com/herjarsa) | #3043 | -| [@freefrank](https://github.com/freefrank) | #3066 (reported the Docker build failure) | +| Contributor | PRs / Contribution | +| --- | --- | +| [@diegosouzapw](https://github.com/diegosouzapw) | maintainer — AgentBridge, Traffic Inspector, Quota Share Engine, Nav Restructure, Plugins integration, releases & upstream ports | +| [@oyi77](https://github.com/oyi77) | #2913, #2947, #2954, #2978, #3015, #3018, #3039, #3041, #3045, #3046, #3049 | +| [@terence71-glitch](https://github.com/terence71-glitch) | #2956, #2960, #2963, #2984, #3000, #3006, #3012, #3048, #3051 | +| [@soyelmismo](https://github.com/soyelmismo) | #2951, #2965, #2973 | +| [@branben](https://github.com/branben) | #2958, #2959 | +| [@makcimbx](https://github.com/makcimbx) | #2937, #2938 | +| [@guanbear](https://github.com/guanbear) | #2931, #3031 | +| [@Lion-killer](https://github.com/Lion-killer) | #2981, #2988 | +| [@JxnLexn](https://github.com/JxnLexn) | per-API-key stream default mode | +| [@androw](https://github.com/androw) | #3017 | +| [@xz-dev](https://github.com/xz-dev) | #2975, #3064 | +| [@S0yora](https://github.com/S0yora) | #2964 | +| [@NekoMonci12](https://github.com/NekoMonci12) | #3008 | +| [@Tentoxa](https://github.com/Tentoxa) | #3010 | +| [@ReqX](https://github.com/ReqX) | #2957 | +| [@NomenAK](https://github.com/NomenAK) | #2943 | +| [@charithharshana](https://github.com/charithharshana) | #2940 | +| [@dhaern](https://github.com/dhaern) | #2927 | +| [@dangeReis](https://github.com/dangeReis) | #3021 | +| [@bobbyunknown](https://github.com/bobbyunknown) | #3029 | +| [@CitrusIce](https://github.com/CitrusIce) | #3035, #3058 | +| [@wussh](https://github.com/wussh) | #3036 | +| [@Chewji9875](https://github.com/Chewji9875) | #3037 | +| [@herjarsa](https://github.com/herjarsa) | #3043 | +| [@freefrank](https://github.com/freefrank) | #3066 (reported the Docker build failure) | A special thanks to everyone who contributed code, reviews, and tests for this release: @androw, @bobbyunknown, @branben, @charithharshana, @Chewji9875, @CitrusIce, @dangeReis, @dhaern, @diegosouzapw, @freefrank, @guanbear, @herjarsa, @JxnLexn, @Lion-killer, @makcimbx, @NekoMonci12, @NomenAK, @oyi77, @ReqX, @S0yora, @soyelmismo, @Tentoxa, @terence71-glitch, @wussh, @xz-dev @@ -1047,7 +976,7 @@ A special thanks to everyone who contributed code, reviews, and tests for this r - **warning-cleanup:** relax node engine constraint to `>=22.0.0` and clean dependencies (keeping `marked-terminal` to prevent TUI REPL crash) (#2792 — thanks @oyi77) - **combo:** normalize upstream Headers into a plain object before classification to avoid Node 24 / undici cross-instance `Cannot read private member #headers` crash on combo failover (#2751) - **translator:** silently drop `tool_search` built-in tool type instead of returning 400 — newer Codex clients send `tool_search` as a Responses API built-in with no Chat Completions equivalent (#2766) -- **usage:** un-invert GitHub Copilot Free / limited plan quota — `limited_user_quotas` is the _remaining_ count, not used, so the dashboard now shows 100% when the quota is untouched and 0% when fully exhausted (#2876 — thanks @androw) +- **usage:** un-invert GitHub Copilot Free / limited plan quota — `limited_user_quotas` is the *remaining* count, not used, so the dashboard now shows 100% when the quota is untouched and 0% when fully exhausted (#2876 — thanks @androw) - **fix(cli):** register openclaw in the CLI tool-detector so it appears in `omniroute status` alongside its existing API and config support ([#2833](https://github.com/diegosouzapw/OmniRoute/issues/2833)) - **oauth (windsurf):** hotfix Windsurf login — drop the dead PKCE flow and promote the import-token flow as the default ([#2884](https://github.com/diegosouzapw/OmniRoute/pull/2884) — thanks @yunaamelia) - **antigravity:** normalize textual SSE tool calls and classify Gemini Antigravity resource exhaustion as a model lockout instead of a connection failure ([#2828](https://github.com/diegosouzapw/OmniRoute/pull/2828) — thanks @Ardem2025) @@ -1082,33 +1011,33 @@ A special thanks to everyone who contributed code, reviews, and tests for this r A special thanks to everyone who contributed to this release. Ranked by commits since `v3.8.6` (105 commits total): -| Contributor | Commits | PRs | -| ---------------------------------------------------------- | ------: | ----------------------------------------------- | -| [@diegosouzapw](https://github.com/diegosouzapw) | 38 | maintainer — releases, upstream ports & fixes | -| [@oyi77](https://github.com/oyi77) | 10 | #2887, #2862, #2866, #2837, #2885, #2792, #2793 | -| [@yunaamelia](https://github.com/yunaamelia) | 7 | #2884 | -| [@herjarsa](https://github.com/herjarsa) | 6 | #2868, #2886, #2865, #2860, #2857, #2801 | -| [@leninejunior](https://github.com/leninejunior) | 4 | #2818, #2824, #2825, #2816 | -| [@jeferssonlemes](https://github.com/jeferssonlemes) | 3 | #2791, #2802, #2815, #2817 | -| [@rdself](https://github.com/rdself) | 3 | #2874, #2875, #2880 | -| Dmitry Kuznetsov | 3 | textual tool-call & lockout hardening | -| [@apoapostolov](https://github.com/apoapostolov) | 2 | #2799, #2800 | -| [@unitythemaker](https://github.com/unitythemaker) | 2 | #2904 | -| Nikolay Alafuzov | 2 | reasoning interleaved gating | -| [@Tushar49](https://github.com/Tushar49) | 2 | #2854, #2855, #2807 | -| [@guanbear](https://github.com/guanbear) | 2 | #2908 | -| [@soyelmismo](https://github.com/soyelmismo) | 2 | #2903, #2842 | -| [@RajvardhanPatil07](https://github.com/RajvardhanPatil07) | 1 | #2861 | -| [@mugnimaestra](https://github.com/mugnimaestra) | 1 | #2888 | -| [@dhaern](https://github.com/dhaern) | 1 | #2878 | -| [@hartmark](https://github.com/hartmark) | 1 | #2795, #2771 | -| [@marchlhw](https://github.com/marchlhw) | 1 | #2821 | -| [@alltomatos](https://github.com/alltomatos) | 1 | i18n pt-BR | -| [@akarray](https://github.com/akarray) | 1 | #2796 | -| [@gogones](https://github.com/gogones) | 1 | #2845 | -| [@disonjer](https://github.com/disonjer) | 1 | #2840 | -| [@nickwizard](https://github.com/nickwizard) | 1 | #2841 | -| [@levonk](https://github.com/levonk) | 1 | #2806 | +| Contributor | Commits | PRs | +| --- | ---: | --- | +| [@diegosouzapw](https://github.com/diegosouzapw) | 38 | maintainer — releases, upstream ports & fixes | +| [@oyi77](https://github.com/oyi77) | 10 | #2887, #2862, #2866, #2837, #2885, #2792, #2793 | +| [@yunaamelia](https://github.com/yunaamelia) | 7 | #2884 | +| [@herjarsa](https://github.com/herjarsa) | 6 | #2868, #2886, #2865, #2860, #2857, #2801 | +| [@leninejunior](https://github.com/leninejunior) | 4 | #2818, #2824, #2825, #2816 | +| [@jeferssonlemes](https://github.com/jeferssonlemes) | 3 | #2791, #2802, #2815, #2817 | +| [@rdself](https://github.com/rdself) | 3 | #2874, #2875, #2880 | +| Dmitry Kuznetsov | 3 | textual tool-call & lockout hardening | +| [@apoapostolov](https://github.com/apoapostolov) | 2 | #2799, #2800 | +| [@unitythemaker](https://github.com/unitythemaker) | 2 | #2904 | +| Nikolay Alafuzov | 2 | reasoning interleaved gating | +| [@Tushar49](https://github.com/Tushar49) | 2 | #2854, #2855, #2807 | +| [@guanbear](https://github.com/guanbear) | 2 | #2908 | +| [@soyelmismo](https://github.com/soyelmismo) | 2 | #2903, #2842 | +| [@RajvardhanPatil07](https://github.com/RajvardhanPatil07) | 1 | #2861 | +| [@mugnimaestra](https://github.com/mugnimaestra) | 1 | #2888 | +| [@dhaern](https://github.com/dhaern) | 1 | #2878 | +| [@hartmark](https://github.com/hartmark) | 1 | #2795, #2771 | +| [@marchlhw](https://github.com/marchlhw) | 1 | #2821 | +| [@alltomatos](https://github.com/alltomatos) | 1 | i18n pt-BR | +| [@akarray](https://github.com/akarray) | 1 | #2796 | +| [@gogones](https://github.com/gogones) | 1 | #2845 | +| [@disonjer](https://github.com/disonjer) | 1 | #2840 | +| [@nickwizard](https://github.com/nickwizard) | 1 | #2841 | +| [@levonk](https://github.com/levonk) | 1 | #2806 | _Reviews & additional contributions: @androw, @Ardem2025, @InkshadeWoods._ A special thanks to everyone who contributed code, reviews, and tests for this release: @@ -1146,6 +1075,7 @@ A special thanks to everyone who contributed code, reviews, and tests for this r A special thanks to everyone who contributed code, reviews, and tests for this release: @akarray, @hartmark, @hijak, @JxnLexn, @kjhq, @rdself, @thanet-s + --- ## [3.8.4] — 2026-05-26 diff --git a/README.md b/README.md index 50e484d2b64..467e920a1e1 100644 --- a/README.md +++ b/README.md @@ -1013,6 +1013,14 @@ Special thanks to **[RTK - Rust Token Killer](https://github.com/rtk-ai/rtk)** b Special thanks to **[Troglodita](https://github.com/leninejunior/troglodita)** by **[Lenine Júnior](https://github.com/leninejunior)** — the PT-BR token compression project ("por que gastar muitos tokens quando poucos resolve?") whose Portuguese-native rules power OmniRoute's pt-BR language pack: pleonasm reduction, filler removal tuned for Brazilian Portuguese grammar, and technical abbreviations for the dev BR community. +## ❤️ Support + +OmniRoute is free and open source, built and maintained in the open. If it saves you time or money, consider supporting development: + +- ⭐ **Star the repo** — it genuinely helps visibility +- 💖 **[GitHub Sponsors](https://github.com/sponsors/diegosouzapw)** — fund ongoing maintenance and new providers +- 🐛 **Report bugs and share feedback** in [Discussions](https://github.com/diegosouzapw/OmniRoute/discussions) + ## 📄 License MIT License - see [LICENSE](LICENSE) for details. diff --git a/bin/cli/commands/setup-open-code.mjs b/bin/cli/commands/setup-open-code.mjs new file mode 100644 index 00000000000..b23103f51eb --- /dev/null +++ b/bin/cli/commands/setup-open-code.mjs @@ -0,0 +1,384 @@ +/** + * omniroute setup opencode — Wire the bundled @omniroute/opencode-plugin + * into a local OpenCode install. + * + * Closes the gap where `npm install -g omniroute` ships the plugin + * inside the omniroute package (`@omniroute/opencode-plugin/dist/`) but + * OpenCode discovers plugins via `~/.config/opencode/plugins/` or + * via entries in `opencode.json`. Without this command, the user has + * to extract the tarball and wire it up by hand (see the plugin README, + * "Install" section). + * + * What it does, in order: + * 1. Resolves the bundled plugin path (source + built dist). + * 2. Resolves the OpenCode config directory (XDG-aware). + * 3. Copies the built plugin into `/plugins/omniroute/`. + * 4. Creates or updates `opencode.json` with a single `plugin` entry + * pointing at the local copy (so OC ≥1.15 picks it up). + * 5. Optionally runs `opencode auth login --provider omniroute` + * so the next `opencode` invocation already has the API key. + * + * Idempotent: re-running with the same `--provider-id` updates the + * entry in place (path + baseURL) without duplicating it. + */ +import { existsSync, mkdirSync, readFileSync, writeFileSync, cpSync } from "node:fs"; +import { dirname, isAbsolute, join, resolve } from "node:path"; +import { fileURLToPath, pathToFileURL } from "node:url"; +import { spawnSync } from "node:child_process"; +import os from "node:os"; + +import { printHeading, printInfo, printSuccess, printError } from "../io.mjs"; +import { t } from "../i18n.mjs"; + +const __filename = fileURLToPath(import.meta.url); +const __dirname = dirname(__filename); + +// We walk up from this file to find the omniroute package root. The script +// lives at `/bin/cli/commands/setup-open-code.mjs`, so the +// package root is three levels up. Using import.meta.url (not process.cwd()) +// means the command works the same way whether you run it from the source +// repo, a global install, or a symlinked location. +const PACKAGE_ROOT = resolve(__dirname, "..", "..", ".."); + +// The bundled plugin ships at PACKAGE_ROOT/@omniroute/opencode-plugin/ +// (see root package.json `files`: ["@omniroute/", ...]). The env override +// exists so tests can point at a fixture without building the real plugin. +const BUNDLED_PLUGIN_DIR = + process.env.OMNIROUTE_OPENCODE_PLUGIN_DIR || + join(PACKAGE_ROOT, "@omniroute", "opencode-plugin"); + +/** + * Resolve the OpenCode config directory. Honours XDG_CONFIG_HOME and the + * platform-specific defaults documented at https://opencode.ai/. + * + * @returns {{ configDir: string, dataDir: string }} + */ +function resolveOpenCodeDirs() { + const home = os.homedir(); + const xdgConfig = process.env.XDG_CONFIG_HOME; + const xdgData = process.env.XDG_DATA_HOME; + const platform = process.platform; + + let configDir; + let dataDir; + if (platform === "darwin") { + // macOS: ~/Library/Application Support/opencode + configDir = join(home, "Library", "Application Support", "opencode"); + dataDir = configDir; // OC uses the same root for config + data on macOS + } else if (platform === "win32") { + const appdata = process.env.APPDATA || join(home, "AppData", "Roaming"); + const localAppdata = process.env.LOCALAPPDATA || join(home, "AppData", "Local"); + configDir = join(appdata, "opencode"); + dataDir = join(localAppdata, "opencode"); + } else { + // Linux + everything else: XDG-style + configDir = xdgConfig ? join(xdgConfig, "opencode") : join(home, ".config", "opencode"); + dataDir = xdgData ? join(xdgData, "opencode") : join(home, ".local", "share", "opencode"); + } + return { configDir, dataDir }; +} + +/** + * Locate the bundled @omniroute/opencode-plugin dist. The plugin may be + * present in two states: + * + * - Built (`dist/index.cjs` + `dist/index.js` exist) — preferred, + * ships from a published omniroute tarball after Step 8.8 of + * `scripts/build/prepublish.ts` runs. + * - Unbuilt (only `src/index.ts`) — local dev / fresh clone. We surface + * a clear error instead of running tsup here, because the CLI runtime + * may not have tsup available (it's a devDependency). + * + * @returns {{ distEntry: string, cjsEntry: string, packageDir: string }} + */ +function resolveBundledPlugin() { + if (!existsSync(BUNDLED_PLUGIN_DIR)) { + throw new Error( + `Bundled @omniroute/opencode-plugin not found at ${BUNDLED_PLUGIN_DIR}.\n` + + `This usually means omniroute was installed from a source tree that does not ` + + `include the workspace package. Try reinstalling omniroute (npm install -g omniroute) ` + + `or run \`cd @omniroute/opencode-plugin && npm install && npm run build\` from the source repo.` + ); + } + + const esmEntry = join(BUNDLED_PLUGIN_DIR, "dist", "index.js"); + const cjsEntry = join(BUNDLED_PLUGIN_DIR, "dist", "index.cjs"); + + if (!existsSync(esmEntry) || !existsSync(cjsEntry)) { + throw new Error( + `@omniroute/opencode-plugin dist/ not built (looked for ${esmEntry}).\n` + + `Run \`cd ${BUNDLED_PLUGIN_DIR} && npm install && npm run build\` and re-run this command.` + ); + } + + // Prefer ESM. OpenCode (≥1.15) loads ESM modules natively. + return { distEntry: esmEntry, cjsEntry, packageDir: BUNDLED_PLUGIN_DIR }; +} + +/** + * Copy the plugin package into `/plugins/omniroute/`. We + * copy the entire package (dist/ + package.json) so the dist file's + * require/import of `zod` and `@opencode-ai/plugin` resolves against the + * copy's own node_modules. Without the copy, OpenCode would need to + * resolve the peer deps from the omniroute package's tree, which is + * unreliable. + */ +function installPluginToOpenCode(pluginInfo, opencodeConfigDir) { + const targetDir = join(opencodeConfigDir, "plugins", "omniroute"); + mkdirSync(dirname(targetDir), { recursive: true }); + mkdirSync(targetDir, { recursive: true }); + + // Copy package.json + dist/. We intentionally do NOT recursively copy + // node_modules from the source — `peerDependenciesMeta` declares zod + + // @opencode-ai/plugin as peers, and the user's OpenCode install already + // provides them. Copying our own node_modules would risk duplicate zod + // instances (the @opencode-ai/plugin contract uses a singleton). + const packageJsonSrc = join(pluginInfo.packageDir, "package.json"); + const distSrc = join(pluginInfo.packageDir, "dist"); + cpSync(packageJsonSrc, join(targetDir, "package.json")); + cpSync(distSrc, join(targetDir, "dist"), { recursive: true }); + + return targetDir; +} + +/** + * Update `opencode.json` to register the plugin. Idempotent: if an entry + * for the same `providerId` already exists, replace it in place. If the + * user has any other plugin entries, preserve them. + * + * @returns {{ configPath: string, changed: boolean }} + */ +function registerPluginInOpenCodeConfig({ + opencodeConfigDir, + pluginTargetDir, + providerId, + baseURL, + displayName, +}) { + const configPath = join(opencodeConfigDir, "opencode.json"); + let cfg = {}; + if (existsSync(configPath)) { + try { + cfg = JSON.parse(readFileSync(configPath, "utf8")); + } catch (err) { + throw new Error( + `Failed to parse existing ${configPath}: ${err.message}\n` + + `Fix or remove the file manually, then re-run \`omniroute setup opencode\`.` + ); + } + } + + const plugins = Array.isArray(cfg.plugin) ? cfg.plugin : []; + + // Plugin entries can be either a string ("@some/pkg") or a tuple + // ("@some/pkg", { options }). The README documents the tuple form, so + // we use that. The "module path" is a file:// URL relative to the + // opencode config dir — that is what opencode ≥1.15 resolves. + const entry = [ + `./plugins/omniroute/dist/index.js`, + { + providerId, + baseURL, + ...(displayName ? { displayName } : {}), + }, + ]; + + // Idempotency: drop any prior entry for the same providerId. We also + // drop a legacy `opencode-omniroute-auth` entry if present — that + // package is the obsolete predecessor of @omniroute/opencode-plugin + // and was the root cause of issue #3711. + const filtered = plugins.filter((p) => { + if (typeof p === "string") { + return !p.includes("opencode-omniroute-auth"); + } + if (Array.isArray(p) && p[1] && typeof p[1] === "object") { + const pid = p[1].providerId; + if (pid === providerId) return false; + // Also drop the legacy auth plugin if it's there. + if (typeof p[0] === "string" && p[0].includes("opencode-omniroute-auth")) { + return false; + } + } + return true; + }); + filtered.push(entry); + cfg.plugin = filtered; + + // Make sure the config dir exists, then write the updated config. + mkdirSync(dirname(configPath), { recursive: true }); + writeFileSync(configPath, JSON.stringify(cfg, null, 2) + "\n", "utf8"); + + return { configPath, changed: true }; +} + +/** + * Optionally invoke `opencode auth login --provider `. We + * shell out (instead of importing) so this command works even if + * OpenCode's CLI surface shifts between minor versions — the user gets + * a clear "could not run opencode" message instead of a hard import + * failure. + */ +function runOpenCodeAuth(providerId) { + const isWin = process.platform === "win32"; + const opencodeBin = isWin ? "opencode.cmd" : "opencode"; + const res = spawnSync(opencodeBin, ["auth", "login", "--provider", providerId], { + stdio: "inherit", + shell: false, + }); + if (res.error) { + // ENOENT = opencode is not on PATH + if (res.error.code === "ENOENT") { + printInfo( + `opencode CLI not found on PATH. Run \`opencode auth login --provider ${providerId}\` manually after installing OpenCode.` + ); + return 1; + } + printError(`opencode auth login failed: ${res.error.message}`); + return 1; + } + return typeof res.status === "number" ? res.status : 1; +} + +/** + * Top-level action handler. Kept exported so the integration test can + * drive it without spawning a subprocess. + * + * @param {object} opts + * @param {string} [opts.providerId="omniroute"] + * @param {string} [opts.baseURL="http://localhost:20128"] (Commander camelCases + * `--base-url` into `baseUrl`, so both spellings are accepted.) + * @param {string} [opts.configDir] Override the OpenCode config dir (tests / non-standard installs). + * @param {string} [opts.displayName] + * @param {boolean} [opts.auth=false] Run `opencode auth login` after wiring. + * @param {boolean} [opts.nonInteractive=false] Skip prompts. + * @returns {Promise<{ exitCode: number, configPath?: string, pluginTargetDir?: string }>} + */ +export async function runSetupOpenCodeCommand(opts = {}) { + const providerId = opts.providerId || "omniroute"; + const baseURL = opts.baseURL || opts.baseUrl || "http://localhost:20128"; + const displayName = opts.displayName || null; + const wantsAuth = Boolean(opts.auth); + const nonInteractive = Boolean(opts.nonInteractive); + + printHeading("OmniRoute → OpenCode Plugin Setup"); + + const resolvedDirs = resolveOpenCodeDirs(); + const opencodeConfigDir = opts.configDir || resolvedDirs.configDir; + const opencodeDataDir = resolvedDirs.dataDir; + printInfo(`OpenCode config dir: ${opencodeConfigDir}`); + printInfo(`OpenCode data dir: ${opencodeDataDir}`); + + // 1. Resolve bundled plugin + let pluginInfo; + try { + pluginInfo = resolveBundledPlugin(); + } catch (err) { + printError(err.message); + return { exitCode: 1 }; + } + printInfo(`Bundled plugin: ${pluginInfo.distEntry}`); + + // 2. Ensure OpenCode config dir exists (opencode will create it on + // first run, but creating it now means we can write opencode.json + // even if OC has never been launched). + if (!existsSync(opencodeConfigDir)) { + mkdirSync(opencodeConfigDir, { recursive: true }); + printInfo(`Created OpenCode config dir (didn't exist yet).`); + } + + // 3. Copy plugin into OpenCode's plugin dir + let pluginTargetDir; + try { + pluginTargetDir = installPluginToOpenCode(pluginInfo, opencodeConfigDir); + printSuccess(`Plugin installed at ${pluginTargetDir}`); + } catch (err) { + printError(`Failed to install plugin: ${err.message}`); + return { exitCode: 1 }; + } + + // 4. Register in opencode.json + let configPath; + try { + const reg = registerPluginInOpenCodeConfig({ + opencodeConfigDir, + pluginTargetDir, + providerId, + baseURL, + displayName, + }); + configPath = reg.configPath; + printSuccess(`opencode.json updated at ${configPath}`); + } catch (err) { + printError(`Failed to update opencode.json: ${err.message}`); + return { exitCode: 1, pluginTargetDir }; + } + + // 5. Optionally run auth login + if (wantsAuth) { + if (nonInteractive) { + printInfo(`Skipping \`opencode auth login\` (non-interactive mode).`); + printInfo(`Run manually: opencode auth login --provider ${providerId}`); + } else { + printHeading("Authenticating with OpenCode"); + const authExit = runOpenCodeAuth(providerId); + if (authExit !== 0) { + return { exitCode: authExit, configPath, pluginTargetDir }; + } + } + } else { + printInfo( + `Next step: opencode auth login --provider ${providerId} (pass --auth to do this automatically)` + ); + } + + printSuccess("OpenCode plugin setup complete"); + printInfo(`Restart OpenCode to pick up the new plugin entry.`); + return { exitCode: 0, configPath, pluginTargetDir }; +} + +/** + * Register the `omniroute setup opencode` subcommand on the parent + * `setup` command. Commander builds the doc/help from the chain, so + * `omniroute setup --help` automatically shows the new subcommand. + * + * @param {import("commander").Command} setupCommand the registered `setup` command + */ +export function registerSetupOpenCode(setupCommand) { + setupCommand + .command("opencode") + .description( + t("setup.opencode") || + "Install and register the bundled @omniroute/opencode-plugin with a local OpenCode install" + ) + .option( + "--provider-id ", + "OpenCode provider id to register (default: omniroute)", + "omniroute" + ) + .option( + "--base-url ", + "OmniRoute base URL the plugin should talk to (default: http://localhost:20128)", + "http://localhost:20128" + ) + .option("--display-name ", "Display name in the OpenCode UI (optional)") + .option( + "--auth", + "Run `opencode auth login --provider ` after wiring (interactive)", + false + ) + .option("--non-interactive", "Do not prompt; skip the auth login step", false) + .action(async (opts, cmd) => { + // The parent `setup` command uses cmd.optsWithGlobals(); we mirror + // that here so global flags (--json, --base-url, --api-key) still + // flow through to the runner. + const globalOpts = cmd.parent?.parent?.optsWithGlobals?.() ?? {}; + const merged = { + ...opts, + output: globalOpts.output, + apiKey: opts.apiKey ?? globalOpts.apiKey, + baseUrl: opts.baseUrl ?? globalOpts.baseUrl, + }; + const { exitCode } = await runSetupOpenCodeCommand(merged); + if (exitCode !== 0) process.exit(exitCode); + }); +} diff --git a/bin/cli/commands/setup.mjs b/bin/cli/commands/setup.mjs index 690c4d9e08d..cda32573b27 100644 --- a/bin/cli/commands/setup.mjs +++ b/bin/cli/commands/setup.mjs @@ -10,6 +10,7 @@ import { getProviderDisplayName, resolveProviderChoice, } from "../provider-catalog.mjs"; +import { registerSetupOpenCode } from "./setup-open-code.mjs"; import { t } from "../i18n.mjs"; const PROJECT_ROOT = resolve(dirname(fileURLToPath(import.meta.url)), "../../.."); @@ -150,6 +151,11 @@ export function registerSetup(program) { const exitCode = await runSetupCommand({ ...opts, output: globalOpts.output }); if (exitCode !== 0) process.exit(exitCode); }); + + // Wire up `omniroute setup opencode` subcommand. Kept inside registerSetup + // so it always travels with the parent command (avoids a separate register + // call in the registry that would silently break if the parent renames). + registerSetupOpenCode(program.commands.find((c) => c.name() === "setup")); } export async function runSetupCommand(opts = {}) { diff --git a/bin/cli/locales/en.json b/bin/cli/locales/en.json index 857e18b65a3..bcbc016701c 100644 --- a/bin/cli/locales/en.json +++ b/bin/cli/locales/en.json @@ -26,7 +26,8 @@ "testFailed": "Provider test failed: {error}", "loginEnabled": "Login: enabled (password updated)", "loginDisabled": "Login: disabled", - "providerInfo": "Provider: {info}" + "providerInfo": "Provider: {info}", + "opencode": "Install and configure the bundled @omniroute/opencode-plugin for OpenCode" }, "doctor": { "title": "OmniRoute Doctor", diff --git a/docs/guides/USER_GUIDE.md b/docs/guides/USER_GUIDE.md index 6256824de54..dcc10ce3f9b 100644 --- a/docs/guides/USER_GUIDE.md +++ b/docs/guides/USER_GUIDE.md @@ -989,6 +989,12 @@ history, or compressing fallback requests; enabling it allows configured hedging skips, and proactive fallback compression to trade routing/request fidelity for lower tail latency. +Disable **Reasoning token buffer** when upstream providers require strict +`max_tokens` / `maxOutputTokens` limits. When enabled, combo routing only adds reasoning-model +headroom for models with a known output cap and leaves the client token limit unchanged when the +safe buffered value would exceed that cap. If the client limit is already above a known cap, +OmniRoute clamps it down to that cap before sending the upstream request. + --- ### Health Dashboard diff --git a/docs/i18n/ar/CHANGELOG.md b/docs/i18n/ar/CHANGELOG.md index 087aa8fe977..a69849f9a10 100644 --- a/docs/i18n/ar/CHANGELOG.md +++ b/docs/i18n/ar/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/az/CHANGELOG.md b/docs/i18n/az/CHANGELOG.md index 6c3e4d337c6..53b25f55cbe 100644 --- a/docs/i18n/az/CHANGELOG.md +++ b/docs/i18n/az/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/bg/CHANGELOG.md b/docs/i18n/bg/CHANGELOG.md index 6c3e4d337c6..53b25f55cbe 100644 --- a/docs/i18n/bg/CHANGELOG.md +++ b/docs/i18n/bg/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/bn/CHANGELOG.md b/docs/i18n/bn/CHANGELOG.md index d1e38220e5f..eac10e6cd8a 100644 --- a/docs/i18n/bn/CHANGELOG.md +++ b/docs/i18n/bn/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/cs/CHANGELOG.md b/docs/i18n/cs/CHANGELOG.md index 856d776d617..e8ca032a15f 100644 --- a/docs/i18n/cs/CHANGELOG.md +++ b/docs/i18n/cs/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/da/CHANGELOG.md b/docs/i18n/da/CHANGELOG.md index 420b57b2392..17cf7768eaf 100644 --- a/docs/i18n/da/CHANGELOG.md +++ b/docs/i18n/da/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/de/CHANGELOG.md b/docs/i18n/de/CHANGELOG.md index dd29ccc8990..c19a046563f 100644 --- a/docs/i18n/de/CHANGELOG.md +++ b/docs/i18n/de/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/es/CHANGELOG.md b/docs/i18n/es/CHANGELOG.md index cbc8be128d3..4e22735726d 100644 --- a/docs/i18n/es/CHANGELOG.md +++ b/docs/i18n/es/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/fa/CHANGELOG.md b/docs/i18n/fa/CHANGELOG.md index 7b8246e1e4b..36e621d9a0a 100644 --- a/docs/i18n/fa/CHANGELOG.md +++ b/docs/i18n/fa/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/fi/CHANGELOG.md b/docs/i18n/fi/CHANGELOG.md index 9ae8fc57e63..6baf5e64750 100644 --- a/docs/i18n/fi/CHANGELOG.md +++ b/docs/i18n/fi/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/fr/CHANGELOG.md b/docs/i18n/fr/CHANGELOG.md index 12446f97d89..3a99dde7667 100644 --- a/docs/i18n/fr/CHANGELOG.md +++ b/docs/i18n/fr/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/gu/CHANGELOG.md b/docs/i18n/gu/CHANGELOG.md index f7115d78d24..8a2ec074358 100644 --- a/docs/i18n/gu/CHANGELOG.md +++ b/docs/i18n/gu/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/he/CHANGELOG.md b/docs/i18n/he/CHANGELOG.md index ce4f834626d..f542d29ede6 100644 --- a/docs/i18n/he/CHANGELOG.md +++ b/docs/i18n/he/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/hi/CHANGELOG.md b/docs/i18n/hi/CHANGELOG.md index 67952135d76..5854502c47d 100644 --- a/docs/i18n/hi/CHANGELOG.md +++ b/docs/i18n/hi/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/hu/CHANGELOG.md b/docs/i18n/hu/CHANGELOG.md index 4153d89e7ad..cac7dfef759 100644 --- a/docs/i18n/hu/CHANGELOG.md +++ b/docs/i18n/hu/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/id/CHANGELOG.md b/docs/i18n/id/CHANGELOG.md index 1a4946a8a84..efbd66686b8 100644 --- a/docs/i18n/id/CHANGELOG.md +++ b/docs/i18n/id/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/in/CHANGELOG.md b/docs/i18n/in/CHANGELOG.md index 287f5e74ad7..f4b5f6aedde 100644 --- a/docs/i18n/in/CHANGELOG.md +++ b/docs/i18n/in/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/it/CHANGELOG.md b/docs/i18n/it/CHANGELOG.md index ea20c5e0a9f..4293c8264f2 100644 --- a/docs/i18n/it/CHANGELOG.md +++ b/docs/i18n/it/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/ja/CHANGELOG.md b/docs/i18n/ja/CHANGELOG.md index bee38308130..84e747513df 100644 --- a/docs/i18n/ja/CHANGELOG.md +++ b/docs/i18n/ja/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/ko/CHANGELOG.md b/docs/i18n/ko/CHANGELOG.md index 2d4a4bc119a..7e05fe336c6 100644 --- a/docs/i18n/ko/CHANGELOG.md +++ b/docs/i18n/ko/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/mr/CHANGELOG.md b/docs/i18n/mr/CHANGELOG.md index 2d6d31b79b5..e6ea8faf0e1 100644 --- a/docs/i18n/mr/CHANGELOG.md +++ b/docs/i18n/mr/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/ms/CHANGELOG.md b/docs/i18n/ms/CHANGELOG.md index 941dbf55ec8..12b6491553e 100644 --- a/docs/i18n/ms/CHANGELOG.md +++ b/docs/i18n/ms/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/nl/CHANGELOG.md b/docs/i18n/nl/CHANGELOG.md index c957a171fdd..87ee99cd32d 100644 --- a/docs/i18n/nl/CHANGELOG.md +++ b/docs/i18n/nl/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/no/CHANGELOG.md b/docs/i18n/no/CHANGELOG.md index 2620b21135a..48336b783ea 100644 --- a/docs/i18n/no/CHANGELOG.md +++ b/docs/i18n/no/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/phi/CHANGELOG.md b/docs/i18n/phi/CHANGELOG.md index d149e3c9e0a..b0e72d3e669 100644 --- a/docs/i18n/phi/CHANGELOG.md +++ b/docs/i18n/phi/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/pl/CHANGELOG.md b/docs/i18n/pl/CHANGELOG.md index 0688bb4b921..985f01dfc05 100644 --- a/docs/i18n/pl/CHANGELOG.md +++ b/docs/i18n/pl/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/pt-BR/CHANGELOG.md b/docs/i18n/pt-BR/CHANGELOG.md index afeeb61e5b3..fc2a5cadbea 100644 --- a/docs/i18n/pt-BR/CHANGELOG.md +++ b/docs/i18n/pt-BR/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/pt/CHANGELOG.md b/docs/i18n/pt/CHANGELOG.md index 46cd4ffbf23..c1e556f9c72 100644 --- a/docs/i18n/pt/CHANGELOG.md +++ b/docs/i18n/pt/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/ro/CHANGELOG.md b/docs/i18n/ro/CHANGELOG.md index ee4c92d4305..7515cb8c775 100644 --- a/docs/i18n/ro/CHANGELOG.md +++ b/docs/i18n/ro/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/ru/CHANGELOG.md b/docs/i18n/ru/CHANGELOG.md index 31a5da7e2d8..8602274b67c 100644 --- a/docs/i18n/ru/CHANGELOG.md +++ b/docs/i18n/ru/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/sk/CHANGELOG.md b/docs/i18n/sk/CHANGELOG.md index 63117457b12..e03ef3ee6db 100644 --- a/docs/i18n/sk/CHANGELOG.md +++ b/docs/i18n/sk/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/sv/CHANGELOG.md b/docs/i18n/sv/CHANGELOG.md index f071c2e757d..4467eb15ed9 100644 --- a/docs/i18n/sv/CHANGELOG.md +++ b/docs/i18n/sv/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/sw/CHANGELOG.md b/docs/i18n/sw/CHANGELOG.md index 6307211fb36..c85af1c1318 100644 --- a/docs/i18n/sw/CHANGELOG.md +++ b/docs/i18n/sw/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/ta/CHANGELOG.md b/docs/i18n/ta/CHANGELOG.md index bb5ca0b2145..30f309ffd8b 100644 --- a/docs/i18n/ta/CHANGELOG.md +++ b/docs/i18n/ta/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/te/CHANGELOG.md b/docs/i18n/te/CHANGELOG.md index 926726b61f9..09ea46c7b10 100644 --- a/docs/i18n/te/CHANGELOG.md +++ b/docs/i18n/te/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/th/CHANGELOG.md b/docs/i18n/th/CHANGELOG.md index ef1d0a62ced..9d785a7add4 100644 --- a/docs/i18n/th/CHANGELOG.md +++ b/docs/i18n/th/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/tr/CHANGELOG.md b/docs/i18n/tr/CHANGELOG.md index 1eb13067872..f59f8e1a7e8 100644 --- a/docs/i18n/tr/CHANGELOG.md +++ b/docs/i18n/tr/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/uk-UA/CHANGELOG.md b/docs/i18n/uk-UA/CHANGELOG.md index e606dd07dd8..1583ff85f79 100644 --- a/docs/i18n/uk-UA/CHANGELOG.md +++ b/docs/i18n/uk-UA/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/ur/CHANGELOG.md b/docs/i18n/ur/CHANGELOG.md index 21e84b8e728..31a19aecf5b 100644 --- a/docs/i18n/ur/CHANGELOG.md +++ b/docs/i18n/ur/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/vi/CHANGELOG.md b/docs/i18n/vi/CHANGELOG.md index ba1ca85bcc3..739c7af6000 100644 --- a/docs/i18n/vi/CHANGELOG.md +++ b/docs/i18n/vi/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/i18n/zh-CN/CHANGELOG.md b/docs/i18n/zh-CN/CHANGELOG.md index 630537e06f2..6cc1a5fbe3e 100644 --- a/docs/i18n/zh-CN/CHANGELOG.md +++ b/docs/i18n/zh-CN/CHANGELOG.md @@ -171,7 +171,7 @@ _Development cycle in progress._ `[403] `. The quota path now skips proactive refresh for rotating providers (`rotationGroupFor`) and reuses the current access_token, deferring genuine expiry to the reactive, serialized 401 path. Defense in - depth: `serializeRefresh` now leaves a settle gap between two *queued* sibling + depth: `serializeRefresh` now leaves a settle gap between two _queued_ sibling refreshes (default 2000 ms, tunable via `CODEX_REFRESH_SPACING_MS`, `"0"` to opt out) while releasing a lone refresh immediately, so the reactive path adds no latency. @@ -234,7 +234,7 @@ _Development cycle in progress._ via a new `RegistryModel.interleavedField` field, so follow-up/tool-use turns replay reasoning_content. Previously `big-pickle` matched no replay pattern and failed with `[400] The reasoning_content in the thinking mode must be passed - back to the API` (its DeepSeek-thinking upstream is not detectable from the +back to the API` (its DeepSeek-thinking upstream is not detectable from the model id, and `requiresReasoningReplay` does not consume `supportsReasoning`). `getResolvedModelCapabilities` now surfaces the registry `interleavedField`. (#2900) - **providers/github-copilot:** built-in GitHub Copilot Claude Opus and Gemini @@ -267,7 +267,7 @@ _Development cycle in progress._ instead of failing with "No credentials for provider: opencode-zen". A configured, active key is still used when present. (#2962) - **translator/responses:** fixed an upstream `[400] Messages with role 'tool' - must be a response to a preceding message with 'tool_calls'` when a Codex +must be a response to a preceding message with 'tool_calls'` when a Codex client sent a `function_call` with an empty/missing `call_id`. The orphaned `function_call_output` previously slipped past the orphan filter. Now empty-`call_id` function calls are skipped (no dangling assistant tool_call) @@ -386,6 +386,12 @@ _Development cycle in progress._ ## [Unreleased] +--- + +## [3.8.23] — TBD + +--- + ### ✨ New Features ### 🔧 Bug Fixes diff --git a/docs/reference/ENVIRONMENT.md b/docs/reference/ENVIRONMENT.md index 333bc4b6978..49aebe2736b 100644 --- a/docs/reference/ENVIRONMENT.md +++ b/docs/reference/ENVIRONMENT.md @@ -685,6 +685,15 @@ Automatic model pricing data synchronization from external sources. --- +## Arena ELO Sync + +| Variable | Default | Source File | Description | +| --------------------------- | ------------- | -------------------------- | ------------------------------------------------------------- | +| `ARENA_ELO_SYNC_ENABLED` | `false` | `src/lib/arenaEloSync.ts` | Opt-in periodic Arena AI leaderboard ELO sync. | +| `ARENA_ELO_SYNC_INTERVAL` | `86400` (24h) | `src/lib/arenaEloSync.ts` | Sync interval in seconds. | + +--- + ## 19. Model Sync (Dev) | Variable | Default | Source File | Description | diff --git a/docs/reference/openapi.yaml b/docs/reference/openapi.yaml index 3b8cdb8ff04..a93fb1dea25 100644 --- a/docs/reference/openapi.yaml +++ b/docs/reference/openapi.yaml @@ -1,7 +1,7 @@ openapi: 3.1.0 info: title: OmniRoute API - version: 3.8.22 + version: 3.8.21 description: | OmniRoute is a local-first AI API proxy router. It provides an OpenAI-compatible endpoint that routes requests to multiple AI providers with load balancing, diff --git a/electron/package.json b/electron/package.json index 888f249bd05..4d007c3c13d 100644 --- a/electron/package.json +++ b/electron/package.json @@ -1,6 +1,6 @@ { "name": "omniroute-desktop", - "version": "3.8.22", + "version": "3.8.21", "description": "OmniRoute Desktop Application", "main": "main.js", "author": { @@ -54,7 +54,6 @@ "loginManager.js", "processTree.js", "sqlite-inspection.js", - "lib/resolveServerEntry.js", "package.json", "node_modules/**/*" ], diff --git a/file-size-baseline.json b/file-size-baseline.json index fbd4cd6e9b2..d2ba7c8977e 100644 --- a/file-size-baseline.json +++ b/file-size-baseline.json @@ -2,40 +2,40 @@ "_comment": "Catraca de tamanho (check-file-size.mjs). frozen so pode encolher; arquivos novos <= cap. --update ratcheta.", "cap": 800, "frozen": { - "open-sse/config/providerRegistry.ts": 4677, - "open-sse/executors/antigravity.ts": 1533, - "open-sse/executors/base.ts": 1175, + "open-sse/config/providerRegistry.ts": 4692, + "open-sse/executors/antigravity.ts": 1553, + "open-sse/executors/base.ts": 1199, "open-sse/executors/chatgpt-web.ts": 2870, "open-sse/executors/claude-web.ts": 1057, "open-sse/executors/codex.ts": 1439, "open-sse/executors/cursor.ts": 1391, - "open-sse/executors/deepseek-web.ts": 1116, + "open-sse/executors/deepseek-web.ts": 1117, "open-sse/executors/duckduckgo-web.ts": 917, "open-sse/executors/grok-web.ts": 1871, "open-sse/executors/muse-spark-web.ts": 1284, - "open-sse/executors/perplexity-web.ts": 867, + "open-sse/executors/perplexity-web.ts": 868, "open-sse/handlers/audioSpeech.ts": 952, - "open-sse/handlers/chatCore.ts": 6023, + "open-sse/handlers/chatCore.ts": 5808, "open-sse/handlers/imageGeneration.ts": 3777, - "open-sse/handlers/responseSanitizer.ts": 1080, - "open-sse/handlers/search.ts": 1441, + "open-sse/handlers/responseSanitizer.ts": 1103, + "open-sse/handlers/search.ts": 1442, "open-sse/handlers/videoGeneration.ts": 1026, "open-sse/mcp-server/schemas/tools.ts": 1437, "open-sse/mcp-server/server.ts": 1457, "open-sse/mcp-server/tools/advancedTools.ts": 1118, - "open-sse/services/accountFallback.ts": 1633, + "open-sse/services/accountFallback.ts": 1708, "open-sse/services/batchProcessor.ts": 828, "open-sse/services/browserBackedChat.ts": 850, "open-sse/services/claudeCodeCompatible.ts": 1202, - "open-sse/services/combo.ts": 4530, + "open-sse/services/combo.ts": 5054, "open-sse/services/rateLimitManager.ts": 1017, - "open-sse/services/tokenRefresh.ts": 1896, - "open-sse/services/usage.ts": 3042, - "open-sse/translator/request/openai-to-gemini.ts": 822, + "open-sse/services/tokenRefresh.ts": 1997, + "open-sse/services/usage.ts": 3341, + "open-sse/translator/request/openai-to-gemini.ts": 844, "open-sse/translator/response/openai-responses.ts": 873, "open-sse/utils/cursorAgentProtobuf.ts": 1499, - "open-sse/utils/stream.ts": 2694, - "src/app/(dashboard)/dashboard/HomePageClient.tsx": 1417, + "open-sse/utils/stream.ts": 2710, + "src/app/(dashboard)/dashboard/HomePageClient.tsx": 1385, "src/app/(dashboard)/dashboard/analytics/ComboHealthTab.tsx": 1020, "src/app/(dashboard)/dashboard/api-manager/ApiManagerPageClient.tsx": 2680, "src/app/(dashboard)/dashboard/cache/media/MediaPageClient.tsx": 1105, @@ -48,61 +48,66 @@ "src/app/(dashboard)/dashboard/endpoint/EndpointPageClient.tsx": 2570, "src/app/(dashboard)/dashboard/health/page.tsx": 1091, "src/app/(dashboard)/dashboard/playground/components/tabs/ApiTab.tsx": 847, - "src/app/(dashboard)/dashboard/providers/[id]/ProviderDetailPageClient.tsx": 4063, - "src/app/(dashboard)/dashboard/providers/[id]/hooks/useProviderConnections.ts": 954, - "src/app/(dashboard)/dashboard/providers/[id]/hooks/useProviderSettings.ts": 264, - "src/app/(dashboard)/dashboard/providers/[id]/hooks/useProviderModels.ts": 155, - "src/app/(dashboard)/dashboard/providers/[id]/providerPageHelpers.ts": 822, + "src/app/(dashboard)/dashboard/providers/[id]/ProviderDetailPageClient.tsx": 782, "src/app/(dashboard)/dashboard/providers/[id]/components/ConnectionRow.tsx": 941, "src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx": 843, "src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx": 1171, + "src/app/(dashboard)/dashboard/providers/[id]/hooks/useProviderConnections.ts": 954, + "src/app/(dashboard)/dashboard/providers/[id]/hooks/useProviderModels.ts": 155, + "src/app/(dashboard)/dashboard/providers/[id]/hooks/useProviderSettings.ts": 264, + "src/app/(dashboard)/dashboard/providers/[id]/providerPageHelpers.ts": 897, "src/app/(dashboard)/dashboard/providers/components/onboarding/ProviderOnboardingWizard.tsx": 906, "src/app/(dashboard)/dashboard/providers/page.tsx": 1925, - "src/app/(dashboard)/dashboard/runtime/RuntimePageClient.tsx": 1127, + "src/app/(dashboard)/dashboard/runtime/RuntimePageClient.tsx": 1198, "src/app/(dashboard)/dashboard/settings/components/AppearanceTab.tsx": 819, "src/app/(dashboard)/dashboard/settings/components/CompressionSettingsTab.tsx": 932, "src/app/(dashboard)/dashboard/settings/components/MemorySkillsTab.tsx": 880, "src/app/(dashboard)/dashboard/settings/components/PricingTab.tsx": 1012, "src/app/(dashboard)/dashboard/settings/components/ProxyRegistryManager.tsx": 1072, - "src/app/(dashboard)/dashboard/settings/components/ResilienceTab.tsx": 851, + "src/app/(dashboard)/dashboard/settings/components/ResilienceTab.tsx": 983, "src/app/(dashboard)/dashboard/settings/components/RoutingTab.tsx": 1580, "src/app/(dashboard)/dashboard/settings/components/SystemStorageTab.tsx": 1924, "src/app/(dashboard)/dashboard/usage/components/BudgetTab.tsx": 1016, "src/app/(dashboard)/dashboard/usage/components/EvalsTab.tsx": 2148, - "src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.tsx": 1015, + "src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.tsx": 1069, "src/app/api/oauth/[provider]/[action]/route.ts": 897, - "src/app/api/providers/[id]/models/route.ts": 2287, + "src/app/api/providers/[id]/models/route.ts": 2426, "src/app/api/providers/[id]/test/route.ts": 842, - "src/app/api/usage/analytics/route.ts": 1355, + "src/app/api/usage/analytics/route.ts": 941, "src/app/api/v1/models/catalog.ts": 1435, "src/lib/cloudflaredTunnel.ts": 934, "src/lib/db/apiKeys.ts": 1490, "src/lib/db/core.ts": 1820, "src/lib/db/migrationRunner.ts": 1100, "src/lib/db/models.ts": 1132, - "src/lib/db/providers.ts": 993, - "src/lib/db/proxies.ts": 1031, - "src/lib/db/settings.ts": 1101, + "src/lib/db/providers.ts": 1050, + "src/lib/db/proxies.ts": 1039, + "src/lib/db/settings.ts": 1108, "src/lib/db/usageAnalytics.ts": 873, "src/lib/evals/evalRunner.ts": 961, "src/lib/memory/retrieval.ts": 1171, "src/lib/modelsDevSync.ts": 934, - "src/lib/providers/validation.ts": 4201, + "src/lib/providers/validation.ts": 4209, "src/lib/tailscaleTunnel.ts": 1189, "src/lib/usage/callLogs.ts": 975, - "src/lib/usage/usageHistory.ts": 840, + "src/lib/usage/providerLimits.ts": 941, + "src/lib/usage/usageHistory.ts": 854, "src/shared/components/OAuthModal.tsx": 956, "src/shared/components/RequestLoggerV2.tsx": 1232, "src/shared/components/analytics/charts.tsx": 1558, "src/shared/constants/cliTools.ts": 875, - "src/shared/constants/pricing.ts": 1447, - "src/shared/constants/providers.ts": 3121, + "src/shared/constants/pricing.ts": 1470, + "src/shared/constants/providers.ts": 3144, "src/shared/constants/sidebarVisibility.ts": 990, - "src/shared/services/cliRuntime.ts": 1073, - "src/shared/validation/schemas.ts": 2490, - "src/sse/handlers/chat.ts": 1381, - "src/sse/services/auth.ts": 2198 + "src/shared/services/cliRuntime.ts": 1084, + "src/shared/validation/schemas.ts": 2515, + "src/sse/handlers/chat.ts": 1389, + "src/sse/services/auth.ts": 2207 }, "_rebaseline_2026_06_09": "Re-baseline consciente pre-release v3.8.19: 9 arquivos cresceram durante o ciclo (features mergeadas: RequestLoggerV2 +281 request-logger rework, stream +101, combo +73, chatCore +45, catalog +32 fable-5/catalog-flag, callLogs +4, accountFallback +2, usageHistory novo 840) + core.ts +7 (fix resetAllDbModuleState, PR 3536). A catraca segue valendo destes valores — proximo crescimento falha. Decisao: encolher (esp. RequestLoggerV2/chatCore) e a issue #3501 ficam para o ciclo seguinte.", - "_rebaseline_2026_06_11_phase1f": "Phase 1f (#3501): ProviderDetailPageClient.tsx 4948→4062 (-886 LOC); 3 novos hooks extraídos. useProviderConnections.ts=954 acima do cap=800 — justificado: extração direta do god-component (zero lógica nova), própria redução do cliente supera o custo. useProviderSettings.ts=263 e useProviderModels.ts=154 já abaixo do cap." + "_rebaseline_2026_06_11_phase1f": "Phase 1f (#3501): ProviderDetailPageClient.tsx 4948→4062 (-886 LOC); 3 novos hooks extraídos. useProviderConnections.ts=954 acima do cap=800 — justificado: extração direta do god-component (zero lógica nova), própria redução do cliente supera o custo. useProviderSettings.ts=263 e useProviderModels.ts=154 já abaixo do cap.", + "_rebaseline_2026_06_12_review_issues": "Re-baseline consciente do /review-issues v3.8.23: 27 arquivos com crescimento herdado (v3.8.22 nunca reconciliado) + fixes deste round (combo.ts #3685, openai-to-gemini.ts #3688, tokenRefresh.ts #3692, validation/proxies de outras merges). providerLimits.ts (941) adicionado como frozen (split coeso de usage). Shrink endereçado separadamente pelo #3501.", + "_rebaseline_2026_06_12_phase1g1j": "Phase 1g-1j (#3501): ProviderDetailPageClient.tsx 4063→3409 (extraídos ProviderPlaygroundPanel, useCommandCodeAuth, useExternalLinkFlow+ExternalLinkModal, useAuthFileHandlers — zero lógica nova). models/route.ts 2344→2426: drift do #3712 (vertex dynamic model discovery) reconciliado aqui.", + "_rebaseline_2026_06_12_phase1n1s": "Phase 1n-1s (#3501): ProviderDetailPageClient.tsx 2554→1376 (extraídos ConnectionsListPanel, ConnectionsHeaderToolbar, ZedImportCard, BatchTestResultsModal, AdaptaTutorialModal + hooks/useApiKeySave + 4 helper closures→providerPageHelpers.ts). providerPageHelpers.ts 822→897 justificado: recebe 4 closures do god-component (getApiLabel/getApiDefaultPath/getApiPath/getHeaderIconProviderId), zero lógica nova, cliente encolhe mais do que helpers crescem.", + "_rebaseline_2026_06_12_phase1t": "Phase 1t (#3501): ProviderDetailPageClient.tsx 1377→782 — META ≤800 ATINGIDA (extraídos ProviderPageHeader, CompatibleNodeCard, ProviderModalsPanel, EmptyConnectionsPlaceholder, UpstreamProxyCard, SearchProviderCard + hooks useConnectionGate/useProviderNodeActions). Drift concorrente reconciliado: ResilienceTab/sse-chat/sse-auth/accountFallback/combo (merges #3629 model-lockout etc.)." } diff --git a/open-sse/config/antigravityModelAliases.ts b/open-sse/config/antigravityModelAliases.ts index 0376d7396de..26859620694 100644 --- a/open-sse/config/antigravityModelAliases.ts +++ b/open-sse/config/antigravityModelAliases.ts @@ -206,43 +206,6 @@ export function toClientAntigravityModelId(modelId: string): string { return ANTIGRAVITY_REVERSE_MODEL_ALIASES[modelId] || modelId; } -// Quota buckets reported by the Antigravity backend are keyed by UPSTREAM model ids — a -// DIFFERENT namespace from the public/client catalog. In that upstream quota namespace -// `gemini-3.5-flash-low` denotes the *Medium* tier's bucket (it is the upstream target of -// the `gemini-3.5-flash-medium` forward alias), even though the same literal is also a -// public "Low" client id. This remap therefore CANNOT be derived from -// ANTIGRAVITY_REVERSE_MODEL_ALIASES (which has no `gemini-3.5-flash-low` entry precisely -// because it is already a valid client id) — it encodes the upstream-bucket → client-tier -// chain explicitly. Keep it the inverse of the `-low/-medium/-high` rows in -// ANTIGRAVITY_MODEL_ALIASES above. (#3821-review LEDGER-5 — was duplicated as an inline -// if-ladder in open-sse/services/usage.ts.) -const ANTIGRAVITY_QUOTA_BUCKET_TO_CLIENT: AntigravityModelAliasMap = Object.freeze({ - "gemini-3.5-flash-extra-low": "gemini-3.5-flash-low", - "gemini-3.5-flash-low": "gemini-3.5-flash-medium", - "gemini-3-flash-agent": "gemini-3.5-flash-high", -}); - -// Retired/hidden upstream preview buckets that must be dropped from client-facing usage. -const ANTIGRAVITY_DROPPED_QUOTA_BUCKETS = new Set([ - "gemini-3.5-flash-preview", - "gemini-3-flash-preview", -]); - -/** - * Map an UPSTREAM Antigravity quota-bucket model id to the client-visible tier id used in - * usage responses, or `null` if the bucket should be hidden from clients. Operates on the - * upstream quota namespace (see ANTIGRAVITY_QUOTA_BUCKET_TO_CLIENT) — do NOT pass client - * ids here. Single source of truth shared by the usage service and the provider-limits - * cache sanitizer. - */ -export function toClientAntigravityQuotaModelId(modelId: string): string | null { - if (!modelId) return null; - if (ANTIGRAVITY_DROPPED_QUOTA_BUCKETS.has(modelId)) return null; - const tierClientId = ANTIGRAVITY_QUOTA_BUCKET_TO_CLIENT[modelId]; - if (tierClientId) return tierClientId; - return toClientAntigravityModelId(modelId); -} - export function getClientVisibleAntigravityModelName( modelId: string, fallbackName?: string diff --git a/open-sse/config/freeModelCatalog.data.ts b/open-sse/config/freeModelCatalog.data.ts index 5b29278eced..d7d51609150 100644 --- a/open-sse/config/freeModelCatalog.data.ts +++ b/open-sse/config/freeModelCatalog.data.ts @@ -418,14 +418,9 @@ export const FREE_MODEL_BUDGETS: FreeModelBudget[] = [ { provider: "friendliai", modelId: "meta-llama-3.1-8b-instruct", displayName: "meta-llama-3.1-8b-instruct", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "friendliai", tos: "avoid" }, { provider: "ai21", modelId: "jamba-large-1.7", displayName: "jamba-large-1.7", monthlyTokens: 0, creditTokens: 10000000, freeType: "one-time-initial", poolKey: "ai21", tos: "avoid" }, { provider: "ai21", modelId: "jamba-mini-2", displayName: "jamba-mini-2", monthlyTokens: 0, creditTokens: 10000000, freeType: "one-time-initial", poolKey: "ai21", tos: "avoid" }, - { provider: "qwen-web", modelId: "qwen-plus", displayName: "Qwen Plus", monthlyTokens: 0, creditTokens: 0, freeType: "discontinued", poolKey: "qwen-web", tos: "avoid" }, - { provider: "qwen-web", modelId: "qwen-max", displayName: "Qwen Max", monthlyTokens: 0, creditTokens: 0, freeType: "discontinued", poolKey: "qwen-web", tos: "avoid" }, - { provider: "qwen-web", modelId: "qwen-turbo", displayName: "Qwen Turbo", monthlyTokens: 0, creditTokens: 0, freeType: "discontinued", poolKey: "qwen-web", tos: "avoid" }, - { provider: "qwen-web", modelId: "qwen3-plus", displayName: "Qwen3 Plus", monthlyTokens: 0, creditTokens: 0, freeType: "discontinued", poolKey: "qwen-web", tos: "avoid" }, - { provider: "qwen-web", modelId: "qwen3-max", displayName: "Qwen3 Max", monthlyTokens: 0, creditTokens: 0, freeType: "discontinued", poolKey: "qwen-web", tos: "avoid" }, - { provider: "qwen-web", modelId: "qwen3-flash", displayName: "Qwen3 Flash", monthlyTokens: 0, creditTokens: 0, freeType: "discontinued", poolKey: "qwen-web", tos: "avoid" }, - { provider: "qwen-web", modelId: "qwen3-coder-plus", displayName: "Qwen3 Coder Plus", monthlyTokens: 0, creditTokens: 0, freeType: "discontinued", poolKey: "qwen-web", tos: "avoid" }, - { provider: "qwen-web", modelId: "qwen3-coder-flash", displayName: "Qwen3 Coder Flash", monthlyTokens: 0, creditTokens: 0, freeType: "discontinued", poolKey: "qwen-web", tos: "avoid" }, + { provider: "qwen-web", modelId: "qwen3.7-max", displayName: "Qwen3.7 Max", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "qwen-web", tos: "avoid" }, + { provider: "qwen-web", modelId: "qwen3.7-plus", displayName: "Qwen3.7 Plus", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "qwen-web", tos: "avoid" }, + { provider: "qwen-web", modelId: "qwen3.6-plus", displayName: "Qwen3.6 Plus", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "qwen-web", tos: "avoid" }, { provider: "gitlawb", modelId: "mimo-v2.5-pro", displayName: "MiMo-V2.5-Pro", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "gitlawb", tos: "unknown" }, { provider: "gitlawb", modelId: "mimo-v2.5", displayName: "MiMo-V2.5", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "gitlawb", tos: "unknown" }, { provider: "gitlawb", modelId: "mimo-v2-pro", displayName: "MiMo-V2-Pro", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "gitlawb", tos: "unknown" }, diff --git a/open-sse/config/geminiRateLimits.json b/open-sse/config/geminiRateLimits.json new file mode 100644 index 00000000000..a1ae15dc7f0 --- /dev/null +++ b/open-sse/config/geminiRateLimits.json @@ -0,0 +1,34 @@ +{ + "gemini-2.5-flash": { "rpm": 5, "rpd": 20, "tpm": 250000 }, + "gemini-2.5-pro": { "rpm": 0, "rpd": 0, "tpm": 0 }, + "gemini-2-flash": { "rpm": 0, "rpd": 0, "tpm": 0 }, + "gemini-2-flash-lite": { "rpm": 0, "rpd": 0, "tpm": 0 }, + "gemini-2.5-flash-tts": { "rpm": 3, "rpd": 10, "tpm": 10000 }, + "gemini-2.5-pro-tts": { "rpm": 0, "rpd": 0, "tpm": 0 }, + "imagen-4-generate": { "rpm": -1, "rpd": 25, "tpm": -1 }, + "imagen-4-ultra-generate": { "rpm": -1, "rpd": 25, "tpm": -1 }, + "imagen-4-fast-generate": { "rpm": -1, "rpd": 25, "tpm": -1 }, + "gemma-4-26b-it": { "rpm": 15, "rpd": 1500, "tpm": -1 }, + "gemma-4-31b-it": { "rpm": 15, "rpd": 1500, "tpm": -1 }, + "gemini-embedding-exp-03-07": { "rpm": 100, "rpd": 1000, "tpm": 30000 }, + "gemini-3.5-flash": { "rpm": 5, "rpd": 20, "tpm": 250000 }, + "gemini-3.1-flash-lite": { "rpm": 15, "rpd": 500, "tpm": 250000 }, + "gemini-3.1-pro": { "rpm": 0, "rpd": 0, "tpm": 0 }, + "gemini-2.5-flash-lite": { "rpm": 10, "rpd": 20, "tpm": 250000 }, + "nano-banana-gemini-2.5-flash-preview-image": { "rpm": 0, "rpd": 0, "tpm": 0 }, + "nano-banana-pro-gemini-3-pro-image": { "rpm": 0, "rpd": 0, "tpm": 0 }, + "nano-banana-2-gemini-3.1-flash-image": { "rpm": 0, "rpd": 0, "tpm": 0 }, + "lyria-3-clip": { "rpm": 0, "rpd": 0, "tpm": 0 }, + "lyria-3-pro": { "rpm": 0, "rpd": 0, "tpm": 0 }, + "veo-3-generate": { "rpm": 0, "rpd": 0, "tpm": -1 }, + "veo-3-fast-generate": { "rpm": 0, "rpd": 0, "tpm": -1 }, + "veo-3-lite-generate": { "rpm": 0, "rpd": 0, "tpm": -1 }, + "gemini-3.1-flash-tts": { "rpm": 3, "rpd": 10, "tpm": 10000 }, + "gemini-robotics-er-1.5-preview": { "rpm": 10, "rpd": 20, "tpm": 250000 }, + "gemini-robotics-er-1.6-preview": { "rpm": 5, "rpd": 20, "tpm": 250000 }, + "computer-use-preview": { "rpm": 0, "rpd": 0, "tpm": 0 }, + "gemini-embedding-exp-04-07": { "rpm": 100, "rpd": 1000, "tpm": 30000 }, + "gemini-3.5-live-translate": { "rpm": -1, "rpd": -1, "tpm": 20000 }, + "gemini-2.5-flash-native-audio-dialog": { "rpm": -1, "rpd": -1, "tpm": 1000000 }, + "gemini-3-flash-live": { "rpm": -1, "rpd": -1, "tpm": 65000 } +} diff --git a/open-sse/config/providerRegistry.ts b/open-sse/config/providerRegistry.ts index d017b25a18c..27bc46650fd 100644 --- a/open-sse/config/providerRegistry.ts +++ b/open-sse/config/providerRegistry.ts @@ -4011,18 +4011,17 @@ const _REGISTRY_EAGER: Record = { alias: "qwen-web", format: "openai", executor: "qwen-web", - baseUrl: "https://chat.qwen.ai/api/chat/completions", + // v2 API (the legacy /api/chat/completions endpoint was retired upstream). + baseUrl: "https://chat.qwen.ai/api/v2/chat/completions", authType: "apikey", authHeader: "bearer", + // Current upstream catalog (GET https://chat.qwen.ai/api/models). Legacy + // ids (qwen-plus, qwen3-max, ...) still resolve via the executor's + // MODEL_ALIASES map for backward compatibility. models: [ - { id: "qwen-plus", name: "Qwen Plus" }, - { id: "qwen-max", name: "Qwen Max" }, - { id: "qwen-turbo", name: "Qwen Turbo" }, - { id: "qwen3-plus", name: "Qwen3 Plus" }, - { id: "qwen3-max", name: "Qwen3 Max" }, - { id: "qwen3-flash", name: "Qwen3 Flash" }, - { id: "qwen3-coder-plus", name: "Qwen3 Coder Plus" }, - { id: "qwen3-coder-flash", name: "Qwen3 Coder Flash" }, + { id: "qwen3.7-max", name: "Qwen3.7 Max" }, + { id: "qwen3.7-plus", name: "Qwen3.7 Plus" }, + { id: "qwen3.6-plus", name: "Qwen3.6 Plus" }, ], }, diff --git a/open-sse/executors/mimocode.ts b/open-sse/executors/mimocode.ts index d75d24174ff..00bc5061b61 100644 --- a/open-sse/executors/mimocode.ts +++ b/open-sse/executors/mimocode.ts @@ -28,6 +28,42 @@ const COOLDOWN_MAX_MS = 60_000; const MIMO_SOURCE = "mimocode-cli-free"; +/** + * Anti-abuse gate marker required by the Xiaomi free endpoint. + * + * `/api/free-ai/openai/chat` returns `403 "Illegal access"` unless the request body + * contains a recognized MiMoCode prompt signature as a substring inside a `system`-role + * message (verified empirically — headers, fingerprint, and JWT are not what is checked). + * This is the canonical MiMoCode agent opener the official CLI sends, and it is on the + * upstream allowlist. We inject it as a leading system message so user requests pass the + * gate. The string MUST stay byte-for-byte identical — the check is case-sensitive and + * truncations are rejected. + */ +export const MIMO_SYSTEM_MARKER = + "You are MiMoCode, an interactive CLI tool that helps users with software engineering tasks."; + +/** + * Ensure the outgoing body carries the MiMoCode anti-abuse marker in a system message. + * Idempotent: if any system message already contains the marker, the body is returned + * unchanged. Bodies without a `messages` array are left untouched. + */ +function injectSystemMarker(body: Record): Record { + const messages = body.messages; + if (!Array.isArray(messages)) return body; + + const hasMarker = messages.some( + (m) => + m != null && + typeof m === "object" && + (m as { role?: unknown }).role === "system" && + typeof (m as { content?: unknown }).content === "string" && + (m as { content: string }).content.includes(MIMO_SYSTEM_MARKER) + ); + if (hasMarker) return body; + + return { ...body, messages: [{ role: "system", content: MIMO_SYSTEM_MARKER }, ...messages] }; +} + const USER_AGENTS = [ "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36", "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36", @@ -67,7 +103,9 @@ function getCpuModel(): string { try { const cpus = os.cpus(); if (cpus.length > 0 && cpus[0].model) return cpus[0].model.trim(); - } catch { /* ignore */ } + } catch { + /* ignore */ + } return "unknown-cpu"; } @@ -80,8 +118,13 @@ export function generateFingerprint(seed?: string): string { let username = "unknown-user"; try { username = os.userInfo().username; - } catch { /* ignore */ } - return crypto.createHash("sha256").update(`${hostname}|${platform}|${arch}|${cpu}|${username}`).digest("hex"); + } catch { + /* ignore */ + } + return crypto + .createHash("sha256") + .update(`${hostname}|${platform}|${arch}|${cpu}|${username}`) + .digest("hex"); } // ── Bootstrap ────────────────────────────────────────────────────────────── @@ -91,7 +134,7 @@ const bootstrapInflight = new Map { const existing = bootstrapInflight.get(fingerprint); if (existing) return existing; @@ -146,7 +189,13 @@ export class MimocodeExecutor extends BaseExecutor { constructor() { super("mimocode", { format: "openai" }); this.baseUrl = this.getBaseUrls()[0] || "https://api.xiaomimimo.com"; - this.accounts.push({ fingerprint: generateFingerprint(), jwt: "", expiresAt: 0, cooldownUntil: 0, consecutiveFails: 0 }); + this.accounts.push({ + fingerprint: generateFingerprint(), + jwt: "", + expiresAt: 0, + cooldownUntil: 0, + consecutiveFails: 0, + }); } private syncAccountsFromCredentials(credentials: ProviderCredentials): void { @@ -155,13 +204,22 @@ export class MimocodeExecutor extends BaseExecutor { const existing = new Set(this.accounts.map((a) => a.fingerprint)); for (const fp of fingerprints) { if (typeof fp === "string" && !existing.has(fp)) { - this.accounts.push({ fingerprint: fp, jwt: "", expiresAt: 0, cooldownUntil: 0, consecutiveFails: 0 }); + this.accounts.push({ + fingerprint: fp, + jwt: "", + expiresAt: 0, + cooldownUntil: 0, + consecutiveFails: 0, + }); existing.add(fp); } } } - private async getJwtForAccount(account: AccountState, signal?: AbortSignal | null): Promise { + private async getJwtForAccount( + account: AccountState, + signal?: AbortSignal | null + ): Promise { if (isAccountReady(account)) return account.jwt; const result = await bootstrapJwt(this.baseUrl, account.fingerprint, signal); account.jwt = result.jwt; @@ -185,7 +243,10 @@ export class MimocodeExecutor extends BaseExecutor { private markCooldown(account: AccountState): void { account.consecutiveFails++; - const backoff = Math.min(COOLDOWN_BASE_MS * Math.pow(2, account.consecutiveFails - 1), COOLDOWN_MAX_MS); + const backoff = Math.min( + COOLDOWN_BASE_MS * Math.pow(2, account.consecutiveFails - 1), + COOLDOWN_MAX_MS + ); account.cooldownUntil = Date.now() + backoff + Math.random() * 1000; } @@ -193,7 +254,12 @@ export class MimocodeExecutor extends BaseExecutor { account.consecutiveFails = 0; } - buildUrl(_model: string, _stream: boolean, _urlIndex = 0, _credentials?: ProviderCredentials | null): string { + buildUrl( + _model: string, + _stream: boolean, + _urlIndex = 0, + _credentials?: ProviderCredentials | null + ): string { return `${this.baseUrl.replace(/\/$/, "")}${CHAT_PATH}`; } @@ -201,7 +267,7 @@ export class MimocodeExecutor extends BaseExecutor { _credentials: ProviderCredentials, stream = true, _clientHeaders?: Record | null, - _model?: string, + _model?: string ): Record { const headers: Record = { "Content-Type": "application/json", @@ -212,9 +278,15 @@ export class MimocodeExecutor extends BaseExecutor { return headers; } - transformRequest(model: string, body: unknown, _stream: boolean, _credentials?: ProviderCredentials | null): unknown { + transformRequest( + model: string, + body: unknown, + _stream: boolean, + _credentials?: ProviderCredentials | null + ): unknown { if (typeof body === "object" && body !== null) { - return { ...(body as Record), model: rewriteModelName(model) }; + const withModel = { ...(body as Record), model: rewriteModelName(model) }; + return injectSystemMarker(withModel); } return body; } @@ -222,15 +294,25 @@ export class MimocodeExecutor extends BaseExecutor { async testConnection( _credentials: ProviderCredentials, _signal?: AbortSignal | null, - log?: ExecuteInput["log"], + log?: ExecuteInput["log"] ): Promise { try { const account = this.accounts[0]; const jwt = await this.getJwtForAccount(account, _signal); const resp = await fetch(this.buildUrl("mimo-auto", false), { method: "POST", - headers: { "Content-Type": "application/json", Authorization: `Bearer ${jwt}`, "X-Mimo-Source": MIMO_SOURCE }, - body: JSON.stringify({ model: "mimo-auto", messages: [{ role: "user", content: "ping" }], stream: false }), + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${jwt}`, + "X-Mimo-Source": MIMO_SOURCE, + }, + body: JSON.stringify( + injectSystemMarker({ + model: "mimo-auto", + messages: [{ role: "user", content: "ping" }], + stream: false, + }) + ), signal: _signal ?? undefined, }); return resp.status === 200; @@ -251,9 +333,14 @@ export class MimocodeExecutor extends BaseExecutor { if (signal?.aborted) { return { - response: new Response(encoder.encode(JSON.stringify({ - error: { message: "Request aborted", type: "abort", code: "ABORTED" }, - })), { status: 499, headers: { "Content-Type": "application/json" } }), + response: new Response( + encoder.encode( + JSON.stringify({ + error: { message: "Request aborted", type: "abort", code: "ABORTED" }, + }) + ), + { status: 499, headers: { "Content-Type": "application/json" } } + ), url: this.buildUrl(model, stream), headers: this.buildHeaders(input.credentials, stream), transformedBody: body, @@ -282,45 +369,81 @@ export class MimocodeExecutor extends BaseExecutor { // On auth failure, re-bootstrap this account and retry once if (resp.status === 401 || resp.status === 403) { - log?.warn?.("MIMOCODE", `Auth failed (${resp.status}) on account ${account.fingerprint.slice(0, 8)}…`); + log?.warn?.( + "MIMOCODE", + `Auth failed (${resp.status}) on account ${account.fingerprint.slice(0, 8)}…` + ); account.jwt = ""; account.expiresAt = 0; account.consecutiveFails = 0; const freshJwt = await this.getJwtForAccount(account, signal); headers["Authorization"] = `Bearer ${freshJwt}`; - resp = await fetch(url, { method: "POST", headers, body: JSON.stringify(reqBody), signal: signal ?? undefined }); + resp = await fetch(url, { + method: "POST", + headers, + body: JSON.stringify(reqBody), + signal: signal ?? undefined, + }); } if (resp.status === 429) { this.markCooldown(account); - log?.warn?.("MIMOCODE", `Rate limited on account ${account.fingerprint.slice(0, 8)}, trying next…`); + log?.warn?.( + "MIMOCODE", + `Rate limited on account ${account.fingerprint.slice(0, 8)}, trying next…` + ); continue; } this.markSuccess(account); const respHeaders: Record = {}; - resp.headers.forEach((v, k) => { respHeaders[k] = v; }); - return { response: resp as unknown as Response, url, headers: respHeaders, transformedBody: reqBody }; + resp.headers.forEach((v, k) => { + respHeaders[k] = v; + }); + return { + response: resp as unknown as Response, + url, + headers: respHeaders, + transformedBody: reqBody, + }; } catch (err) { this.markCooldown(account); if (attempt === this.accounts.length - 1) { const msg = err instanceof Error ? err.message : String(err); log?.error?.("MIMOCODE", `Executor error: ${msg}`); return { - response: new Response(encoder.encode(JSON.stringify({ - error: { message: msg, type: "upstream_error", code: "EXECUTOR_ERROR" }, - })), { status: 502, headers: { "Content-Type": "application/json" } }), - url, headers: this.buildHeaders(input.credentials, stream), transformedBody: body, + response: new Response( + encoder.encode( + JSON.stringify({ + error: { message: msg, type: "upstream_error", code: "EXECUTOR_ERROR" }, + }) + ), + { status: 502, headers: { "Content-Type": "application/json" } } + ), + url, + headers: this.buildHeaders(input.credentials, stream), + transformedBody: body, }; } } } return { - response: new Response(encoder.encode(JSON.stringify({ - error: { message: "All accounts exhausted", type: "upstream_error", code: "NO_ACCOUNTS" }, - })), { status: 502, headers: { "Content-Type": "application/json" } }), - url, headers: this.buildHeaders(input.credentials, stream), transformedBody: body, + response: new Response( + encoder.encode( + JSON.stringify({ + error: { + message: "All accounts exhausted", + type: "upstream_error", + code: "NO_ACCOUNTS", + }, + }) + ), + { status: 502, headers: { "Content-Type": "application/json" } } + ), + url, + headers: this.buildHeaders(input.credentials, stream), + transformedBody: body, }; } } diff --git a/open-sse/executors/qwen-web.ts b/open-sse/executors/qwen-web.ts index 67c7f50c1e0..82db1baebb2 100644 --- a/open-sse/executors/qwen-web.ts +++ b/open-sse/executors/qwen-web.ts @@ -1,173 +1,367 @@ /** - * QwenWebExecutor — Alibaba Tongyi Qwen Chat via chat.qwen.ai + * QwenWebExecutor — Alibaba Tongyi Qwen Chat via chat.qwen.ai (v2 API) * - * Routes requests through Qwen's consumer chat API. - * Chinese market provider with strong vision, coding, and reasoning models. + * Routes requests through Qwen's consumer chat API. The legacy v1 endpoint + * (`/api/chat/completions`) was retired upstream in 2026 and now answers 504 + * HTML from Alibaba's gateway for every request, regardless of credentials + * (#3288 / discussion #2768). The current contract is a two-step v2 flow: * - * Auth: Token from chat.qwen.ai Local Storage or tongyi_sso_ticket cookie - * Endpoint: POST https://chat.qwen.ai/api/chat/completions - * Format: OpenAI-compatible + * 1. POST /api/v2/chats/new → create a chat, returns chat_id + * 2. POST /api/v2/chat/completions?chat_id= → phase-based SSE stream + * + * The v2 endpoints sit behind Alibaba's "baxia" WAF, which requires the full + * browser cookie jar from a real logged-in session (cna, ssxmod_itna, + * ssxmod_itna2, token, ...). We therefore replay the captured/pasted Cookie + * header verbatim plus the bearer token, mirroring how grok-web replays its + * anti-bot cookies. + * + * SSE chunks carry `choices[0].delta` with a `phase` field: `think` / + * `thinking_summary` map to reasoning, `answer` (or a null phase) carries the + * assistant content. + * + * Reference implementations: gpt4free `g4f/Provider/Qwen.py`, + * Chat2API `proxy/adapters/qwen-ai.ts`. + * + * Auth: full Cookie header from chat.qwen.ai + bearer token (localStorage + * `token`, also mirrored to a `token` cookie). + * Format: OpenAI-compatible (translated from Qwen's phase protocol). */ import { BaseExecutor, type ExecuteInput } from "./base.ts"; import { makeExecutorErrorResult as makeErrorResult } from "../utils/error.ts"; import { prepareToolMessages, buildToolAwareResult } from "../translator/webTools.ts"; +import { buildQwenCookieHeader, extractQwenToken } from "@/lib/providers/webCookieAuth"; const BASE_URL = "https://chat.qwen.ai"; -const CHAT_URL = `${BASE_URL}/api/chat/completions`; +const CHATS_NEW_URL = `${BASE_URL}/api/v2/chats/new`; +const CHAT_COMPLETIONS_URL = `${BASE_URL}/api/v2/chat/completions`; const USER_AGENT = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36"; +// Anti-bot headers the v2 endpoint expects. `bx-umidtoken` is normally minted +// per-session from sg-wum.alibaba.com; a captured value travels with the cookie +// jar, but we also send a static fallback so the header is always present. +const BX_VERSION = "2.5.36"; +const BX_UMIDTOKEN_FALLBACK = "T2gA0000000000000000000000000000000000000000"; + +const MODEL_ALIASES: Record = { + // Legacy OmniRoute ids → current upstream catalog (GET /api/models). + "qwen-plus": "qwen3.7-plus", + "qwen-max": "qwen3.7-max", + "qwen-turbo": "qwen3.6-plus", + "qwen3-plus": "qwen3.7-plus", + "qwen3-max": "qwen3.7-max", + "qwen3-flash": "qwen3.6-plus", + "qwen3-coder-plus": "qwen3.7-max", + "qwen3-coder-flash": "qwen3.6-plus", + qwen: "qwen3.7-max", + qwen3: "qwen3.7-max", +}; + +const DEFAULT_MODEL = "qwen3.7-max"; + +function mapModel(modelId: string): string { + return MODEL_ALIASES[modelId] || modelId; +} + +function uuid(): string { + return crypto.randomUUID(); +} + +/** Detect Alibaba's WAF / retired-v1 gateway page so we never surface raw HTML. */ +function isWafResponse(status: number, contentType: string, bodyText: string): boolean { + if (contentType.includes("text/html")) return true; + if (status === 504) return true; + return /aliyun_waf|baxia| { + const headers: Record = { + "Content-Type": "application/json", + Accept: "*/*", + "User-Agent": USER_AGENT, + Origin: BASE_URL, + Referer: chatId ? `${BASE_URL}/c/${chatId}` : `${BASE_URL}/`, + source: "web", + "x-request-id": uuid(), + "bx-v": BX_VERSION, + "bx-umidtoken": BX_UMIDTOKEN_FALLBACK, + }; + if (token) headers["Authorization"] = `Bearer ${token}`; + if (cookieHeader) headers["Cookie"] = cookieHeader; + return headers; + } + async execute(input: ExecuteInput) { const { body, credentials, signal, stream: wantStream } = input; const bodyObj = (body || {}) as Record; - const rawToken = String(credentials?.apiKey ?? credentials?.accessToken ?? "").trim(); + + const rawCred = String(credentials?.apiKey ?? "").trim(); + const cookieHeader = buildQwenCookieHeader(rawCred); + let token = extractQwenToken(rawCred); + if (!token && credentials?.accessToken) token = String(credentials.accessToken).trim(); const messages = (bodyObj.messages as Array<{ role: string; content: string }>) || []; - const modelId = (bodyObj.model as string) || "qwen-plus"; + const requestedModel = (bodyObj.model as string) || DEFAULT_MODEL; + const modelId = mapModel(requestedModel); const { hasTools, requestedTools, effectiveMessages } = prepareToolMessages(bodyObj, messages); - const reqBody = { - messages: effectiveMessages.map((m) => ({ role: m.role, content: String(m.content ?? "") })), - model: modelId, - stream: wantStream, - max_tokens: (bodyObj.max_tokens as number) || 4096, - }; + // Qwen Web is single-turn: fold the conversation into one user prompt. + const prompt = this.foldMessages(effectiveMessages); - const reqHeaders: Record = { - "Content-Type": "application/json", - "User-Agent": USER_AGENT, - Accept: wantStream ? "text/event-stream" : "application/json", - Referer: `${BASE_URL}/`, - Origin: BASE_URL, - }; - if (rawToken) { - reqHeaders["Authorization"] = `Bearer ${rawToken}`; + // ── Step 1: create a chat ──────────────────────────────────────────────── + let chatId: string; + try { + const newChatRes = await fetch(CHATS_NEW_URL, { + method: "POST", + headers: this.buildHeaders(token, cookieHeader), + body: JSON.stringify({ + title: "New Chat", + models: [modelId], + chat_mode: "normal", + chat_type: "t2t", + timestamp: Date.now(), + }), + signal, + }); + + const ct = newChatRes.headers.get("content-type") || ""; + if (!newChatRes.ok || ct.includes("text/html")) { + const text = await newChatRes.text().catch(() => ""); + if (isWafResponse(newChatRes.status, ct, text)) { + return makeErrorResult(401, WAF_ERROR_MESSAGE, body, CHATS_NEW_URL); + } + return makeErrorResult( + newChatRes.status || 502, + `Qwen create-chat failed: ${text.slice(0, 300)}`, + body, + CHATS_NEW_URL + ); + } + + const data = (await newChatRes.json()) as { data?: { id?: string } }; + chatId = data?.data?.id ?? ""; + if (!chatId) { + return makeErrorResult(502, "Qwen create-chat returned no chat id", body, CHATS_NEW_URL); + } + } catch (err) { + return makeErrorResult( + 502, + `Qwen create-chat error: ${err instanceof Error ? err.message : "unknown"}`, + body, + CHATS_NEW_URL + ); } + // ── Step 2: send the message ───────────────────────────────────────────── + const completionUrl = `${CHAT_COMPLETIONS_URL}?chat_id=${chatId}`; + const msgPayload = this.buildMessagePayload(chatId, modelId, prompt, requestedModel); + let upstream: Response; try { - upstream = await fetch(CHAT_URL, { + upstream = await fetch(completionUrl, { method: "POST", - headers: reqHeaders, - body: JSON.stringify(reqBody), + headers: this.buildHeaders(token, cookieHeader, chatId), + body: JSON.stringify(msgPayload), signal, }); } catch (err) { return makeErrorResult( 502, - `Qwen fetch failed: ${err instanceof Error ? err.message : "unknown"}`, + `Qwen completion fetch failed: ${err instanceof Error ? err.message : "unknown"}`, body, - CHAT_URL + completionUrl ); } - if (!upstream.ok) { + const ct = upstream.headers.get("content-type") || ""; + if (!upstream.ok || ct.includes("text/html")) { const errText = await upstream.text().catch(() => ""); - if (upstream.status === 401) { - return makeErrorResult( - 401, - "Qwen authentication failed. Your token may have expired. " + - "Get a fresh token from chat.qwen.ai (DevTools → Application → Local Storage → token)", - body, - CHAT_URL - ); + if (isWafResponse(upstream.status, ct, errText)) { + return makeErrorResult(401, WAF_ERROR_MESSAGE, body, completionUrl); } - return makeErrorResult(upstream.status, `Qwen error: ${errText}`, body, CHAT_URL); + return makeErrorResult( + upstream.status || 502, + `Qwen error: ${errText.slice(0, 300)}`, + body, + completionUrl + ); } if (!wantStream) { - const data = (await upstream.json()) as Record; - const rawContent = - (data?.choices as Array<{ message?: { content?: string } }>)?.[0]?.message?.content || - (data?.content as string) || - ""; + const { content } = await this.collectStream(upstream); + const finalText = content; if (hasTools) { - const { content, toolCalls, finishReason } = buildToolAwareResult(rawContent, requestedTools, "qwen"); - const message: Record = { role: "assistant", content }; - if (toolCalls) { message.tool_calls = toolCalls; message.content = null; } - return { - response: new Response( - JSON.stringify({ - id: `chatcmpl-qwen-${Date.now()}`, object: "chat.completion", - created: Math.floor(Date.now() / 1000), model: modelId, - choices: [{ index: 0, message, finish_reason: finishReason }], - }), - { headers: { "Content-Type": "application/json" } } - ), - url: CHAT_URL, headers: reqHeaders, transformedBody: reqBody, - }; + const { + content: toolContent, + toolCalls, + finishReason, + } = buildToolAwareResult(finalText, requestedTools, "qwen"); + const message: Record = { role: "assistant", content: toolContent }; + if (toolCalls) { + message.tool_calls = toolCalls; + message.content = null; + } + return this.jsonResponse(modelId, message, finishReason, completionUrl, msgPayload); } - return { - response: new Response( - JSON.stringify({ - id: `chatcmpl-qwen-${Date.now()}`, object: "chat.completion", - created: Math.floor(Date.now() / 1000), model: modelId, - choices: [{ index: 0, message: { role: "assistant", content: rawContent }, finish_reason: "stop" }], - }), - { headers: { "Content-Type": "application/json" } } - ), - url: CHAT_URL, headers: reqHeaders, transformedBody: reqBody, - }; + return this.jsonResponse( + modelId, + { role: "assistant", content: finalText }, + "stop", + completionUrl, + msgPayload + ); } - // Streaming - const encoder = new TextEncoder(); - const decoder = new TextDecoder(); + // Streaming: transform Qwen phase SSE → OpenAI chat.completion.chunk SSE. + const stream = this.buildClientStream(upstream, modelId, hasTools, requestedTools, signal); + return { + response: new Response(stream, { + headers: { + "Content-Type": "text/event-stream", + "Cache-Control": "no-cache", + Connection: "keep-alive", + }, + }), + url: completionUrl, + headers: this.buildHeaders(token, cookieHeader, chatId), + transformedBody: msgPayload, + }; + } - if (hasTools) { - let fullContent = ""; - const reader = upstream.body?.getReader(); - if (reader) { - let buf = ""; - try { - while (true) { - const { done, value } = await reader.read(); - if (done) break; - buf += decoder.decode(value, { stream: true }); - for (const line of buf.split("\n")) { - if (!line.startsWith("data:")) continue; - const d = line.slice(5).trim(); - if (d === "[DONE]") continue; - try { fullContent += JSON.parse(d).choices?.[0]?.delta?.content || ""; } catch {} - } - buf = buf.split("\n").pop() || ""; - } - } catch {} + private foldMessages(messages: Array<{ role: string; content: unknown }>): string { + let systemContent = ""; + let userContent = ""; + for (const m of messages) { + const text = String(m.content ?? ""); + if (m.role === "system") { + systemContent += (systemContent ? "\n\n" : "") + text; + } else if (m.role === "user") { + userContent = text; } + } + return systemContent ? `${systemContent}\n\nUser: ${userContent}` : userContent; + } - const { content, toolCalls, finishReason } = buildToolAwareResult(fullContent, requestedTools, "qwen"); - const stream = new ReadableStream({ - start(controller) { - const id = `chatcmpl-qwen-${Date.now()}`; - const created = Math.floor(Date.now() / 1000); - const delta = toolCalls - ? { role: "assistant", content: null, tool_calls: toolCalls } - : { role: "assistant", content }; - controller.enqueue(encoder.encode(`data: ${JSON.stringify({ id, object: "chat.completion.chunk", created, model: modelId, choices: [{ index: 0, delta, finish_reason: null }] })}\n\n`)); - controller.enqueue(encoder.encode(`data: ${JSON.stringify({ id, object: "chat.completion.chunk", created, model: modelId, choices: [{ index: 0, delta: {}, finish_reason: finishReason }] })}\n\n`)); - controller.enqueue(encoder.encode("data: [DONE]\n\n")); - controller.close(); + private buildMessagePayload( + chatId: string, + modelId: string, + prompt: string, + requestedModel: string + ): Record { + const fid = uuid(); + const enableThinking = /think|reason|r1/i.test(requestedModel); + const featureConfig: Record = { + thinking_enabled: enableThinking, + output_schema: "phase", + auto_thinking: enableThinking, + research_mode: "normal", + auto_search: false, + }; + return { + stream: true, + incremental_output: true, + chat_id: chatId, + chat_mode: "normal", + model: modelId, + parent_id: null, + messages: [ + { + fid, + parentId: null, + childrenIds: [], + role: "user", + content: prompt, + user_action: "chat", + files: [], + timestamp: Math.floor(Date.now() / 1000), + models: [modelId], + chat_type: "t2t", + feature_config: featureConfig, + sub_chat_type: "t2t", + parent_id: null, }, - }); - return { - response: new Response(stream, { headers: { "Content-Type": "text/event-stream", "Cache-Control": "no-cache", Connection: "keep-alive" } }), - url: CHAT_URL, headers: reqHeaders, transformedBody: reqBody, - }; + ], + }; + } + + /** Read the whole upstream SSE stream, returning the joined answer + reasoning. */ + private async collectStream(upstream: Response): Promise<{ content: string; reasoning: string }> { + const reader = upstream.body?.getReader(); + const decoder = new TextDecoder(); + let content = ""; + let reasoning = ""; + if (!reader) return { content, reasoning }; + + let buffer = ""; + try { + while (true) { + const { done, value } = await reader.read(); + if (done) break; + buffer += decoder.decode(value, { stream: true }); + const lines = buffer.split("\n"); + buffer = lines.pop() || ""; + for (const line of lines) { + const delta = parseSseDelta(line); + if (!delta) continue; + if (delta.kind === "answer") content += delta.text; + else if (delta.kind === "think") reasoning += delta.text; + } + } + } catch { + /* upstream closed mid-stream — return what we have */ } + return { content, reasoning }; + } + + /** Transform the Qwen phase SSE into OpenAI chat.completion.chunk SSE. */ + private buildClientStream( + upstream: Response, + modelId: string, + hasTools: boolean, + requestedTools: unknown, + signal: AbortSignal | null | undefined + ): ReadableStream { + const encoder = new TextEncoder(); + const decoder = new TextDecoder(); + const id = `chatcmpl-qwen-${Date.now()}`; + const created = Math.floor(Date.now() / 1000); + const emitChunk = (delta: Record, finishReason: string | null) => + `data: ${JSON.stringify({ + id, + object: "chat.completion.chunk", + created, + model: modelId, + choices: [{ index: 0, delta, finish_reason: finishReason }], + })}\n\n`; - const stream = new ReadableStream({ + return new ReadableStream({ async start(controller) { const reader = upstream.body?.getReader(); - if (!reader) { controller.close(); return; } + if (!reader) { + controller.enqueue(encoder.encode("data: [DONE]\n\n")); + controller.close(); + return; + } let buffer = ""; + let fullContent = ""; + controller.enqueue(encoder.encode(emitChunk({ role: "assistant", content: "" }, null))); try { while (true) { const { done, value } = await reader.read(); @@ -176,43 +370,95 @@ export class QwenWebExecutor extends BaseExecutor { const lines = buffer.split("\n"); buffer = lines.pop() || ""; for (const line of lines) { - if (!line.startsWith("data:")) continue; - const data = line.slice(5).trim(); - if (data === "[DONE]") { controller.enqueue(encoder.encode("data: [DONE]\n\n")); continue; } - try { - const parsed = JSON.parse(data); - const text = parsed.choices?.[0]?.delta?.content || parsed.choices?.[0]?.text || ""; - if (text) { - const chunk = { - id: `chatcmpl-qwen-${Date.now()}`, - object: "chat.completion.chunk", - created: Math.floor(Date.now() / 1000), - model: modelId, - choices: [{ index: 0, delta: { content: text }, finish_reason: null }], - }; - controller.enqueue(encoder.encode(`data: ${JSON.stringify(chunk)}\n\n`)); + const delta = parseSseDelta(line); + if (!delta || !delta.text) continue; + if (delta.kind === "answer") { + fullContent += delta.text; + if (!hasTools) { + controller.enqueue(encoder.encode(emitChunk({ content: delta.text }, null))); } - } catch { - /* skip unparseable chunks */ + } else if (delta.kind === "think" && !hasTools) { + controller.enqueue( + encoder.encode(emitChunk({ reasoning_content: delta.text }, null)) + ); } } } } catch (err) { - if (!signal?.aborted) controller.error(err); - } finally { - controller.enqueue(encoder.encode("data: [DONE]\n\n")); - controller.close(); + if (!signal?.aborted) { + controller.error(err); + return; + } } + + if (hasTools) { + const { content, toolCalls, finishReason } = buildToolAwareResult( + fullContent, + requestedTools, + "qwen" + ); + const delta = toolCalls + ? { role: "assistant", content: null, tool_calls: toolCalls } + : { role: "assistant", content }; + controller.enqueue(encoder.encode(emitChunk(delta, null))); + controller.enqueue(encoder.encode(emitChunk({}, finishReason))); + } else { + controller.enqueue(encoder.encode(emitChunk({}, "stop"))); + } + controller.enqueue(encoder.encode("data: [DONE]\n\n")); + controller.close(); }, }); + } + private jsonResponse( + modelId: string, + message: Record, + finishReason: string, + url: string, + transformedBody: unknown + ) { return { - response: new Response(stream, { - headers: { "Content-Type": "text/event-stream", "Cache-Control": "no-cache", Connection: "keep-alive" }, - }), - url: CHAT_URL, - headers: reqHeaders, - transformedBody: reqBody, + response: new Response( + JSON.stringify({ + id: `chatcmpl-qwen-${Date.now()}`, + object: "chat.completion", + created: Math.floor(Date.now() / 1000), + model: modelId, + choices: [{ index: 0, message, finish_reason: finishReason }], + }), + { headers: { "Content-Type": "application/json" } } + ), + url, + headers: {} as Record, + transformedBody, }; } } + +/** Parse one SSE line into a typed delta, or null if it carries no content. */ +function parseSseDelta(line: string): { kind: "answer" | "think"; text: string } | null { + if (!line.startsWith("data:")) return null; + const payload = line.slice(5).trim(); + if (!payload || payload === "[DONE]") return null; + let parsed: { + choices?: Array<{ delta?: { phase?: string | null; content?: unknown } }>; + }; + try { + parsed = JSON.parse(payload); + } catch { + return null; + } + const delta = parsed?.choices?.[0]?.delta; + if (!delta) return null; + const phase = delta.phase; + const content = typeof delta.content === "string" ? delta.content : ""; + if (phase === "think" || phase === "thinking_summary") { + return { kind: "think", text: content }; + } + // `answer` phase or a null/absent phase both carry assistant content. + if (phase === "answer" || phase === null || phase === undefined) { + return { kind: "answer", text: content }; + } + return null; +} diff --git a/open-sse/executors/vertex.ts b/open-sse/executors/vertex.ts index 7412fa24b94..02994026d05 100644 --- a/open-sse/executors/vertex.ts +++ b/open-sse/executors/vertex.ts @@ -21,6 +21,28 @@ export function parseSAFromApiKey(apiKey: string): ServiceAccount { } } +/** + * A Service Account credential is a JSON object (type/client_email/private_key). A Vertex AI + * Express-mode API key is an opaque non-JSON string. Distinguishing them lets the executor + * support BOTH: Service Account JSON (JWT → OAuth → project-scoped endpoint + Bearer auth) and + * Express keys (project-less publisher endpoint + x-goog-api-key auth), instead of failing every + * Express key with "requires a valid Service Account JSON". + */ +export function looksLikeServiceAccountJson(apiKey: string): boolean { + if (!apiKey || typeof apiKey !== "string") return false; + try { + const parsed = JSON.parse(apiKey); + return !!parsed && typeof parsed === "object" && !Array.isArray(parsed); + } catch { + return false; + } +} + +/** True for a Vertex AI Express-mode API key (a non-empty, non-JSON, non-OAuth credential). */ +export function isExpressApiKey(apiKey?: string | null): boolean { + return typeof apiKey === "string" && apiKey.trim().length > 0 && !looksLikeServiceAccountJson(apiKey); +} + export async function getAccessToken(sa: ServiceAccount): Promise { if (!sa.client_email || !sa.private_key) { throw new Error( @@ -110,7 +132,13 @@ export class VertexExecutor extends BaseExecutor { async execute(input: ExecuteInput) { const { credentials, log } = input; - if (credentials.apiKey && !credentials.accessToken) { + // Defensive: trim stray surrounding whitespace from a pasted credential. + if (typeof credentials.apiKey === "string") { + credentials.apiKey = credentials.apiKey.trim(); + } + // Service Account JSON → mint a short-lived OAuth token (Bearer). An Express-mode API key is + // sent as-is via x-goog-api-key (see buildHeaders), so no token exchange is needed for it. + if (credentials.apiKey && !credentials.accessToken && looksLikeServiceAccountJson(credentials.apiKey)) { try { const sa = parseSAFromApiKey(credentials.apiKey); credentials.accessToken = await getAccessToken(sa); @@ -123,6 +151,19 @@ export class VertexExecutor extends BaseExecutor { } buildUrl(model: string, stream: boolean, urlIndex = 0, credentials: any = null) { + // Vertex AI Express mode: project-less v1 publisher endpoint with the API key passed as a + // ?key= query parameter (verified working contract — same as the CaptionAI GeminiClient). The + // Express key is NOT accepted as a Bearer/OAuth credential or via x-goog-api-key on this API. + if (isExpressApiKey(credentials?.apiKey) && !credentials?.accessToken) { + const expressKey = encodeURIComponent(String(credentials.apiKey).trim()); + if (isPartnerModel(model)) { + // Partner (Anthropic/etc.) models are not available via Express keys; best-effort. + return `https://aiplatform.googleapis.com/v1/publishers/openapi/chat/completions?key=${expressKey}`; + } + const op = stream ? "streamGenerateContent?alt=sse&" : "generateContent?"; + return `https://aiplatform.googleapis.com/v1/publishers/google/models/${model}:${op}key=${expressKey}`; + } + const region = credentials?.providerSpecificData?.region || "us-central1"; let project = "unknown-project"; @@ -146,6 +187,7 @@ export class VertexExecutor extends BaseExecutor { if (credentials.accessToken) { headers["Authorization"] = `Bearer ${credentials.accessToken}`; } + // Express-mode keys are carried in the ?key= query parameter (see buildUrl), not a header. if (stream) { headers["Accept"] = "text/event-stream"; } diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index 5198a21a7d6..8a9239c05ed 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -180,7 +180,7 @@ import { isCacheableForRead, isCacheableForWrite, } from "@/lib/semanticCache"; -import { saveIdempotency } from "@/lib/idempotencyLayer"; +import { getIdempotencyKey, saveIdempotency } from "@/lib/idempotencyLayer"; import { createProgressTransform, wantsProgress } from "../utils/progressTracker.ts"; import { createPiiSseTransform } from "@/lib/streamingPiiTransform"; import { isFeatureFlagEnabled } from "@/shared/utils/featureFlags"; @@ -192,12 +192,7 @@ import { getModelFamily, } from "../services/modelFamilyFallback.ts"; import { computeRequestHash, deduplicate, shouldDeduplicate } from "../services/requestDedup.ts"; -import { - compressContext, - estimateTokens, - getTokenLimit, - resolveComboContextLimit, -} from "../services/contextManager.ts"; +import { compressContext, estimateTokens, getTokenLimit } from "../services/contextManager.ts"; import { getBackgroundTaskReason, getDegradedModel, @@ -1872,9 +1867,7 @@ export async function handleChatCore({ }; // ── Phase 9.2: Idempotency check ── - // Resolve the idempotency key once here and reuse it at the Phase 9.2 save site below, - // rather than re-deriving it. (#3821-review LEDGER-6) - const { hit: idempotencyHit, idempotencyKey } = await checkIdempotencyCache({ + const idempotencyHit = await checkIdempotencyCache({ clientRawRequest, provider, model, @@ -2746,30 +2739,21 @@ export async function handleChatCore({ if (!comboConfig && comboName.startsWith("combo/")) { comboConfig = await getComboByName(comboName.substring(6)); } - let comboTargetLimits: number[] = []; if (comboConfig) { const allCombosData = await getCombosCached(); const targets = resolveComboTargets( comboConfig as unknown as { name: string; models: unknown[] }, allCombosData as unknown as { name: string; models: unknown[] }[] ); - comboTargetLimits = targets.map((t: { modelStr?: string }) => { + const limits = targets.map((t: { modelStr?: string }) => { const parsed = parseModel(t.modelStr); return getTokenLimit(parsed.provider, parsed.model); }); + if (limits.length > 0) { + contextLimit = Math.min(...limits); + log?.info?.("CONTEXT", `Combo min limit: ${contextLimit}`); + } } - // chatCore executes per concrete target (handleSingleModel resolves - // provider/effectiveModel before delegating). Compress against THIS - // target's window; min(...allTargets) is only a defensive fallback — - // the old unconditional min compressed a 1M-target request at the - // smallest sibling's window ("agent keeps forgetting things"). - const resolved = resolveComboContextLimit({ - provider, - model: effectiveModel, - comboTargetLimits, - }); - contextLimit = resolved.limit; - log?.info?.("CONTEXT", `Combo context limit: ${resolved.limit} (source=${resolved.source})`); } catch (err) { log?.warn?.("CONTEXT", "Failed to resolve combo limits for compression: " + err); } @@ -5248,8 +5232,10 @@ export async function handleChatCore({ } // ── Phase 9.2: Save for idempotency ── - // Reuse the key resolved by checkIdempotencyCache() above (single derivation per - // request). (#3821-review LEDGER-6) + // The idempotency *check* moved into checkIdempotencyCache() during the + // chatCore modularization (#3598); re-derive the key here for the save path. + // getIdempotencyKey is pure (reads idempotency-key/x-request-id headers). + const idempotencyKey = getIdempotencyKey(clientRawRequest?.headers); saveIdempotency(idempotencyKey, translatedResponse, 200); reqLogger.logConvertedResponse(translatedResponse); persistAttemptLogs({ diff --git a/open-sse/handlers/chatCore/idempotency.ts b/open-sse/handlers/chatCore/idempotency.ts index 35a4b3c1015..26ee4ef4393 100644 --- a/open-sse/handlers/chatCore/idempotency.ts +++ b/open-sse/handlers/chatCore/idempotency.ts @@ -2,12 +2,6 @@ import { getIdempotencyKey, checkIdempotency } from "@/lib/idempotencyLayer"; import { calculateCost } from "@/lib/usage/costCalculator"; import { buildOmniRouteResponseMetaHeaders } from "@/domain/omnirouteResponseMeta"; -/** - * Resolve the request's idempotency key once and check the idempotency store. Returns the - * resolved `idempotencyKey` alongside the cache `hit` so the caller can reuse the SAME key - * for the later save path instead of re-deriving it — eliminating the dual-derivation that - * the chatCore modularization (#3598) introduced. (#3821-review LEDGER-6) - */ export async function checkIdempotencyCache({ clientRawRequest, provider, @@ -16,13 +10,13 @@ export async function checkIdempotencyCache({ startTime, log, }: { - clientRawRequest: unknown; + clientRawRequest: any; provider: string; model: string; - effectiveServiceTier: unknown; + effectiveServiceTier: any; startTime: number; - log: unknown; -}): Promise<{ hit: { success: true; response: Response } | null; idempotencyKey: string }> { + log: any; +}) { const idempotencyKey = getIdempotencyKey(clientRawRequest?.headers); const cachedIdemp = checkIdempotency(idempotencyKey); if (cachedIdemp) { @@ -39,26 +33,23 @@ export async function checkIdempotencyCache({ }) : 0; return { - idempotencyKey, - hit: { - success: true, - response: new Response(JSON.stringify(cachedIdemp.response), { - status: cachedIdemp.status, - headers: { - "Content-Type": "application/json", - "X-OmniRoute-Idempotent": "true", - ...buildOmniRouteResponseMetaHeaders({ - provider, - model, - cacheHit: false, - latencyMs: Date.now() - startTime, - usage: idempotentUsage, - costUsd: idempotentCost, - }), - }, - }), - }, + success: true, + response: new Response(JSON.stringify(cachedIdemp.response), { + status: cachedIdemp.status, + headers: { + "Content-Type": "application/json", + "X-OmniRoute-Idempotent": "true", + ...buildOmniRouteResponseMetaHeaders({ + provider, + model, + cacheHit: false, + latencyMs: Date.now() - startTime, + usage: idempotentUsage, + costUsd: idempotentCost, + }), + }, + }), }; } - return { hit: null, idempotencyKey }; + return null; } diff --git a/open-sse/handlers/chatCore/memorySkillsInjection.ts b/open-sse/handlers/chatCore/memorySkillsInjection.ts index 28719d810d0..65ad48a2567 100644 --- a/open-sse/handlers/chatCore/memorySkillsInjection.ts +++ b/open-sse/handlers/chatCore/memorySkillsInjection.ts @@ -25,14 +25,14 @@ export async function injectMemoryAndSkills({ backgroundReason, log, }: { - body: Record; + body: any; memoryOwnerId: string | null; provider: string; effectiveModel: string; sourceFormat: string; targetFormat: string; backgroundReason: string | null; - log: unknown; + log: any; }) { const memorySettings = memoryOwnerId ? await getMemorySettings().catch(() => DEFAULT_MEMORY_SETTINGS) diff --git a/open-sse/handlers/chatCore/sanitization.ts b/open-sse/handlers/chatCore/sanitization.ts index 62b43615ed6..c24a4bf856f 100644 --- a/open-sse/handlers/chatCore/sanitization.ts +++ b/open-sse/handlers/chatCore/sanitization.ts @@ -2,10 +2,10 @@ import { FORMATS } from "../../translator/formats.ts"; import { sanitizeOpenAITool } from "../../services/toolSchemaSanitizer.ts"; export function sanitizeChatRequestBody( - body: Record, + body: any, sourceFormat: string, targetFormat: string -): Record { +): any { const prefersResponsesTokenField = sourceFormat === FORMATS.OPENAI_RESPONSES || targetFormat === FORMATS.OPENAI_RESPONSES; diff --git a/open-sse/handlers/chatCore/semanticCache.ts b/open-sse/handlers/chatCore/semanticCache.ts index 9ce133c1a27..8a552d04b91 100644 --- a/open-sse/handlers/chatCore/semanticCache.ts +++ b/open-sse/handlers/chatCore/semanticCache.ts @@ -25,17 +25,17 @@ export async function checkSemanticCache({ persistAttemptLogs, }: { semanticCacheEnabled: boolean; - body: Record; - clientRawRequest: unknown; + body: any; + clientRawRequest: any; model: string; provider: string; stream: boolean; - reqLogger: unknown; - effectiveServiceTier: unknown; + reqLogger: any; + effectiveServiceTier: any; connectionId: string | null; startTime: number; - log: unknown; - persistAttemptLogs: (args: unknown) => void; + log: any; + persistAttemptLogs: (args: any) => void; }) { if (semanticCacheEnabled && isCacheableForRead(body, clientRawRequest?.headers)) { const signature = generateSignature( @@ -88,4 +88,4 @@ export async function checkSemanticCache({ } } return null; -} +} \ No newline at end of file diff --git a/open-sse/handlers/imageGeneration.ts b/open-sse/handlers/imageGeneration.ts index 3b2da49eaa1..b460d7b3b90 100644 --- a/open-sse/handlers/imageGeneration.ts +++ b/open-sse/handlers/imageGeneration.ts @@ -1,3776 +1 @@ -import { randomUUID } from "crypto"; -/** - * Image Generation Handler - * - * Handles POST /v1/images/generations requests. - * Proxies to upstream image generation providers using OpenAI-compatible format. - * - * Request format (OpenAI-compatible): - * { - * "model": "openai/gpt-image-2", - * "prompt": "a beautiful sunset over mountains", - * "n": 1, - * "size": "1024x1024", - * "quality": "standard", // optional: "standard" | "hd" - * "response_format": "url" // optional: "url" | "b64_json" - * } - */ - -import { getImageProvider, parseImageModel } from "../config/imageRegistry.ts"; -import { HTTP_STATUS } from "../config/constants.ts"; -import { applyAntigravityClientProfileHeaders } from "../services/antigravityClientProfile.ts"; -import { getAntigravityEnvelopeUserAgent } from "../services/antigravityIdentity.ts"; -import { kieExecutor } from "../executors/kie.ts"; -import { mapImageSize } from "../translator/image/sizeMapper.ts"; -import { getCodexClientVersion, getCodexUserAgent } from "../config/codexClient.ts"; -import { ChatGptWebExecutor } from "../executors/chatgpt-web.ts"; -import { getChatGptImage, findChatGptImageBySha256 } from "../services/chatgptImageCache.ts"; -import { createHash } from "node:crypto"; -import { saveCallLog } from "@/lib/usageDb"; -import { sleep } from "../utils/sleep.ts"; -import { - getKieErrorMessage, - getKieErrorStatus, - isJsonObject, - parseKieResultJson, -} from "../utils/kieTask.ts"; -import { - submitComfyWorkflow, - pollComfyResult, - fetchComfyOutput, - extractComfyOutputFiles, -} from "../utils/comfyuiClient.ts"; -import { fetchRemoteImage } from "@/shared/network/remoteImageFetch"; -import { FetchTimeoutError, fetchWithTimeout, getConfiguredTimeout } from "@/shared/utils/fetchTimeout"; -import { sanitizeErrorMessage, sanitizeUpstreamDetails } from "../utils/error.ts"; - -interface KieImageOptions { - model: string; - provider: string; - providerConfig: { - baseUrl: string; - statusUrl?: string; - }; - body: Record & { - prompt?: unknown; - size?: unknown; - n?: unknown; - timeout_ms?: unknown; - poll_interval_ms?: unknown; - }; - credentials?: { - apiKey?: string; - accessToken?: string; - } | null; - log?: { - info: (scope: string, message: string) => void; - error: (scope: string, message: string) => void; - } | null; -} - -const OPENAI_IMAGE_TO_IMAGE_MODELS = new Set([ - "black-forest-labs/FLUX.2-max", - "black-forest-labs/FLUX.2-pro", - "black-forest-labs/FLUX.2-flex", - "black-forest-labs/FLUX.2-dev", - "openai/gpt-image-1.5", - "Wan-AI/Wan2.6-image", - "Qwen/Qwen-Image-2.0-Pro", - "Qwen/Qwen-Image-2.0", - "google/flash-image-3.1", - "google/gemini-3-pro-image", - "flux-kontext-max", - "flux-kontext", - "flux-kontext-pro", - "qwen-image", -]); - -const IMAGE_ASPECT_RATIO_PATTERN = /^\d+:\d+$/; - -/** - * Resolve the upstream images endpoint for a custom (OpenAI-compatible) image - * provider node (#3205). - * - * Custom provider nodes store their base URL the same way the chat path does: - * in `credentials.providerSpecificData.baseUrl` (e.g. `https://example.com/v1`), - * NOT as a top-level `credentials.baseUrl`. Older callers may still pass a - * top-level `baseUrl`, so we honor that as a secondary source. When neither is - * present we fall back to `fallback` (the built-in Gemini OpenAI endpoint). - * - * Resolution order: providerSpecificData.baseUrl → credentials.baseUrl → fallback. - * - * A node base URL like `https://example.com/v1` is normalized and the - * OpenAI-compatible `/images/generations` path appended (mirroring - * `buildOpenAICompatibleUrl` in services/provider.ts). A node URL that already - * ends in `/images/generations` is returned as-is (no double-append). The - * `fallback` value is assumed to already be a complete URL and is returned - * verbatim. - */ -export function resolveImageBaseUrl( - credentials: - | { baseUrl?: unknown; providerSpecificData?: { baseUrl?: unknown } | null } - | null - | undefined, - fallback: string, - endpoint: "generations" | "edits" = "generations" -): string { - const psd = credentials?.providerSpecificData; - const psdBaseUrl = - psd && typeof psd === "object" && typeof psd.baseUrl === "string" && psd.baseUrl.trim() - ? psd.baseUrl.trim() - : null; - const topLevelBaseUrl = - typeof credentials?.baseUrl === "string" && credentials.baseUrl.trim() - ? credentials.baseUrl.trim() - : null; - const nodeBaseUrl = psdBaseUrl || topLevelBaseUrl; - - if (!nodeBaseUrl) return fallback; - - // A single configured node serves both image routes: honor a base URL that already - // points at the requested OpenAI image path, and rewrite one that points at the other - // image endpoint (e.g. `.../images/generations` requested for edits) (#3214/#3215). - const suffix = `/images/${endpoint}`; - // Trim trailing slashes without a backtracking-prone regex (`/\/+$/` is a - // polynomial-ReDoS pattern on long runs of "/" — CodeQL js/polynomial-redos). - let normalized = nodeBaseUrl; - while (normalized.endsWith("/")) normalized = normalized.slice(0, -1); - if (normalized.endsWith(suffix)) return normalized; - const stripped = normalized.replace(/\/images\/(?:generations|edits)$/, ""); - return `${stripped}${suffix}`; -} - -function normalizeImageAspectRatio(value: unknown, fallbackSize: unknown): string { - if (typeof value === "string") { - const trimmedValue = value.trim(); - if (IMAGE_ASPECT_RATIO_PATTERN.test(trimmedValue)) return trimmedValue; - } - return mapImageSize(typeof fallbackSize === "string" ? fallbackSize : null); -} - -function parseJsonOrNull(value: string): unknown | null { - try { - return JSON.parse(value); - } catch { - return null; - } -} - -function sanitizeImageProviderError(errorText: string): unknown { - const parsed = parseJsonOrNull(errorText); - if (parsed !== null) { - return sanitizeUpstreamDetails(parsed) || sanitizeErrorMessage(errorText); - } - return sanitizeErrorMessage(errorText); -} - -const BFL_MODEL_ENDPOINTS = { - "flux-2-max": "/v1/flux-2-max", - "flux-2-pro": "/v1/flux-2-pro", - "flux-2-flex": "/v1/flux-2-flex", - "flux-2-klein-9b": "/v1/flux-2-klein-9b", - "flux-2-klein-4b": "/v1/flux-2-klein-4b", - "flux-kontext-pro": "/v1/flux-kontext-pro", - "flux-kontext-max": "/v1/flux-kontext-max", - "flux-pro-1.1": "/v1/flux-pro-1.1", - "flux-pro-1.1-ultra": "/v1/flux-pro-1.1-ultra", - "flux-dev": "/v1/flux-dev", - "flux-pro": "/v1/flux-pro", -}; - -const BFL_EDIT_MODELS = new Set([ - "flux-2-max", - "flux-2-pro", - "flux-2-flex", - "flux-kontext-pro", - "flux-kontext-max", -]); - -const BFL_FAILURE_STATUSES = new Set(["Error", "Failed", "Content Moderated", "Request Moderated"]); - -function formatImageProviderError(err) { - const sanitized = sanitizeErrorMessage(err); - const message = (sanitized || "").replace(/^Error:\s*/i, "").trim(); - return message ? `Image provider error: ${message}` : "Image provider error"; -} - -const STABILITY_GENERATION_ENDPOINTS = { - "sd3.5-large": "/v2beta/stable-image/generate/sd3", - "sd3.5-large-turbo": "/v2beta/stable-image/generate/sd3", - "sd3.5-medium": "/v2beta/stable-image/generate/sd3", - "sd3.5-flash": "/v2beta/stable-image/generate/sd3", - "stable-image-ultra": "/v2beta/stable-image/generate/ultra", - "stable-image-core": "/v2beta/stable-image/generate/core", -}; - -const STABILITY_EDIT_ENDPOINTS = { - inpaint: "/v2beta/stable-image/edit/inpaint", - outpaint: "/v2beta/stable-image/edit/outpaint", - erase: "/v2beta/stable-image/edit/erase", - "search-and-replace": "/v2beta/stable-image/edit/search-and-replace", - "search-and-recolor": "/v2beta/stable-image/edit/search-and-recolor", - "remove-background": "/v2beta/stable-image/edit/remove-background", - "replace-background-and-relight": "/v2beta/stable-image/edit/replace-background-and-relight", - fast: "/v2beta/stable-image/upscale/fast", - conservative: "/v2beta/stable-image/upscale/conservative", - creative: "/v2beta/stable-image/upscale/creative", - sketch: "/v2beta/stable-image/control/sketch", - structure: "/v2beta/stable-image/control/structure", - style: "/v2beta/stable-image/control/style", - "style-transfer": "/v2beta/stable-image/control/style-transfer", -}; - -const STABILITY_CONTROL_MODELS = new Set(["sketch", "structure", "style", "style-transfer"]); - -function appendOptionalFormValue(formData, key, value) { - if (value === undefined || value === null || value === "") return; - formData.append(key, String(value)); -} - -function appendImageFormValue(formData, key, source, filename) { - formData.append( - key, - new Blob([source.buffer], { - type: source.contentType || "application/octet-stream", - }), - filename - ); -} - -const FAL_PRESET_SIZES = { - "1024x1024": "square_hd", - "512x512": "square", - "1792x1024": "landscape_16_9", - "1024x1792": "portrait_16_9", - "1024x768": "landscape_4_3", - "768x1024": "portrait_4_3", - "1536x1024": "landscape_3_2", - "1024x1536": "portrait_3_2", - "576x1024": "portrait_16_9", - "1024x576": "landscape_16_9", -}; - -/** - * Handle image generation request - * @param {object} options - * @param {object} options.body - Request body - * @param {object} options.credentials - Provider credentials { apiKey, accessToken } - * @param {object} options.log - Logger - * @param {string} [options.resolvedProvider] - Pre-resolved provider ID (from route layer custom model resolution) - */ -export async function handleImageGeneration({ - body, - credentials, - log, - resolvedProvider = null, - signal = null, - clientHeaders = null, -}) { - let provider, model; - - if (resolvedProvider) { - // Provider was already resolved by the route layer (custom model from DB) - // Extract model name from the full "provider/model" string - provider = resolvedProvider; - const modelStr = body.model || ""; - model = modelStr.startsWith(provider + "/") ? modelStr.slice(provider.length + 1) : modelStr; - } else { - // Standard path: resolve from built-in image registry - const parsed = parseImageModel(body.model); - provider = parsed.provider; - model = parsed.model; - } - - if (!provider) { - return { - success: false, - status: 400, - error: `Invalid image model: ${body.model}. Use format: provider/model`, - }; - } - - const providerConfig = getImageProvider(provider); - - // For custom models without a built-in provider config, use OpenAI-compatible handler - // with a synthetic config based on the provider's credentials - if (!providerConfig) { - if (!resolvedProvider) { - return { - success: false, - status: 400, - error: `Unknown image provider: ${provider}`, - }; - } - - // Custom model: use OpenAI-compatible format with provider's base URL - // The credentials were already resolved by the route layer - if (log) { - log.info("IMAGE", `Custom model ${provider}/${model} — using OpenAI-compatible handler`); - } - - const syntheticConfig = { - id: provider, - // #3205: custom OpenAI-compatible nodes store their base URL in - // credentials.providerSpecificData.baseUrl (same as the chat path — - // see executors/default.ts:buildUrl / services/provider.ts:buildProviderUrl). - // Previously only the (always-absent) top-level credentials.baseUrl was - // read, so every custom image node fell back to the Gemini endpoint and - // returned "Please pass a valid API key". - baseUrl: resolveImageBaseUrl( - credentials, - `https://generativelanguage.googleapis.com/v1beta/openai/images/generations` - ), - authType: "apikey", - authHeader: "bearer", - format: "openai", - }; - - return handleOpenAIImageGeneration({ - model, - provider, - providerConfig: syntheticConfig, - body, - credentials, - log, - }); - } - - if (providerConfig.format === "gemini-image") { - return handleGeminiImageGeneration({ model, providerConfig, body, credentials, log }); - } - - if (providerConfig.format === "imagen3") { - return handleImagen3ImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, - }); - } - - if (providerConfig.format === "hyperbolic") { - return handleHyperbolicImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, - }); - } - - if (providerConfig.format === "fal-ai") { - return handleFalAIImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, - }); - } - - if (providerConfig.format === "stability-ai") { - return handleStabilityAIImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, - }); - } - - if (providerConfig.format === "black-forest-labs") { - return handleBlackForestLabsImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, - }); - } - - if (providerConfig.format === "recraft") { - return handleRecraftImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, - }); - } - - if (providerConfig.format === "topaz") { - return handleTopazImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, - }); - } - - if (providerConfig.format === "chatgpt-web") { - return handleChatGptWebImageGeneration({ - model, - provider, - body, - credentials, - log, - signal, - clientHeaders, - }); - } - - if (providerConfig.format === "nanobanana") { - return handleNanoBananaImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, - }); - } - - if (providerConfig.format === "kie-image") { - return handleKieImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, - }); - } - - if (providerConfig.format === "sdwebui") { - return handleSDWebUIImageGeneration({ model, provider, providerConfig, body, log }); - } - - if (providerConfig.format === "comfyui") { - return handleComfyUIImageGeneration({ model, provider, providerConfig, body, log }); - } - - if (providerConfig.format === "codex-responses") { - return handleCodexImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, - }); - } - - if (providerConfig.format === "haiper-image") { - return handleHaiperImageGeneration({ model, provider, providerConfig, body, credentials, log }); - } - if (providerConfig.format === "leonardo-image") { - return handleLeonardoImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, - }); - } - if (providerConfig.format === "ideogram-image") { - return handleIdeogramImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, - }); - } - - return handleOpenAIImageGeneration({ model, provider, providerConfig, body, credentials, log }); -} - -function normalizeKieImageResult(recordData: unknown): string[] { - const record = isJsonObject(recordData) ? recordData : {}; - const data = isJsonObject(record.data) ? record.data : {}; - const response = isJsonObject(data.response) ? data.response : {}; - const resultJson = parseKieResultJson(recordData); - const urls = new Set(); - - const add = (val: unknown) => { - if (typeof val === "string" && val.startsWith("http")) urls.add(val); - if (Array.isArray(val)) { - val.forEach((v) => { - if (typeof v === "string" && v.startsWith("http")) urls.add(v); - }); - } - }; - - // Check resultJson (common in Market API) - add(resultJson?.resultUrls); - add(resultJson?.imageUrls); - add(resultJson?.resultUrl); - add(resultJson?.imageUrl); - - // Check data.response (common in 4o-image API) - add(response.resultUrls); - add(response.resultUrl); - - // Check direct data fields - add(data.resultImageUrls); - add(data.resultImageUrl); - add(data.url); - - return Array.from(urls); -} - -async function handleKieImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, -}: KieImageOptions) { - const startTime = Date.now(); - const token = credentials?.apiKey || credentials?.accessToken; - const timeoutMs = normalizePositiveNumber(body.timeout_ms, 300000); - const pollIntervalMs = normalizePositiveNumber(body.poll_interval_ms, 2500); - const prompt = typeof body.prompt === "string" ? body.prompt : String(body.prompt ?? ""); - const size = typeof body.size === "string" ? body.size : undefined; - - if (!token) { - return saveImageErrorResult({ - provider, - model, - status: 401, - startTime, - error: "KIE API key is required", - }); - } - - // Check if model is a Market model (unified API) - const fullRegistry = getImageProvider(provider); - const modelEntry = fullRegistry?.models?.find((m) => m.id === model); - const isMarket = modelEntry?.isMarket || model.includes("/"); - - const { imageUrl } = extractImageInputs(body); - let baseUrl = ""; - let payload: Record = {}; - - if (isMarket) { - // Unified Market API endpoint - baseUrl = `${providerConfig.baseUrl.replace(/\/$/, "")}/api/v1/jobs/createTask`; - const input: Record = { - prompt, - aspect_ratio: mapImageSize(size, "1:1"), - }; - if (imageUrl) { - input.image_url = imageUrl; - } - payload = { - model, - input, - }; - } else { - // Legacy/Direct endpoint - const modelPath = model.replace("-t2i", "").replace("-i2i", ""); - baseUrl = providerConfig.baseUrl.includes(model) - ? providerConfig.baseUrl - : `https://api.kie.ai/api/v1/${modelPath}/generate`; - - payload = { - prompt, - size: mapImageSize(size, "1:1"), - nVariants: body.n || 1, - }; - } - - if (log) { - const promptPreview = String(body.prompt ?? "").slice(0, 60); - log.info( - "IMAGE", - `${provider}/${model} (${isMarket ? "market" : "direct"}) | prompt: "${promptPreview}..."` - ); - } - - try { - const endpoint = isMarket ? "/api/v1/jobs/createTask" : new URL(baseUrl).pathname; - const createBaseUrl = isMarket ? providerConfig.baseUrl : baseUrl.replace(endpoint, ""); - const createData = await kieExecutor.createTask({ - baseUrl: createBaseUrl, - token, - payload, - endpoint, - }); - const taskId = createData?.data?.taskId || createData?.taskId; - - if (!taskId) { - const errorMessage = - createData?.msg || - createData?.message || - createData?.error || - "KIE image generation did not return taskId"; - if (log) { - log.error("IMAGE", `KIE createTask failed: ${JSON.stringify(createData)}`); - } - return saveImageErrorResult({ - provider, - model, - status: 502, - startTime, - error: errorMessage, - requestBody: payload, - }); - } - - // Use statusUrl from providerConfig if available, fallback to dynamic derivation - const statusUrl = isMarket - ? `${providerConfig.baseUrl.replace(/\/$/, "")}/api/v1/jobs/recordInfo` - : providerConfig.statusUrl && !providerConfig.statusUrl.includes("jobs/recordInfo") - ? providerConfig.statusUrl - : baseUrl.replace(/\/generate$/, "/record-info"); - - const { data: recordData, state } = await kieExecutor.pollTask({ - statusUrl, - taskId: String(taskId), - token, - timeoutMs, - pollIntervalMs, - }); - - if (state === "success") { - if (log) { - log.info("IMAGE", `KIE poll success for task ${taskId}`); - } - const urls = normalizeKieImageResult(recordData); - const images = urls.map((url: string) => ({ url, revised_prompt: prompt })); - - return saveImageSuccessResult({ - provider, - model, - startTime, - requestBody: payload, - responseBody: { images_count: images.length }, - images, - }); - } - - const record = isJsonObject(recordData) ? recordData : {}; - const recordDataBody = isJsonObject(record.data) ? record.data : {}; - const errorMessage = - recordDataBody.errorMessage || - recordDataBody.failMsg || - record.msg || - "KIE image task failed"; - - if (log) { - log.error("IMAGE", `KIE poll failed for task ${taskId}: ${JSON.stringify(recordData)}`); - } - - return saveImageErrorResult({ - provider, - model, - status: 502, - startTime, - error: String(errorMessage), - requestBody: payload, - }); - } catch (err: unknown) { - return saveImageErrorResult({ - provider, - model, - status: getKieErrorStatus(err, 502), - startTime, - error: `Image provider error: ${getKieErrorMessage(err, "KIE image generation failed")}`, - }); - } -} -/** - * Handle Gemini-format image generation (Antigravity / Nano Banana) - * Uses Gemini's generateContent API with responseModalities: ["TEXT", "IMAGE"] - */ -async function handleGeminiImageGeneration({ model, providerConfig, body, credentials, log }) { - const startTime = Date.now(); - const url = providerConfig.baseUrl; - const provider = "antigravity"; - const credentialRecord = credentials || {}; - const token = credentialRecord.accessToken || credentialRecord.apiKey; - const providerSpecificData = credentialRecord.providerSpecificData; - const providerSpecificProjectId = - providerSpecificData && typeof providerSpecificData === "object" - ? (providerSpecificData as Record).projectId - : null; - const credentialProjectId = - typeof credentialRecord.projectId === "string" ? credentialRecord.projectId.trim() : ""; - const providerProjectId = - typeof providerSpecificProjectId === "string" ? providerSpecificProjectId.trim() : ""; - const projectId = credentialProjectId || providerProjectId || null; - const candidateCount = - typeof body.n === "number" && Number.isFinite(body.n) && body.n > 0 ? Math.floor(body.n) : 1; - const promptText = typeof body.prompt === "string" ? body.prompt : String(body.prompt ?? ""); - - // Summarized request for call log - const logRequestBody = { - model: body.model, - prompt: promptText.slice(0, 200), - size: body.size || "default", - n: candidateCount, - }; - - if (!projectId || typeof projectId !== "string") { - return saveImageErrorResult({ - provider, - model, - status: 400, - startTime, - error: - "Missing Google projectId for Antigravity account. Please reconnect OAuth in Providers so OmniRoute can fetch your Cloud Code project.", - requestBody: logRequestBody, - }); - } - - const antigravityBody = { - project: projectId, - requestId: `image_gen/${Date.now()}/${randomUUID()}/0`, - request: { - contents: [ - { - role: "user", - parts: [{ text: promptText }], - }, - ], - generationConfig: { - candidateCount, - imageConfig: { - aspectRatio: normalizeImageAspectRatio(body.aspect_ratio, body.size), - }, - }, - }, - model, - userAgent: getAntigravityEnvelopeUserAgent(credentialRecord), - requestType: "image_gen", - }; - - const headers = { - "Content-Type": "application/json", - Authorization: `Bearer ${token}`, - }; - applyAntigravityClientProfileHeaders(headers, credentialRecord, antigravityBody); - delete headers["x-goog-user-project"]; - - if (log) { - const promptPreview = promptText.slice(0, 60); - log.info( - "IMAGE", - `antigravity/${model} (gemini) | prompt: "${promptPreview}..." | format: gemini-image` - ); - } - - try { - const response = await fetch(url, { - method: "POST", - headers, - body: JSON.stringify(antigravityBody), - }); - - if (!response.ok) { - const errorText = await response.text(); - const safeError = sanitizeImageProviderError(errorText); - const safeErrorLog = - typeof safeError === "string" ? safeError : JSON.stringify(safeError ?? {}); - if (log) { - log.error("IMAGE", `antigravity error ${response.status}: ${safeErrorLog.slice(0, 200)}`); - } - - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: response.status, - model: `antigravity/${model}`, - provider, - duration: Date.now() - startTime, - error: safeErrorLog.slice(0, 500), - requestBody: logRequestBody, - }).catch(() => {}); - - return { success: false, status: response.status, error: safeError }; - } - - const data = await response.json(); - const responseBody = data.response || data; - - // Extract image data from Antigravity's wrapped Gemini response. - const images = []; - const candidates = responseBody.candidates || []; - for (const candidate of candidates) { - const parts = candidate.content?.parts || []; - for (const part of parts) { - if (part.inlineData) { - images.push({ - b64_json: part.inlineData.data, - revised_prompt: parts.find((p) => p.text)?.text || promptText, - }); - } - } - } - - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 200, - model: `antigravity/${model}`, - provider, - duration: Date.now() - startTime, - tokens: { prompt_tokens: 0, completion_tokens: 0 }, - requestBody: logRequestBody, - responseBody: { images_count: images.length }, - }).catch(() => {}); - - return { - success: true, - data: { - created: Math.floor(Date.now() / 1000), - data: images, - }, - }; - } catch (err) { - if (log) { - log.error("IMAGE", `antigravity fetch error: ${err.message}`); - } - - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 502, - model: `antigravity/${model}`, - provider, - duration: Date.now() - startTime, - error: err.message, - requestBody: logRequestBody, - }).catch(() => {}); - - return { - success: false, - status: 502, - error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, - }; - } -} - -/** - * Handle OpenAI-compatible image generation (standard providers + Nebius fallback) - */ -async function handleOpenAIImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, -}) { - const startTime = Date.now(); - - // Summarized request for call log - const logRequestBody = { - model: body.model, - prompt: - typeof body.prompt === "string" - ? body.prompt.slice(0, 200) - : String(body.prompt ?? "").slice(0, 200), - size: body.size || "default", - n: body.n || 1, - quality: body.quality || undefined, - }; - - // Build upstream request (OpenAI-compatible format) - const upstreamBody: Record = { - model: model, - prompt: body.prompt, - }; - - // Pass optional parameters - if (body.n !== undefined) upstreamBody.n = body.n; - if (body.size !== undefined) upstreamBody.size = body.size; - if (body.quality !== undefined) upstreamBody.quality = body.quality; - if (body.response_format !== undefined) upstreamBody.response_format = body.response_format; - if (body.style !== undefined) upstreamBody.style = body.style; - - const { imageUrl } = extractImageInputs(body); - if (imageUrl && OPENAI_IMAGE_TO_IMAGE_MODELS.has(model)) { - upstreamBody.image_url = imageUrl; - } - - // Build headers - const headers = { - "Content-Type": "application/json", - }; - - const token = credentials.apiKey || credentials.accessToken; - if (providerConfig.authHeader === "bearer") { - headers["Authorization"] = `Bearer ${token}`; - } else if (providerConfig.authHeader === "x-api-key") { - headers["x-api-key"] = token; - } - - if (log) { - const promptPreview = - typeof body.prompt === "string" - ? body.prompt.slice(0, 60) - : String(body.prompt ?? "").slice(0, 60); - log.info( - "IMAGE", - `${provider}/${model} | prompt: "${promptPreview}..." | size: ${body.size || "default"}` - ); - } - - const requestBody = JSON.stringify(upstreamBody); - - // Try primary URL - let result = await fetchImageEndpoint( - providerConfig.baseUrl, - headers, - requestBody, - provider, - log - ); - - // Fallback for providers with fallbackUrl (e.g., Nebius) - if ( - !result.success && - providerConfig.fallbackUrl && - [404, 410, 502, 503].includes(result.status) - ) { - if (log) { - log.info("IMAGE", `${provider}: primary URL failed (${result.status}), trying fallback...`); - } - result = await fetchImageEndpoint( - providerConfig.fallbackUrl, - headers, - requestBody, - provider, - log - ); - } - - // Save call log after result is determined - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: result.status || (result.success ? 200 : 502), - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - tokens: { prompt_tokens: 0, completion_tokens: 0 }, - error: result.success - ? null - : typeof result.error === "string" - ? result.error.slice(0, 500) - : null, - requestBody: logRequestBody, - responseBody: result.success ? { images_count: result.data?.data?.length || 0 } : null, - }).catch(() => {}); - - return result; -} - -/** - * OpenAI-compatible image *edit* forwarder for custom providers (#3214 / #3215). - * - * Mirrors `handleOpenAIImageGeneration` but posts multipart/form-data to the node's - * `/images/edits` endpoint and returns the upstream OpenAI-compatible response. Kept - * separate from the chatgpt-web edit flow, which continues a saved conversation node - * rather than forwarding a stateless edit. The fetch helper leaves Content-Type unset so - * `fetch` derives the multipart boundary from the FormData body. - */ -export async function handleOpenAIImageEdit({ - model, - provider, - credentials, - prompt, - imageBytes, - imageMime, - size, - responseFormat, - n = 1, - log, -}: { - model: string; - provider: string; - credentials: - | { - apiKey?: string; - accessToken?: string; - baseUrl?: unknown; - providerSpecificData?: { baseUrl?: unknown } | null; - } - | null - | undefined; - prompt: string; - imageBytes: Buffer; - imageMime?: string | null; - size?: string | null; - responseFormat?: string | null; - n?: number; - log?: { info: (tag: string, message: string) => void } | null; -}) { - const startTime = Date.now(); - const url = resolveImageBaseUrl( - credentials, - `https://generativelanguage.googleapis.com/v1beta/openai/images/edits`, - "edits" - ); - - // Build the multipart body as a Buffer with an explicit boundary instead of a global - // `FormData`. In production `globalThis.fetch` is patched with node_modules/undici's fetch, - // whose `FormData` class differs from `globalThis.FormData` — passing a native FormData - // makes undici serialize it as the string "[object FormData]" (text/plain), dropping every - // field (including `model`, which reaches the upstream empty). A Buffer body is accepted - // verbatim by any fetch implementation. (#3273) - const boundary = `----OmniRouteImageEdit${randomUUID().replace(/-/g, "")}`; - const CRLF = "\r\n"; - const partBuffers: Buffer[] = []; - const appendField = (name: string, value: string) => { - partBuffers.push( - Buffer.from( - `--${boundary}${CRLF}Content-Disposition: form-data; name="${name}"${CRLF}${CRLF}${value}${CRLF}` - ) - ); - }; - appendField("model", model); - appendField("prompt", prompt); - if (size) appendField("size", size); - if (responseFormat) appendField("response_format", responseFormat); - appendField("n", String(n || 1)); - partBuffers.push( - Buffer.from( - `--${boundary}${CRLF}Content-Disposition: form-data; name="image"; filename="image.png"${CRLF}` + - `Content-Type: ${imageMime || "image/png"}${CRLF}${CRLF}` - ) - ); - partBuffers.push(imageBytes); - partBuffers.push(Buffer.from(`${CRLF}--${boundary}--${CRLF}`)); - const multipartBody = Buffer.concat(partBuffers); - - const headers: Record = { - "Content-Type": `multipart/form-data; boundary=${boundary}`, - }; - const token = credentials?.apiKey || credentials?.accessToken; - if (token) headers["Authorization"] = `Bearer ${token}`; - - if (log) { - log.info("IMAGE", `${provider}/${model} (edit) | prompt: "${prompt.slice(0, 60)}..." -> ${url}`); - } - - const result = await fetchImageEndpoint( - url, - headers, - multipartBody as unknown as BodyInit, - provider, - log - ); - - saveCallLog({ - method: "POST", - path: "/v1/images/edits", - status: result.status || (result.success ? 200 : 502), - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - tokens: { prompt_tokens: 0, completion_tokens: 0 }, - error: result.success - ? null - : typeof result.error === "string" - ? result.error.slice(0, 500) - : null, - requestBody: { model, prompt: prompt.slice(0, 200), size: size || "default", n: n || 1 }, - responseBody: result.success ? { images_count: result.data?.data?.length || 0 } : null, - }).catch(() => {}); - - return result; -} - -const CHATGPT_WEB_IMAGE_MARKDOWN_RE = /!\[[^\]]*\]\(([^)\s]+)\)/g; -const CHATGPT_WEB_IMAGE_ID_RE = /\/v1\/chatgpt-web\/image\/([a-f0-9]{16,64})(?=[?\s"'<>)]|$)/i; - -function extractMarkdownImageUrls(text: string): string[] { - const urls: string[] = []; - // String.prototype.matchAll consumes a fresh iterator and ignores the - // regex's lastIndex, so no manual reset is required. - for (const match of text.matchAll(CHATGPT_WEB_IMAGE_MARKDOWN_RE)) { - if (match[1]) urls.push(match[1]); - } - return urls; -} - -function buildChatGptWebImagePrompt(body): string { - const prompt = String(body.prompt || "").trim(); - const details: string[] = [`Create an image for this prompt: ${prompt}`]; - if (typeof body.size === "string" && body.size.trim()) { - details.push(`Requested size: ${body.size.trim()}.`); - } - if (typeof body.quality === "string" && body.quality.trim()) { - details.push(`Requested quality: ${body.quality.trim()}.`); - } - if (typeof body.style === "string" && body.style.trim()) { - details.push(`Requested style: ${body.style.trim()}.`); - } - return details.join("\n"); -} - -async function handleChatGptWebImageGeneration({ - model, - provider, - body, - credentials, - log, - signal, - clientHeaders, -}) { - const startTime = Date.now(); - const prompt = typeof body.prompt === "string" ? body.prompt.trim() : ""; - if (!prompt) { - return saveImageErrorResult({ - provider, - model, - status: 400, - startTime, - error: "Prompt is required for ChatGPT Web image generation", - }); - } - - if (!credentials?.apiKey) { - return saveImageErrorResult({ - provider, - model, - status: 401, - startTime, - error: "ChatGPT Web credentials missing session cookie", - }); - } - - // Each image is one chatgpt.com chat turn (~30s). Cap at 4 (matches OpenAI's - // own limit for GPT Image models) so a stray n=1000 doesn't pin the - // executor for hours before the upstream HTTP timeout fires. - const CHATGPT_WEB_IMAGE_N_MAX = 4; - const rawCount = Number.isInteger(body.n) && (body.n as number) > 0 ? (body.n as number) : 1; - if (rawCount > CHATGPT_WEB_IMAGE_N_MAX) { - return saveImageErrorResult({ - provider, - model, - status: 400, - startTime, - error: `ChatGPT Web image generation supports n=1..${CHATGPT_WEB_IMAGE_N_MAX} (got ${rawCount}); each n is a separate ~30s chat turn.`, - }); - } - const requestedCount = rawCount; - if (log && requestedCount > 1) { - log.warn( - "IMAGE", - `ChatGPT Web returns one image per chat turn; requested n=${requestedCount} will run sequentially` - ); - } - - const wantsBase64 = body.response_format === "b64_json"; - const images: Array<{ url?: string; b64_json?: string }> = []; - const requestBody = { - model, - prompt: prompt.slice(0, 500), - size: body.size || undefined, - quality: body.quality || undefined, - }; - - for (let i = 0; i < requestedCount; i++) { - const executor = new ChatGptWebExecutor(); - const result = await executor.execute({ - model, - body: { - messages: [{ role: "user", content: buildChatGptWebImagePrompt(body) }], - }, - stream: false, - credentials, - signal, - log, - clientHeaders, - }); - - const responseText = await result.response.text(); - if (result.response.status >= 400) { - return saveImageErrorResult({ - provider, - model, - status: result.response.status, - startTime, - error: responseText, - requestBody, - }); - } - - let content = ""; - try { - const json = JSON.parse(responseText); - content = String(json?.choices?.[0]?.message?.content || ""); - } catch { - content = responseText; - } - - const urls = extractMarkdownImageUrls(content); - if (urls.length === 0) { - return saveImageErrorResult({ - provider, - model, - status: 502, - startTime, - error: `ChatGPT Web completed without returning image markdown: ${content.slice(0, 300)}`, - requestBody, - }); - } - - for (const url of urls) { - if (!wantsBase64) { - images.push({ url }); - continue; - } - const id = url.match(CHATGPT_WEB_IMAGE_ID_RE)?.[1]; - const cached = id ? getChatGptImage(id) : null; - if (!cached) { - return saveImageErrorResult({ - provider, - model, - status: 502, - startTime, - error: "ChatGPT Web image bytes expired before b64_json conversion", - requestBody, - }); - } - images.push({ b64_json: cached.bytes.toString("base64") }); - } - } - - return saveImageSuccessResult({ - provider, - model, - startTime, - requestBody, - responseBody: { images_count: images.length }, - images, - }); -} - -/** - * Handle a multipart /v1/images/edits request for chatgpt-web. Open WebUI - * uploads the prior image's bytes; we hash them and look up our cache. - * - * The hash match is reliable because Open WebUI's image-gen pipeline - * downloads our /v1/chatgpt-web/image/ URL byte-for-byte and re-serves - * those exact bytes through its own file store. When the user asks to edit - * the image, OWUI uploads the same bytes back to us via multipart — same - * hash, we find the conversation context, and drive the executor with a - * synthetic chat thread that triggers continuation mode. - * - * No-match cases (cache evicted by TTL, or the user uploaded a foreign - * image) get a clear 400. We can't actually edit an image we don't have a - * conversation context for — chatgpt.com's image_gen tool needs the - * original conversation node, and we don't have a path to upload bytes - * directly. - */ -export async function handleImageEdit({ - provider, - model, - body, - imageBytes, - credentials, - log, - signal = null, - clientHeaders = null, -}: { - provider: string; - model: string; - body: Record; - imageBytes: Buffer; - imageMime?: string; // accepted for symmetry with route layer; not used - credentials: any; - log: any; - signal?: AbortSignal | null; - clientHeaders?: Record | null; -}) { - const startTime = Date.now(); - const prompt = typeof body.prompt === "string" ? body.prompt.trim() : ""; - if (!prompt) { - return saveImageErrorResult({ - provider, - model, - status: 400, - startTime, - error: "Prompt is required for image edit", - }); - } - - if (!credentials?.apiKey) { - return saveImageErrorResult({ - provider, - model, - status: 401, - startTime, - error: "ChatGPT Web credentials missing session cookie", - }); - } - - const imageHash = createHash("sha256").update(imageBytes).digest("hex"); - const cached = findChatGptImageBySha256(imageHash); - - const wantsBase64 = body.response_format === "b64_json"; - const requestBody = { - model, - prompt: prompt.slice(0, 500), - size: body.size || undefined, - image_hash: imageHash.slice(0, 16), - image_bytes: imageBytes.length, - cached_match: Boolean(cached?.entry.context), - }; - - if (!cached?.entry.context) { - // chatgpt-web's image_gen tool can only edit an image when we continue - // the original conversation node. If we never generated this image (or - // its 30-minute TTL elapsed), there's no node to continue. Return a - // clear, actionable error — much better than silently spawning an - // unrelated image and confusing the user. - log?.warn?.( - "IMAGE", - `chatgpt-web edit: no cached match for sha256=${imageHash.slice(0, 16)} (bytes=${imageBytes.length}); returning 400` - ); - return saveImageErrorResult({ - provider, - model, - status: 400, - startTime, - error: - "chatgpt-web image edit only works for images recently generated through this OmniRoute instance " + - "(cache window: 30 minutes). Re-generate the image and try the edit immediately, or disable image-edit " + - "in your client to use plain chat-completion edit prompts instead.", - requestBody, - }); - } - - // Build a synthetic chat thread that surfaces the cached image URL on - // the assistant turn. The executor's parseOpenAIMessages picks up the - // URL, findCachedImageContext resolves it to {conversationId, - // parentMessageId}, and looksLikeImageEditRequest fires on the user - // prompt — together producing a continuation request that actually - // edits the saved image. - // - // The synthetic user prompt is anchored with both an edit verb AND an - // image-gen verb so the executor's heuristics fire regardless of what - // wording the caller used ("now make it brighter", "tweak this", ...): - // - looksLikeImageEditRequest: matches "edit" + "image" within 120 chars - // - looksLikeImageGenRequest: matches "generate" + "image" within 40 chars - // Either match alone would set forImageGen, but covering both is cheap - // insurance for prompts that don't fit common phrasings. - const messages: Array<{ role: string; content: string }> = [ - { - role: "assistant", - // The base URL is irrelevant — only the path is parsed by - // CACHED_IMAGE_URL_RE in the executor's findCachedImageContext. - content: `![image](http://internal/v1/chatgpt-web/image/${cached.id})`, - }, - { - role: "user", - content: `Edit the image and generate the new image: ${prompt}`, - }, - ]; - - const executor = new ChatGptWebExecutor(); - const result = await executor.execute({ - model, - body: { messages }, - stream: false, - credentials, - signal, - log, - clientHeaders, - }); - - const responseText = await result.response.text(); - if (result.response.status >= 400) { - return saveImageErrorResult({ - provider, - model, - status: result.response.status, - startTime, - error: responseText, - requestBody, - }); - } - - let content = ""; - try { - const json = JSON.parse(responseText); - content = String(json?.choices?.[0]?.message?.content || ""); - } catch { - content = responseText; - } - - const urls = extractMarkdownImageUrls(content); - if (urls.length === 0) { - return saveImageErrorResult({ - provider, - model, - status: 502, - startTime, - error: `ChatGPT Web edit completed without returning image markdown: ${content.slice(0, 300)}`, - requestBody, - }); - } - - const images: Array<{ url?: string; b64_json?: string }> = []; - for (const url of urls) { - if (!wantsBase64) { - images.push({ url }); - continue; - } - const id = url.match(CHATGPT_WEB_IMAGE_ID_RE)?.[1]; - const cachedNew = id ? getChatGptImage(id) : null; - if (!cachedNew) { - return saveImageErrorResult({ - provider, - model, - status: 502, - startTime, - error: "ChatGPT Web image bytes expired before b64_json conversion", - requestBody, - }); - } - images.push({ b64_json: cachedNew.bytes.toString("base64") }); - } - - return saveImageSuccessResult({ - provider, - model, - startTime, - requestBody, - responseBody: { images_count: images.length, edit_match: Boolean(cached?.entry.context) }, - images, - }); -} - -async function handleFalAIImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, -}) { - const startTime = Date.now(); - const token = credentials.apiKey || credentials.accessToken; - const { imageUrl, imageUrls } = extractImageInputs(body); - const upstreamBody: Record = { - prompt: body.prompt, - sync_mode: body.sync_mode ?? true, - }; - - if (body.n !== undefined) upstreamBody.num_images = Number(body.n) || 1; - if (body.negative_prompt) upstreamBody.negative_prompt = body.negative_prompt; - if (body.seed !== undefined) upstreamBody.seed = body.seed; - if (body.style) upstreamBody.style = normalizeRecraftStyle(body.style); - - const outputFormat = normalizeRequestedImageFormat(body, "png"); - if (outputFormat) upstreamBody.output_format = outputFormat; - - if (model.includes("flux-pro/v1.1") && !model.includes("ultra")) { - upstreamBody.image_size = mapFalImageSize(body.size, "landscape_4_3"); - } else if ( - model.includes("bytedance/") || - model.includes("stable-diffusion") || - model.includes("ideogram") || - model.includes("recraft/v3") - ) { - upstreamBody.image_size = mapFalImageSize(body.size, "square_hd"); - } else { - upstreamBody.aspect_ratio = body.aspect_ratio || mapFalAspectRatio(body.size, "1:1"); - } - - if (body.quality === "hd" && model.includes("ultra")) { - upstreamBody.raw = true; - } - - if (imageUrl && model.includes("flux-pro/v1.1-ultra")) { - upstreamBody.image_url = imageUrl; - } - - if (imageUrls.length > 0 && model.includes("ideogram")) { - upstreamBody.image_urls = imageUrls; - } - - if (log) { - const promptPreview = String(body.prompt ?? "").slice(0, 60); - log.info("IMAGE", `${provider}/${model} (fal-ai) | prompt: "${promptPreview}..."`); - } - - try { - const response = await fetch(`${providerConfig.baseUrl.replace(/\/$/, "")}/${model}`, { - method: "POST", - headers: { - "Content-Type": "application/json", - Authorization: `Key ${token}`, - }, - body: JSON.stringify(upstreamBody), - }); - - if (!response.ok) { - const errorText = await response.text(); - if (log) - log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); - return saveImageErrorResult({ - provider, - model, - status: response.status, - startTime, - error: errorText, - requestBody: upstreamBody, - }); - } - - const payload = await response.json(); - const images = await normalizeProviderImagePayload(payload, body, log); - return saveImageSuccessResult({ - provider, - model, - startTime, - requestBody: upstreamBody, - responseBody: { images_count: images.length }, - created: payload.created, - images, - }); - } catch (err) { - if (log) log.error("IMAGE", `${provider} fetch error: ${err.message}`); - return saveImageErrorResult({ - provider, - model, - status: 502, - startTime, - error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, - }); - } -} - -async function handleStabilityAIImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, -}) { - const startTime = Date.now(); - const token = credentials.apiKey || credentials.accessToken; - const endpoint = STABILITY_GENERATION_ENDPOINTS[model] || STABILITY_EDIT_ENDPOINTS[model]; - - if (!endpoint) { - return { - success: false, - status: 400, - error: `Unsupported Stability AI image model: ${model}`, - }; - } - - const { imageUrl, maskUrl } = extractImageInputs(body); - const upstreamBody: Record = { - output_format: - model === "remove-background" - ? normalizeRequestedImageFormat(body, "png", ["png", "webp"]) - : normalizeRequestedImageFormat(body, "png"), - }; - const formData = new FormData(); - - appendOptionalFormValue(formData, "output_format", upstreamBody.output_format); - if (body.prompt) { - upstreamBody.prompt = body.prompt; - appendOptionalFormValue(formData, "prompt", body.prompt); - } - if (body.negative_prompt) { - upstreamBody.negative_prompt = body.negative_prompt; - appendOptionalFormValue(formData, "negative_prompt", body.negative_prompt); - } - if (body.seed !== undefined) { - upstreamBody.seed = body.seed; - appendOptionalFormValue(formData, "seed", body.seed); - } - - try { - if (STABILITY_GENERATION_ENDPOINTS[model]) { - if (model.startsWith("sd3.5")) { - upstreamBody.model = model; - appendOptionalFormValue(formData, "model", model); - } - - if (imageUrl) { - const imageSource = await resolveImageSource(imageUrl); - upstreamBody.mode = "image-to-image"; - appendOptionalFormValue(formData, "mode", "image-to-image"); - upstreamBody.image = imageSource.base64; - appendImageFormValue(formData, "image", imageSource, "image"); - if (body.strength !== undefined) { - upstreamBody.strength = body.strength; - appendOptionalFormValue(formData, "strength", body.strength); - } - } else { - upstreamBody.mode = "text-to-image"; - appendOptionalFormValue(formData, "mode", "text-to-image"); - } - - if (!model.startsWith("sd3.5") || !imageUrl) { - const aspectRatio = body.aspect_ratio || mapImageSize(body.size); - upstreamBody.aspect_ratio = aspectRatio; - appendOptionalFormValue(formData, "aspect_ratio", aspectRatio); - } - - if (body.style_preset) { - upstreamBody.style_preset = body.style_preset; - appendOptionalFormValue(formData, "style_preset", body.style_preset); - } - } else { - if (imageUrl) { - const imageSource = await resolveImageSource(imageUrl); - upstreamBody.image = imageSource.base64; - appendImageFormValue(formData, "image", imageSource, "image"); - } - - if (maskUrl && shouldIncludeStabilityMask(model)) { - const maskSource = await resolveImageSource(maskUrl); - upstreamBody.mask = maskSource.base64; - appendImageFormValue(formData, "mask", maskSource, "mask"); - } - - if (body.search_prompt) { - upstreamBody.search_prompt = body.search_prompt; - appendOptionalFormValue(formData, "search_prompt", body.search_prompt); - } - if (body.grow_mask !== undefined) { - upstreamBody.grow_mask = body.grow_mask; - appendOptionalFormValue(formData, "grow_mask", body.grow_mask); - } - if (body.control_strength !== undefined) { - upstreamBody.control_strength = body.control_strength; - appendOptionalFormValue(formData, "control_strength", body.control_strength); - } - if (body.creativity !== undefined) { - upstreamBody.creativity = body.creativity; - appendOptionalFormValue(formData, "creativity", body.creativity); - } - if (body.left !== undefined) { - upstreamBody.left = body.left; - appendOptionalFormValue(formData, "left", body.left); - } - if (body.right !== undefined) { - upstreamBody.right = body.right; - appendOptionalFormValue(formData, "right", body.right); - } - if (body.up !== undefined) { - upstreamBody.up = body.up; - appendOptionalFormValue(formData, "up", body.up); - } - if (body.down !== undefined) { - upstreamBody.down = body.down; - appendOptionalFormValue(formData, "down", body.down); - } - if (body.style_preset) { - upstreamBody.style_preset = body.style_preset; - appendOptionalFormValue(formData, "style_preset", body.style_preset); - } - - if (STABILITY_CONTROL_MODELS.has(model) && !upstreamBody.prompt) { - upstreamBody.prompt = body.prompt || ""; - appendOptionalFormValue(formData, "prompt", body.prompt || ""); - } - } - - if (log) { - const promptPreview = String(body.prompt ?? "").slice(0, 60); - log.info("IMAGE", `${provider}/${model} (stability-ai) | prompt: "${promptPreview}..."`); - } - - const response = await fetch(`${providerConfig.baseUrl.replace(/\/$/, "")}${endpoint}`, { - method: "POST", - headers: { - Accept: "application/json", - Authorization: `Bearer ${token}`, - }, - body: formData, - }); - - if (!response.ok) { - const errorText = await response.text(); - if (log) - log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); - return saveImageErrorResult({ - provider, - model, - status: response.status, - startTime, - error: errorText, - requestBody: upstreamBody, - }); - } - - const contentType = response.headers.get("content-type") || ""; - let payload; - if (contentType.includes("application/json")) { - payload = await response.json(); - } else { - const buffer = Buffer.from(await response.arrayBuffer()); - payload = { image: buffer.toString("base64") }; - } - - const images = await normalizeProviderImagePayload(payload, body, log); - return saveImageSuccessResult({ - provider, - model, - startTime, - requestBody: upstreamBody, - responseBody: { images_count: images.length }, - created: payload.created, - images, - }); - } catch (err) { - if (log) log.error("IMAGE", `${provider} fetch error: ${err.message}`); - return saveImageErrorResult({ - provider, - model, - status: 502, - startTime, - error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, - }); - } -} - -async function handleBlackForestLabsImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, -}) { - const startTime = Date.now(); - const token = credentials.apiKey || credentials.accessToken; - const endpoint = BFL_MODEL_ENDPOINTS[model]; - - if (!endpoint) { - return { - success: false, - status: 400, - error: `Unsupported Black Forest Labs image model: ${model}`, - }; - } - - const { imageUrl, maskUrl } = extractImageInputs(body); - const upstreamBody: Record = { - prompt: body.prompt, - output_format: normalizeRequestedImageFormat(body, "png"), - }; - - try { - if (BFL_EDIT_MODELS.has(model) && imageUrl) { - upstreamBody.input_image = (await resolveImageSource(imageUrl)).base64; - } else if (imageUrl && isHttpUrl(imageUrl)) { - upstreamBody.image_url = imageUrl; - } - - if (maskUrl && (model === "flux-pro-1.0-fill" || model === "flux-kontext-pro")) { - upstreamBody.mask = (await resolveImageSource(maskUrl)).base64; - } - - if (model === "flux-kontext-pro" || model === "flux-kontext-max") { - upstreamBody.aspect_ratio = body.aspect_ratio || mapImageSize(body.size); - } else if (typeof body.size === "string" && body.size.includes("x")) { - const { width, height } = parseSizeToDimensions(body.size, 1024); - upstreamBody.width = width; - upstreamBody.height = height; - } - - if (body.seed !== undefined) upstreamBody.seed = body.seed; - if (body.n !== undefined && model.includes("ultra")) - upstreamBody.num_images = Number(body.n) || 1; - if (body.quality === "hd" && model.includes("ultra")) upstreamBody.raw = true; - if (body.left !== undefined) upstreamBody.left = body.left; - if (body.right !== undefined) upstreamBody.right = body.right; - if (body.top !== undefined) upstreamBody.top = body.top; - if (body.bottom !== undefined) upstreamBody.bottom = body.bottom; - if (body.steps !== undefined) upstreamBody.steps = body.steps; - if (body.guidance !== undefined) upstreamBody.guidance = body.guidance; - if (body.grow_mask !== undefined) upstreamBody.grow_mask = body.grow_mask; - if (body.safety_tolerance !== undefined) upstreamBody.safety_tolerance = body.safety_tolerance; - - if (log) { - const promptPreview = String(body.prompt ?? "").slice(0, 60); - log.info("IMAGE", `${provider}/${model} (black-forest-labs) | prompt: "${promptPreview}..."`); - } - - const response = await fetch(`${providerConfig.baseUrl.replace(/\/$/, "")}${endpoint}`, { - method: "POST", - headers: { - "Content-Type": "application/json", - Accept: "application/json", - "x-key": token, - }, - body: JSON.stringify(upstreamBody), - }); - - if (!response.ok) { - const errorText = await response.text(); - if (log) - log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); - return saveImageErrorResult({ - provider, - model, - status: response.status, - startTime, - error: errorText, - requestBody: upstreamBody, - }); - } - - const initialPayload = await response.json(); - const finalPayload = initialPayload.polling_url - ? await pollBlackForestLabsResult({ - pollingUrl: initialPayload.polling_url, - token, - body, - log, - }) - : initialPayload; - - const images = await normalizeProviderImagePayload(finalPayload, body, log); - return saveImageSuccessResult({ - provider, - model, - startTime, - requestBody: upstreamBody, - responseBody: { images_count: images.length }, - created: finalPayload.created, - images, - }); - } catch (err) { - if (log) log.error("IMAGE", `${provider} fetch error: ${err.message}`); - return saveImageErrorResult({ - provider, - model, - status: 502, - startTime, - error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, - }); - } -} - -async function handleRecraftImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, -}) { - const startTime = Date.now(); - const token = credentials.apiKey || credentials.accessToken; - const upstreamBody: Record = { - model, - prompt: body.prompt, - }; - - if (body.n !== undefined) upstreamBody.n = body.n; - if (body.size !== undefined) upstreamBody.size = body.size; - if (body.response_format !== undefined) upstreamBody.response_format = body.response_format; - if (body.style !== undefined) upstreamBody.style = body.style; - - if (log) { - const promptPreview = String(body.prompt ?? "").slice(0, 60); - log.info("IMAGE", `${provider}/${model} (recraft) | prompt: "${promptPreview}..."`); - } - - try { - const response = await fetch( - `${providerConfig.baseUrl.replace(/\/$/, "")}/v1/images/generations`, - { - method: "POST", - headers: { - "Content-Type": "application/json", - Authorization: `Bearer ${token}`, - }, - body: JSON.stringify(upstreamBody), - } - ); - - if (!response.ok) { - const errorText = await response.text(); - if (log) - log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); - return saveImageErrorResult({ - provider, - model, - status: response.status, - startTime, - error: errorText, - requestBody: upstreamBody, - }); - } - - const payload = await response.json(); - const images = await normalizeProviderImagePayload(payload, body, log); - return saveImageSuccessResult({ - provider, - model, - startTime, - requestBody: upstreamBody, - responseBody: { images_count: images.length }, - created: payload.created, - images, - }); - } catch (err) { - if (log) log.error("IMAGE", `${provider} fetch error: ${err.message}`); - return saveImageErrorResult({ - provider, - model, - status: 502, - startTime, - error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, - }); - } -} - -async function handleTopazImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, -}) { - const startTime = Date.now(); - const token = credentials.apiKey || credentials.accessToken; - const { imageUrl } = extractImageInputs(body); - - if (!imageUrl) { - return { - success: false, - status: 400, - error: `Topaz model ${model} requires an input image`, - }; - } - - try { - const imageSource = await resolveImageSource(imageUrl); - const formData = new FormData(); - const blob = new Blob([imageSource.buffer], { type: imageSource.contentType || "image/png" }); - formData.append("image", blob, "image.png"); - - if (typeof body.size === "string" && body.size.includes("x")) { - const { width, height } = parseSizeToDimensions(body.size, 1024); - formData.append("output_width", String(width)); - formData.append("output_height", String(height)); - } - - if (log) { - const promptPreview = String(body.prompt ?? "enhance image").slice(0, 60); - log.info("IMAGE", `${provider}/${model} (topaz) | prompt: "${promptPreview}..."`); - } - - const response = await fetch(`${providerConfig.baseUrl.replace(/\/$/, "")}/image/v1/enhance`, { - method: "POST", - headers: { - Accept: "image/jpeg", - "X-API-Key": token, - }, - body: formData, - }); - - if (!response.ok) { - const errorText = await response.text(); - if (log) - log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); - return saveImageErrorResult({ - provider, - model, - status: response.status, - startTime, - error: errorText, - }); - } - - const contentType = response.headers.get("content-type") || "image/jpeg"; - const buffer = Buffer.from(await response.arrayBuffer()); - const base64 = buffer.toString("base64"); - const wantsBase64 = body.response_format === "b64_json"; - const images = [ - wantsBase64 - ? { b64_json: base64, revised_prompt: body.prompt } - : { url: `data:${contentType};base64,${base64}`, revised_prompt: body.prompt }, - ]; - - return saveImageSuccessResult({ - provider, - model, - startTime, - responseBody: { images_count: images.length }, - images, - }); - } catch (err) { - if (log) log.error("IMAGE", `${provider} fetch error: ${err.message}`); - return saveImageErrorResult({ - provider, - model, - status: 502, - startTime, - error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, - }); - } -} - -async function pollBlackForestLabsResult({ pollingUrl, token, body, log }) { - const timeoutMs = normalizePositiveNumber(body.timeout_ms, 300000); - const pollIntervalMs = normalizePositiveNumber(body.poll_interval_ms, 1500); - const deadline = Date.now() + timeoutMs; - - while (Date.now() < deadline) { - const response = await fetch(pollingUrl, { - method: "GET", - headers: { - "x-key": token, - }, - }); - - if (!response.ok) { - const errorText = await response.text(); - throw new Error(`BFL polling failed (${response.status}): ${errorText}`); - } - - const payload = await response.json(); - const status = payload?.status; - - if (status === "Ready") { - return payload; - } - - if (BFL_FAILURE_STATUSES.has(status)) { - throw new Error(`BFL image generation failed: ${status}`); - } - - if (log) { - log.info("IMAGE", `black-forest-labs polling status: ${String(status || "Pending")}`); - } - - await sleep(pollIntervalMs); - } - - throw new Error(`BFL polling timed out after ${timeoutMs}ms`); -} - -function extractImageInputs(body) { - const imageUrls = []; - const seen = new Set(); - - const pushCandidate = (candidate) => { - if (typeof candidate !== "string") return; - const trimmed = candidate.trim(); - if (!trimmed || seen.has(trimmed)) return; - seen.add(trimmed); - imageUrls.push(trimmed); - }; - - pushCandidate(body?.image_url); - pushCandidate(body?.image); - - if (Array.isArray(body?.imageUrls)) { - for (const candidate of body.imageUrls) pushCandidate(candidate); - } - - if (Array.isArray(body?.image_urls)) { - for (const candidate of body.image_urls) pushCandidate(candidate); - } - - if (Array.isArray(body?.messages)) { - for (const msg of body.messages) { - if (!Array.isArray(msg?.content)) continue; - for (const part of msg.content) { - if (part?.type === "image_url") { - pushCandidate(part?.image_url?.url); - } - } - } - } - - return { - imageUrl: imageUrls[0] || null, - imageUrls, - maskUrl: - typeof body?.mask_url === "string" - ? body.mask_url - : typeof body?.mask === "string" - ? body.mask - : null, - }; -} - -async function resolveImageSource(source) { - if (typeof source !== "string" || source.trim().length === 0) { - throw new Error("Invalid image source"); - } - - const trimmed = source.trim(); - const dataUriMatch = /^data:([^;]+);base64,(.+)$/i.exec(trimmed); - if (dataUriMatch) { - const [, contentType, base64] = dataUriMatch; - return { - buffer: Buffer.from(base64, "base64"), - base64, - contentType, - }; - } - - if (isHttpUrl(trimmed)) { - const remoteImage = await fetchRemoteImage(trimmed); - return { - buffer: remoteImage.buffer, - base64: remoteImage.buffer.toString("base64"), - contentType: remoteImage.contentType, - }; - } - - return { - buffer: Buffer.from(trimmed, "base64"), - base64: trimmed, - contentType: "application/octet-stream", - }; -} - -function parseSizeToDimensions(size, fallback = 1024) { - if (typeof size !== "string" || !size.includes("x")) { - return { width: fallback, height: fallback }; - } - - const [widthRaw, heightRaw] = size.split("x"); - const width = Number(widthRaw); - const height = Number(heightRaw); - return { - width: Number.isFinite(width) && width > 0 ? width : fallback, - height: Number.isFinite(height) && height > 0 ? height : fallback, - }; -} - -function normalizeRequestedImageFormat( - body, - fallback = "png", - allowedFormats = ["jpeg", "png", "webp"] -) { - const formatCandidate = - typeof body?.output_format === "string" - ? body.output_format.toLowerCase() - : typeof body?.response_format === "string" && - !["url", "b64_json"].includes(body.response_format.toLowerCase()) - ? body.response_format.toLowerCase() - : fallback; - - if (allowedFormats.includes(formatCandidate)) { - return formatCandidate; - } - - return fallback; -} - -function mapFalImageSize(size, fallback = "square_hd") { - if (typeof size !== "string") return fallback; - if (FAL_PRESET_SIZES[size]) return FAL_PRESET_SIZES[size]; - if (size.includes("x")) { - const { width, height } = parseSizeToDimensions(size, 1024); - return { width, height }; - } - return fallback; -} - -function mapFalAspectRatio(size, fallback = "1:1") { - if (!size) return fallback; - return mapImageSize(size); -} - -function normalizeRecraftStyle(style) { - if (style === "vivid") return "digital_illustration"; - if (style === "natural") return "realistic_image"; - return style; -} - -function shouldIncludeStabilityMask(model) { - return new Set([ - "inpaint", - "erase", - "search-and-replace", - "search-and-recolor", - "replace-background-and-relight", - ]).has(model); -} - -async function normalizeProviderImagePayload(payload, body, log) { - const candidates = []; - - const pushCandidate = (value) => { - if (value === undefined || value === null) return; - candidates.push(value); - }; - - if (Array.isArray(payload?.data)) { - for (const item of payload.data) pushCandidate(item); - } - - if (Array.isArray(payload?.images)) { - for (const item of payload.images) pushCandidate(item); - } - - if (payload?.image) pushCandidate({ b64_json: payload.image }); - if (payload?.url) pushCandidate({ url: payload.url }); - if (payload?.sample) pushCandidate({ url: payload.sample }); - if (payload?.result?.sample) pushCandidate({ url: payload.result.sample }); - if (Array.isArray(payload?.result?.images)) { - for (const item of payload.result.images) pushCandidate(item); - } - - const normalized = []; - for (const candidate of candidates) { - const item = await normalizeProviderImageCandidate(candidate, body); - if (item) normalized.push(item); - } - - if (normalized.length === 0 && log) { - log.warn( - "IMAGE", - `Provider returned no recognizable image payload: ${JSON.stringify(payload).slice(0, 240)}` - ); - } - - return normalized; -} - -async function normalizeProviderImageCandidate(candidate, body) { - const wantsBase64 = body?.response_format === "b64_json"; - let url = null; - let b64 = null; - - if (typeof candidate === "string") { - const dataUriMatch = /^data:[^;]+;base64,(.+)$/i.exec(candidate); - if (dataUriMatch) { - b64 = dataUriMatch[1]; - } else if (isHttpUrl(candidate)) { - url = candidate; - } else { - b64 = candidate; - } - } else if (candidate && typeof candidate === "object") { - url = - firstString(candidate.url, candidate.image_url, candidate.sample, candidate.file_url) || null; - b64 = - firstString(candidate.b64_json, candidate.image, candidate.base64, candidate.data) || null; - } - - if (wantsBase64 && !b64 && url) { - b64 = (await resolveImageSource(url)).base64; - } - - if (url && !wantsBase64) { - return { url, revised_prompt: body?.prompt }; - } - - if (b64) { - return { b64_json: b64, revised_prompt: body?.prompt }; - } - - if (url) { - return { url, revised_prompt: body?.prompt }; - } - - return null; -} - -function firstString(...values) { - for (const value of values) { - if (typeof value === "string" && value.length > 0) return value; - } - return null; -} - -function isHttpUrl(value) { - return typeof value === "string" && /^https?:\/\//i.test(value); -} - -/** - * Codex image generation — translate GPT-Image-style /v1/images/generations - * request into a /v1/responses call with the `image_generation` hosted tool, - * parse the SSE stream, and return the base64 PNG in OpenAI image response shape. - * - * Requires ChatGPT OAuth credentials (Codex provider connection). The hosted - * image_generation tool is only served upstream under ChatGPT auth; API-key - * users will receive a 400 from OpenAI. - */ -export function extractImageGenerationCalls( - sseText: string -): Array<{ b64: string; revisedPrompt: string | null }> { - const results: Array<{ b64: string; revisedPrompt: string | null }> = []; - const lines = String(sseText || "").split("\n"); - for (const line of lines) { - const trimmed = line.trim(); - if (!trimmed.startsWith("data:")) continue; - const payload = trimmed.slice(5).trim(); - if (!payload || payload === "[DONE]") continue; - let evt: Record; - try { - evt = JSON.parse(payload) as Record; - } catch { - continue; - } - if (evt?.type !== "response.output_item.done") continue; - const item = evt.item as Record | undefined; - if (!item || item.type !== "image_generation_call") continue; - const result = typeof item.result === "string" ? item.result : ""; - if (!result) continue; - const revisedPrompt = typeof item.revised_prompt === "string" ? item.revised_prompt : null; - results.push({ b64: result, revisedPrompt }); - } - return results; -} - -// The image_generation hosted tool accepts { "auto" | "low" | "medium" | "high" } -// for `quality`. Legacy image clients often send "standard" / "hd". Map those values -// so OpenWebUI's quality dropdown doesn't silently get rejected upstream. -function mapLegacyImageQualityToImageTool(value: string): string { - const normalized = value.toLowerCase(); - if (normalized === "standard") return "medium"; - if (normalized === "hd") return "high"; - return normalized; -} - -async function handleCodexImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, -}) { - const startTime = Date.now(); - const prompt = typeof body.prompt === "string" ? body.prompt : ""; - if (!prompt.trim()) { - return saveImageErrorResult({ - provider, - model, - status: 400, - startTime, - error: "Prompt is required for Codex image generation", - }); - } - - const requestedCount = - Number.isInteger(body.n) && (body.n as number) > 0 ? (body.n as number) : 1; - if (log && requestedCount > 1) { - log.warn( - "IMAGE", - `Codex hosted image_generation returns one image per call; requested n=${requestedCount} will fan out in parallel` - ); - } - - const token = credentials?.accessToken || credentials?.apiKey; - if (!token) { - return saveImageErrorResult({ - provider, - model, - status: 401, - startTime, - error: "Codex credentials missing accessToken — reconnect the Codex provider", - }); - } - - const workspaceId = - credentials?.providerSpecificData && - typeof credentials.providerSpecificData === "object" && - !Array.isArray(credentials.providerSpecificData) - ? (credentials.providerSpecificData as Record).workspaceId - : undefined; - - // Forward size/quality from the GPT-Image-style body into the hosted tool so - // OpenWebUI's size/quality selectors actually take effect. Everything else - // (model, n, background, moderation, output_compression) is left to the - // Codex backend's defaults — today that's `gpt-image-2`. - const toolConfig: Record = { type: "image_generation", output_format: "png" }; - if (typeof body.size === "string" && body.size.trim()) { - toolConfig.size = body.size.trim(); - } - if (typeof body.quality === "string" && body.quality.trim()) { - toolConfig.quality = mapLegacyImageQualityToImageTool(body.quality.trim()); - } - - const upstreamBody: Record = { - model, - instructions: - "You must call the image_generation tool exactly once to fulfill the user's request. Do not add narration.", - input: [ - { - role: "user", - content: [{ type: "input_text", text: prompt }], - }, - ], - tools: [toolConfig], - stream: true, - store: false, - }; - - const headers: Record = { - "Content-Type": "application/json", - Accept: "text/event-stream", - Authorization: `Bearer ${token}`, - Version: getCodexClientVersion(), - "User-Agent": getCodexUserAgent(), - originator: "codex_cli_rs", - }; - if (typeof workspaceId === "string" && workspaceId) { - headers["chatgpt-account-id"] = workspaceId; - headers["session_id"] = workspaceId; - } - - if (log) { - log.info( - "IMAGE", - `${provider}/${model} (codex-responses) | prompt: "${prompt.slice(0, 60)}..."` - ); - } - - const fetchOneImage = async () => { - let response: Response; - try { - response = await fetch(providerConfig.baseUrl, { - method: "POST", - headers, - body: JSON.stringify(upstreamBody), - }); - } catch (err) { - if (log) log.error("IMAGE", `${provider} fetch error: ${(err as Error).message}`); - return { - ok: false as const, - error: { - provider, - model, - status: 502, - startTime, - error: `Image provider error: ${(err as Error).message}`, - requestBody: upstreamBody, - }, - }; - } - - if (!response.ok) { - const errorText = await response.text(); - if (log) - log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); - return { - ok: false as const, - error: { - provider, - model, - status: response.status, - startTime, - error: errorText, - requestBody: upstreamBody, - }, - }; - } - - const rawSSE = await response.text(); - const items = extractImageGenerationCalls(rawSSE); - if (items.length === 0) { - return { - ok: false as const, - error: { - provider, - model, - status: 502, - startTime, - error: - "Codex completed without producing an image_generation_call — the model may have declined the tool", - requestBody: upstreamBody, - }, - }; - } - - return { ok: true as const, items }; - }; - - const imageResults = await Promise.all( - Array.from({ length: requestedCount }, () => fetchOneImage()) - ); - - const collected: Array<{ b64_json: string; revised_prompt?: string }> = []; - for (const imageResult of imageResults) { - if (!imageResult.ok) return saveImageErrorResult(imageResult.error); - for (const item of imageResult.items) { - collected.push({ - b64_json: item.b64, - ...(item.revisedPrompt ? { revised_prompt: item.revisedPrompt } : {}), - }); - } - } - - const wantsUrl = body.response_format !== "b64_json"; - const data = wantsUrl - ? collected.map((item) => ({ - url: `data:image/png;base64,${item.b64_json}`, - ...(item.revised_prompt ? { revised_prompt: item.revised_prompt } : {}), - })) - : collected; - - return saveImageSuccessResult({ - provider, - model, - startTime, - requestBody: upstreamBody, - responseBody: { images_count: data.length }, - images: data, - }); -} - -function saveImageSuccessResult({ - provider, - model, - startTime, - requestBody = null, - responseBody = null, - created = null, - images, -}) { - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 200, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - requestBody, - responseBody, - }).catch(() => {}); - - return { - success: true, - data: { - created: created || Math.floor(Date.now() / 1000), - data: images, - }, - }; -} - -function saveImageErrorResult({ provider, model, status, startTime, error, requestBody = null }) { - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: typeof error === "string" ? error.slice(0, 500) : String(error).slice(0, 500), - requestBody, - }).catch(() => {}); - - return { - success: false, - status, - error, - }; -} - -/** - * Fetch a single image endpoint and normalize response - */ -async function fetchImageEndpoint(url, headers, body, provider, log) { - try { - let response; - try { - response = await fetchWithTimeout(url, { - method: "POST", - headers, - body, - timeoutMs: getConfiguredTimeout(), - }); - } catch (err: unknown) { - const isAbortError = - typeof err === "object" && - err !== null && - "name" in err && - (err as { name?: unknown }).name === "AbortError"; - if (err instanceof FetchTimeoutError || isAbortError) { - const message = err instanceof Error ? err.message : String(err); - if (log) { - log.error("IMAGE", `${provider} fetch error: ${message}`); - } - return { - success: false, - status: 504, - error: `Image provider error: ${sanitizeErrorMessage(message || err)}`, - }; - } - throw err; - } - - if (!response.ok) { - const errorText = await response.text(); - if (log) { - log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); - } - return { - success: false, - status: response.status, - error: errorText, - }; - } - - const data = await response.json(); - - // Normalize response to OpenAI format - return { - success: true, - data: { - created: data.created || Math.floor(Date.now() / 1000), - data: data.data || [], - }, - }; - } catch (err: unknown) { - const message = err instanceof Error ? err.message : String(err); - if (log) { - log.error("IMAGE", `${provider} fetch error: ${message}`); - } - return { - success: false, - status: 502, - error: `Image provider error: ${sanitizeErrorMessage(message || err)}`, - }; - } -} - -/** - * Handle Hyperbolic image generation - * Uses { model_name, prompt, height, width } and returns { images: [{ image: base64 }] } - */ -async function handleHyperbolicImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, -}) { - const startTime = Date.now(); - const token = credentials.apiKey || credentials.accessToken; - - const [width, height] = (body.size || "1024x1024").split("x").map(Number); - - const upstreamBody = { - model_name: model, - prompt: body.prompt, - height: height || 1024, - width: width || 1024, - backend: "auto", - }; - - if (log) { - const promptPreview = String(body.prompt ?? "").slice(0, 60); - log.info("IMAGE", `${provider}/${model} (hyperbolic) | prompt: "${promptPreview}..."`); - } - - try { - const response = await fetch(providerConfig.baseUrl, { - method: "POST", - headers: { - "Content-Type": "application/json", - Authorization: `Bearer ${token}`, - }, - body: JSON.stringify(upstreamBody), - }); - - if (!response.ok) { - const errorText = await response.text(); - if (log) - log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); - - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: response.status, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: errorText.slice(0, 500), - }).catch(() => {}); - - return { success: false, status: response.status, error: errorText }; - } - - const data = await response.json(); - // Transform { images: [{ image: base64 }] } → OpenAI format - const images = (data.images || []).map((img) => ({ - b64_json: img.image, - revised_prompt: body.prompt, - })); - - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 200, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - responseBody: { images_count: images.length }, - }).catch(() => {}); - - return { - success: true, - data: { created: Math.floor(Date.now() / 1000), data: images }, - }; - } catch (err) { - if (log) log.error("IMAGE", `${provider} fetch error: ${err.message}`); - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 502, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: err.message, - }).catch(() => {}); - return { - success: false, - status: 502, - error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, - }; - } -} - -/** - * Handle NanoBanana image generation - * NanoBanana is async (submit task -> poll status -> return final image URL/base64) - */ -async function handleNanoBananaImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, -}) { - const startTime = Date.now(); - const token = credentials.apiKey || credentials.accessToken; - - // Route to pro URL for "nanobanana-pro" model - const isPro = model === "nanobanana-pro"; - const submitUrl = isPro && providerConfig.proUrl ? providerConfig.proUrl : providerConfig.baseUrl; - const statusUrl = providerConfig.statusUrl; - - const aspectRatio = - typeof body.aspectRatio === "string" - ? body.aspectRatio - : typeof body.aspect_ratio === "string" - ? body.aspect_ratio - : mapImageSize(body.size); - - let resolution = - typeof body.resolution === "string" - ? body.resolution - : inferResolutionFromSize(body.size) || "1K"; - if (body.quality === "hd" && resolution === "1K") { - resolution = "2K"; - } - - const upstreamBody = isPro - ? { - prompt: body.prompt, - resolution, - aspectRatio, - ...(Array.isArray(body.imageUrls) ? { imageUrls: body.imageUrls } : {}), - } - : { - prompt: body.prompt, - type: - Array.isArray(body.imageUrls) && body.imageUrls.length > 0 - ? "IMAGETOIAMGE" - : "TEXTTOIAMGE", - numImages: Number.isFinite(body.n) ? Math.max(1, Number(body.n)) : 1, - image_size: aspectRatio, - ...(Array.isArray(body.imageUrls) ? { imageUrls: body.imageUrls } : {}), - }; - - if (log) { - const promptPreview = String(body.prompt ?? "").slice(0, 60); - log.info( - "IMAGE", - `${provider}/${model} (nanobanana ${isPro ? "pro" : "flash"}) | prompt: "${promptPreview}..."` - ); - } - - try { - const submitResp = await fetch(submitUrl, { - method: "POST", - headers: { - "Content-Type": "application/json", - Authorization: `Bearer ${token}`, - }, - body: JSON.stringify(upstreamBody), - }); - - if (!submitResp.ok) { - const errorText = await submitResp.text(); - if (log) { - log.error( - "IMAGE", - `${provider} submit error ${submitResp.status}: ${errorText.slice(0, 200)}` - ); - } - - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: submitResp.status, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: errorText.slice(0, 500), - }).catch(() => {}); - - return { success: false, status: submitResp.status, error: errorText }; - } - - const submitData = await submitResp.json(); - - // Backward compatibility: handle providers returning image payload synchronously - const hasSyncPayload = - Boolean(submitData?.image) || - Array.isArray(submitData?.images) || - Array.isArray(submitData?.data) || - Boolean(submitData?.data?.[0]?.url) || - Boolean(submitData?.data?.[0]?.b64_json); - - if (hasSyncPayload) { - const syncResult = normalizeNanoBananaSyncPayload(submitData, body.prompt); - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 200, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - responseBody: { images_count: syncResult.data?.length || 0, mode: "sync" }, - }).catch(() => {}); - return { - success: true, - data: { created: Math.floor(Date.now() / 1000), data: syncResult.data }, - }; - } - - const taskId = submitData?.data?.taskId || submitData?.taskId; - if (!taskId) { - const errorText = `NanoBanana submit did not return taskId: ${JSON.stringify(submitData).slice(0, 400)}`; - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 502, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: errorText, - }).catch(() => {}); - return { success: false, status: 502, error: errorText }; - } - - if (!statusUrl) { - const errorText = "NanoBanana statusUrl is not configured"; - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 500, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: errorText, - }).catch(() => {}); - return { success: false, status: 500, error: errorText }; - } - - const timeoutMs = normalizePositiveNumber( - body.timeout_ms, - normalizePositiveNumber(process.env.NANOBANANA_POLL_TIMEOUT_MS, 120000) - ); - const pollIntervalMs = normalizePositiveNumber( - body.poll_interval_ms, - normalizePositiveNumber(process.env.NANOBANANA_POLL_INTERVAL_MS, 2500) - ); - - let lastTaskData = null; - const deadline = Date.now() + timeoutMs; - - while (Date.now() < deadline) { - const pollResp = await fetch(`${statusUrl}?taskId=${encodeURIComponent(taskId)}`, { - method: "GET", - headers: { Authorization: `Bearer ${token}` }, - }); - - if (!pollResp.ok) { - const errorText = await pollResp.text(); - if (log) { - log.error( - "IMAGE", - `${provider} poll error ${pollResp.status}: ${errorText.slice(0, 200)}` - ); - } - return { success: false, status: pollResp.status, error: errorText }; - } - - const pollData = await pollResp.json(); - const taskData = pollData?.data || pollData; - lastTaskData = taskData; - - const successFlag = Number(taskData?.successFlag); - if (successFlag === 1) { - const normalized = await normalizeNanoBananaTaskResult(taskData, body, log); - - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 200, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - responseBody: { images_count: normalized.length, mode: "async", taskId }, - }).catch(() => {}); - - return { - success: true, - data: { - created: Math.floor(Date.now() / 1000), - data: normalized, - }, - }; - } - - if (successFlag === 2 || successFlag === 3) { - const errorText = - taskData?.errorMessage || `NanoBanana task failed (successFlag=${String(successFlag)})`; - - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 502, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: errorText.slice(0, 500), - responseBody: { taskId, successFlag, errorCode: taskData?.errorCode ?? null }, - }).catch(() => {}); - - return { success: false, status: 502, error: errorText }; - } - - await sleep(pollIntervalMs); - } - - const timeoutError = `NanoBanana task timeout after ${timeoutMs}ms (taskId=${taskId}, successFlag=${String(lastTaskData?.successFlag ?? "unknown")})`; - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 504, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: timeoutError, - responseBody: { taskId, lastSuccessFlag: lastTaskData?.successFlag ?? null }, - }).catch(() => {}); - - return { success: false, status: 504, error: timeoutError }; - } catch (err) { - if (log) log.error("IMAGE", `${provider} fetch error: ${err.message}`); - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 502, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: err.message, - }).catch(() => {}); - return { - success: false, - status: 502, - error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, - }; - } -} - -function normalizeNanoBananaSyncPayload(data, prompt) { - const images = []; - - if (data.image) { - images.push({ b64_json: data.image, revised_prompt: prompt }); - } else if (Array.isArray(data.images)) { - for (const img of data.images) { - images.push({ - b64_json: typeof img === "string" ? img : img?.image || img?.data, - revised_prompt: prompt, - }); - } - } else if (Array.isArray(data.data)) { - for (const img of data.data) { - if (!img) continue; - images.push(img); - } - } - - return { data: images.filter(Boolean) }; -} - -async function normalizeNanoBananaTaskResult(taskData, body, log) { - const response = taskData?.response || {}; - - const urlCandidates = [ - response?.resultImageUrl, - response?.originImageUrl, - taskData?.resultImageUrl, - taskData?.originImageUrl, - ].filter((v) => typeof v === "string" && v.length > 0); - - if (Array.isArray(response?.resultImageUrls)) { - for (const u of response.resultImageUrls) { - if (typeof u === "string" && u.length > 0) urlCandidates.push(u); - } - } - - const b64Candidates = [ - response?.resultImageBase64, - response?.resultImage, - taskData?.resultImageBase64, - taskData?.resultImage, - ].filter((v) => typeof v === "string" && v.length > 0); - - if (Array.isArray(response?.resultImageBase64List)) { - for (const b64 of response.resultImageBase64List) { - if (typeof b64 === "string" && b64.length > 0) b64Candidates.push(b64); - } - } - - const wantsBase64 = body.response_format === "b64_json"; - - if (wantsBase64) { - if (b64Candidates.length > 0) { - return b64Candidates.map((b64) => ({ b64_json: b64, revised_prompt: body.prompt })); - } - - if (urlCandidates.length > 0) { - const firstUrl = urlCandidates[0]; - const remoteImage = await fetchRemoteImage(firstUrl); - const base64 = remoteImage.buffer.toString("base64"); - return [{ b64_json: base64, revised_prompt: body.prompt }]; - } - } - - if (urlCandidates.length > 0) { - return urlCandidates.map((url) => ({ url, revised_prompt: body.prompt })); - } - - if (b64Candidates.length > 0) { - return b64Candidates.map((b64) => ({ b64_json: b64, revised_prompt: body.prompt })); - } - - if (log) { - log.warn( - "IMAGE", - `NanoBanana task completed without image payload: ${JSON.stringify(taskData).slice(0, 240)}` - ); - } - - return []; -} - -function inferResolutionFromSize(size) { - if (typeof size !== "string") return null; - const [wRaw, hRaw] = size.split("x"); - const width = Number(wRaw); - const height = Number(hRaw); - if (!Number.isFinite(width) || !Number.isFinite(height) || width <= 0 || height <= 0) return null; - - const longestSide = Math.max(width, height); - if (longestSide <= 1024) return "1K"; - if (longestSide <= 2048) return "2K"; - return "4K"; -} - -function normalizePositiveNumber(value, fallback) { - const n = Number(value); - if (!Number.isFinite(n) || n <= 0) return fallback; - return Math.floor(n); -} - -/** - * Handle SD WebUI image generation (local, no auth) - * POST {baseUrl} with { prompt, negative_prompt, width, height, steps } - * Response: { images: ["base64..."] } - */ -async function handleSDWebUIImageGeneration({ model, provider, providerConfig, body, log }) { - const startTime = Date.now(); - const [width, height] = (body.size || "512x512").split("x").map(Number); - - const upstreamBody = { - prompt: body.prompt, - negative_prompt: body.negative_prompt || "", - width: width || 512, - height: height || 512, - steps: body.steps || 20, - cfg_scale: body.cfg_scale || 7, - sampler_name: body.sampler || "Euler a", - batch_size: body.n || 1, - override_settings: { - sd_model_checkpoint: model, - }, - }; - - if (log) { - const promptPreview = String(body.prompt ?? "").slice(0, 60); - log.info("IMAGE", `${provider}/${model} (sdwebui) | prompt: "${promptPreview}..."`); - } - - try { - const response = await fetch(providerConfig.baseUrl, { - method: "POST", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify(upstreamBody), - }); - - if (!response.ok) { - const errorText = await response.text(); - if (log) - log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); - - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: response.status, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: errorText.slice(0, 500), - }).catch(() => {}); - - return { success: false, status: response.status, error: errorText }; - } - - const data = await response.json(); - // SD WebUI returns { images: ["base64...", ...] } - const images = (data.images || []).map((b64) => ({ - b64_json: b64, - revised_prompt: body.prompt, - })); - - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 200, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - responseBody: { images_count: images.length }, - }).catch(() => {}); - - return { - success: true, - data: { created: Math.floor(Date.now() / 1000), data: images }, - }; - } catch (err) { - if (log) log.error("IMAGE", `${provider} sdwebui error: ${err.message}`); - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 502, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: err.message, - }).catch(() => {}); - return { - success: false, - status: 502, - error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, - }; - } -} - -/** - * Handle ComfyUI image generation (local, no auth) - * Submits a txt2img workflow, polls for completion, fetches output - */ -async function handleComfyUIImageGeneration({ model, provider, providerConfig, body, log }) { - const startTime = Date.now(); - const [width, height] = (body.size || "1024x1024").split("x").map(Number); - - // Default txt2img workflow template for ComfyUI - const workflow = { - "3": { - class_type: "KSampler", - inputs: { - seed: parseInt(randomUUID().replace(/-/g, "").substring(0, 8), 16) % 2 ** 32, - steps: body.steps || 20, - cfg: body.cfg_scale || 7, - sampler_name: "euler", - scheduler: "normal", - denoise: 1, - model: ["4", 0], - positive: ["6", 0], - negative: ["7", 0], - latent_image: ["5", 0], - }, - }, - "4": { - class_type: "CheckpointLoaderSimple", - inputs: { ckpt_name: model }, - }, - "5": { - class_type: "EmptyLatentImage", - inputs: { width: width || 1024, height: height || 1024, batch_size: body.n || 1 }, - }, - "6": { - class_type: "CLIPTextEncode", - inputs: { text: body.prompt, clip: ["4", 1] }, - }, - "7": { - class_type: "CLIPTextEncode", - inputs: { text: body.negative_prompt || "", clip: ["4", 1] }, - }, - "8": { - class_type: "VAEDecode", - inputs: { samples: ["3", 0], vae: ["4", 2] }, - }, - "9": { - class_type: "SaveImage", - inputs: { filename_prefix: "omniroute", images: ["8", 0] }, - }, - }; - - if (log) { - const promptPreview = String(body.prompt ?? "").slice(0, 60); - log.info("IMAGE", `${provider}/${model} (comfyui) | prompt: "${promptPreview}..."`); - } - - try { - const promptId = await submitComfyWorkflow(providerConfig.baseUrl, workflow); - const historyEntry = await pollComfyResult(providerConfig.baseUrl, promptId); - const outputFiles = extractComfyOutputFiles(historyEntry); - - const images = []; - for (const file of outputFiles) { - const buffer = await fetchComfyOutput( - providerConfig.baseUrl, - file.filename, - file.subfolder, - file.type - ); - const base64 = Buffer.from(buffer).toString("base64"); - images.push({ b64_json: base64, revised_prompt: body.prompt }); - } - - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 200, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - responseBody: { images_count: images.length }, - }).catch(() => {}); - - return { - success: true, - data: { created: Math.floor(Date.now() / 1000), data: images }, - }; - } catch (err) { - if (log) log.error("IMAGE", `${provider} comfyui error: ${err.message}`); - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 502, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: err.message, - }).catch(() => {}); - return { - success: false, - status: 502, - error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, - }; - } -} - -async function handleHaiperImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, -}) { - const startTime = Date.now(); - const token = credentials?.apiKey || ""; - const prompt = typeof body.prompt === "string" ? body.prompt : String(body.prompt ?? ""); - if (log) { - log.info("IMAGE", `${provider}/${model} (haiper) | prompt: "${prompt.slice(0, 60)}..."`); - } - try { - const res = await fetch(providerConfig.baseUrl, { - method: "POST", - headers: { "Content-Type": "application/json", HAIPER_KEY: token }, - body: JSON.stringify({ prompt, aspect_ratio: body.aspect_ratio || "16:9" }), - }); - if (!res.ok) { - const errorText = await res.text(); - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: res.status, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: errorText.slice(0, 500), - }).catch(() => {}); - return { success: false, status: res.status, error: errorText }; - } - const { job_id } = await res.json(); - const deadline = Date.now() + 300000; - while (Date.now() < deadline) { - await sleep(5000); - const statusRes = await fetch(`${providerConfig.statusUrl}/${job_id}`, { - headers: { HAIPER_KEY: token }, - }); - const status = await statusRes.json(); - if (status.status === "completed" || status.status === "succeeded") { - const imgUrl = status.creation_url || status.output?.image_url; - if (imgUrl) { - const imgRes = await fetch(imgUrl); - if (!imgRes.ok) { - return { - success: false, - status: imgRes.status, - error: `Failed to download image: ${imgRes.status}`, - }; - } - const buf = await imgRes.arrayBuffer(); - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 200, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - }).catch(() => {}); - return { - success: true, - data: { - created: Math.floor(Date.now() / 1000), - data: [{ b64_json: Buffer.from(buf).toString("base64") }], - }, - }; - } - } - if (status.status === "failed") { - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 502, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: "Haiper image generation failed", - }).catch(() => {}); - return { success: false, status: 502, error: "Haiper image generation failed" }; - } - } - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 504, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: "Haiper image generation timed out", - }).catch(() => {}); - return { success: false, status: 504, error: "Haiper image generation timed out" }; - } catch (err) { - if (log) log.error("IMAGE", `${provider} haiper error: ${err.message}`); - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 502, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: err.message, - }).catch(() => {}); - return { - success: false, - status: 502, - error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, - }; - } -} - -async function handleLeonardoImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, -}) { - const startTime = Date.now(); - const token = credentials?.apiKey || ""; - const prompt = typeof body.prompt === "string" ? body.prompt : String(body.prompt ?? ""); - if (log) { - log.info("IMAGE", `${provider}/${model} (leonardo) | prompt: "${prompt.slice(0, 60)}..."`); - } - try { - const res = await fetch(providerConfig.baseUrl, { - method: "POST", - headers: { "Content-Type": "application/json", Authorization: `Bearer ${token}` }, - body: JSON.stringify({ - modelId: model || "phoenix", - prompt, - width: body.width || 1024, - height: body.height || 1024, - num_images: 1, - }), - }); - if (!res.ok) { - const errorText = await res.text(); - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: res.status, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: errorText.slice(0, 500), - }).catch(() => {}); - return { success: false, status: res.status, error: errorText }; - } - const { sdGenerationJob } = await res.json(); - const genId = sdGenerationJob?.generationId; - if (!genId) { - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 502, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: "No generation ID returned", - }).catch(() => {}); - return { success: false, status: 502, error: "No generation ID returned" }; - } - const deadline = Date.now() + 300000; - while (Date.now() < deadline) { - await sleep(5000); - const statusRes = await fetch(`${providerConfig.baseUrl}/${genId}`, { - headers: { Authorization: `Bearer ${token}` }, - }); - const status = await statusRes.json(); - const gen = status.generations_by_pk || status; - if (gen.status === "COMPLETE") { - const imgUrl = gen.generated_images?.[0]?.url; - if (imgUrl) { - const imgRes = await fetch(imgUrl); - if (!imgRes.ok) { - return { - success: false, - status: imgRes.status, - error: `Failed to download image: ${imgRes.status}`, - }; - } - const buf = await imgRes.arrayBuffer(); - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 200, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - }).catch(() => {}); - return { - success: true, - data: { - created: Math.floor(Date.now() / 1000), - data: [{ b64_json: Buffer.from(buf).toString("base64") }], - }, - }; - } - } - if (gen.status === "FAILED") { - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 502, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: "Leonardo image generation failed", - }).catch(() => {}); - return { success: false, status: 502, error: "Leonardo image generation failed" }; - } - } - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 504, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: "Leonardo image generation timed out", - }).catch(() => {}); - return { success: false, status: 504, error: "Leonardo image generation timed out" }; - } catch (err) { - if (log) log.error("IMAGE", `${provider} leonardo error: ${err.message}`); - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 502, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: err.message, - }).catch(() => {}); - return { - success: false, - status: 502, - error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, - }; - } -} - -async function handleIdeogramImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, -}) { - const startTime = Date.now(); - const token = credentials?.apiKey || ""; - const prompt = typeof body.prompt === "string" ? body.prompt : String(body.prompt ?? ""); - if (log) { - log.info("IMAGE", `${provider}/${model} (ideogram) | prompt: "${prompt.slice(0, 60)}..."`); - } - try { - const res = await fetch(providerConfig.baseUrl, { - method: "POST", - headers: { "Content-Type": "application/json", "Api-Key": token }, - body: JSON.stringify({ prompt, aspect_ratio: "ASPECT_16_9", model: model || "V_3" }), - }); - if (!res.ok) { - const errorText = await res.text(); - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: res.status, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: errorText.slice(0, 500), - }).catch(() => {}); - return { success: false, status: res.status, error: errorText }; - } - const data = await res.json(); - if (data.data && data.data.length > 0) { - const imgUrl = data.data[0].url; - const imgRes = await fetch(imgUrl); - if (!imgRes.ok) { - return { - success: false, - status: imgRes.status, - error: `Failed to download image: ${imgRes.status}`, - }; - } - const buf = await imgRes.arrayBuffer(); - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 200, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - }).catch(() => {}); - return { - success: true, - data: { - created: Math.floor(Date.now() / 1000), - data: [{ b64_json: Buffer.from(buf).toString("base64") }], - }, - }; - } - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 502, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: "No images returned from Ideogram", - }).catch(() => {}); - return { success: false, status: 502, error: "No images returned from Ideogram" }; - } catch (err) { - if (log) log.error("IMAGE", `${provider} ideogram error: ${err.message}`); - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 502, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: err.message, - }).catch(() => {}); - return { - success: false, - status: 502, - error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, - }; - } -} - -type Imagen3ImageGenArgs = { - model: string; - provider: string; - providerConfig: { baseUrl: string }; - body: { prompt?: string; size?: string; n?: number }; - credentials: { apiKey?: string; accessToken?: string }; - log?: { - info?: (tag: string, msg: string) => void; - error?: (tag: string, msg: string) => void; - } | null; -}; - -type Imagen3NormalizedImage = { - b64_json?: unknown; - url?: unknown; - revised_prompt?: string; -}; - -/** - * Handle Imagen 3 image generation - */ -async function handleImagen3ImageGeneration({ - model, - provider, - providerConfig, - body, - credentials, - log, -}: Imagen3ImageGenArgs) { - const startTime = Date.now(); - const token = credentials.apiKey || credentials.accessToken; - const aspectRatio = mapImageSize(body.size); - - const upstreamBody = { - prompt: body.prompt, - aspect_ratio: aspectRatio, - number_of_images: body.n ?? 1, - }; - - if (log) { - const promptPreview = String(body.prompt ?? "").slice(0, 60); - log.info( - "IMAGE", - `${provider}/${model} (imagen3) | prompt: "${promptPreview}..." | aspect_ratio: ${aspectRatio}` - ); - } - - try { - const response = await fetch(providerConfig.baseUrl, { - method: "POST", - headers: { - "Content-Type": "application/json", - Authorization: `Bearer ${token}`, - }, - body: JSON.stringify(upstreamBody), - }); - - if (!response.ok) { - const errorText = await response.text(); - if (log) - log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); - - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: response.status, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: errorText.slice(0, 500), - requestBody: upstreamBody, - }).catch(() => {}); - - return { success: false, status: response.status, error: errorText }; - } - - const data = await response.json(); - - // Normalize response to OpenAI format - const images: Imagen3NormalizedImage[] = []; - if (Array.isArray(data.images)) { - images.push( - ...data.images.map((img: Record) => ({ - b64_json: img.image ?? img.b64_json ?? img.url ?? img, - revised_prompt: body.prompt, - })) - ); - } else if (Array.isArray(data.data)) { - images.push(...data.data); - } else if (data.url || data.b64_json || data.image) { - images.push({ - b64_json: data.image || data.b64_json || data.url, - url: data.url, - revised_prompt: body.prompt, - }); - } - - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 200, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - responseBody: { images_count: images.length }, - }).catch(() => {}); - - return { - success: true, - data: { created: data.created || Math.floor(Date.now() / 1000), data: images }, - }; - } catch (err: unknown) { - const errMsg = err instanceof Error ? err.message : String(err); - if (log) log.error("IMAGE", `${provider} fetch error: ${errMsg}`); - - saveCallLog({ - method: "POST", - path: "/v1/images/generations", - status: 502, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: errMsg, - }).catch(() => {}); - - return { success: false, status: 502, error: `Image provider error: ${errMsg}` }; - } -} +export * from "./imageGeneration/index.ts"; diff --git a/open-sse/handlers/imageGeneration/blackForestLabs.ts b/open-sse/handlers/imageGeneration/blackForestLabs.ts new file mode 100644 index 00000000000..c009fc344bb --- /dev/null +++ b/open-sse/handlers/imageGeneration/blackForestLabs.ts @@ -0,0 +1,221 @@ + +/** + * Image Generation Handler + * + * Handles POST /v1/images/generations requests. + * Proxies to upstream image generation providers using OpenAI-compatible format. + * + * Request format (OpenAI-compatible): + * { + * "model": "openai/gpt-image-2", + * "prompt": "a beautiful sunset over mountains", + * "n": 1, + * "size": "1024x1024", + * "quality": "standard", // optional: "standard" | "hd" + * "response_format": "url" // optional: "url" | "b64_json" + * } + */ + +import { getImageProvider, parseImageModel } from "../../config/imageRegistry.ts"; + +import { mapImageSize } from "../../translator/image/sizeMapper.ts"; + +import { sleep } from "../../utils/sleep.ts"; + +import { sanitizeErrorMessage, sanitizeUpstreamDetails } from "../../utils/error.ts"; + +import { extractImageInputs, normalizeRequestedImageFormat, resolveImageSource, isHttpUrl, parseSizeToDimensions, normalizeProviderImagePayload, normalizePositiveNumber } from "./utils.ts"; +import { saveImageErrorResult, saveImageSuccessResult } from "./logging.ts"; + + + +const BFL_MODEL_ENDPOINTS = { + "flux-2-max": "/v1/flux-2-max", + "flux-2-pro": "/v1/flux-2-pro", + "flux-2-flex": "/v1/flux-2-flex", + "flux-2-klein-9b": "/v1/flux-2-klein-9b", + "flux-2-klein-4b": "/v1/flux-2-klein-4b", + "flux-kontext-pro": "/v1/flux-kontext-pro", + "flux-kontext-max": "/v1/flux-kontext-max", + "flux-pro-1.1": "/v1/flux-pro-1.1", + "flux-pro-1.1-ultra": "/v1/flux-pro-1.1-ultra", + "flux-dev": "/v1/flux-dev", + "flux-pro": "/v1/flux-pro", +}; + + + +const BFL_EDIT_MODELS = new Set([ + "flux-2-max", + "flux-2-pro", + "flux-2-flex", + "flux-kontext-pro", + "flux-kontext-max", +]); + + + +const BFL_FAILURE_STATUSES = new Set(["Error", "Failed", "Content Moderated", "Request Moderated"]); + + + +export async function handleBlackForestLabsImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, +}) { + const startTime = Date.now(); + const token = credentials.apiKey || credentials.accessToken; + const endpoint = BFL_MODEL_ENDPOINTS[model]; + + if (!endpoint) { + return { + success: false, + status: 400, + error: `Unsupported Black Forest Labs image model: ${model}`, + }; + } + + const { imageUrl, maskUrl } = extractImageInputs(body); + const upstreamBody: Record = { + prompt: body.prompt, + output_format: normalizeRequestedImageFormat(body, "png"), + }; + + try { + if (BFL_EDIT_MODELS.has(model) && imageUrl) { + upstreamBody.input_image = (await resolveImageSource(imageUrl)).base64; + } else if (imageUrl && isHttpUrl(imageUrl)) { + upstreamBody.image_url = imageUrl; + } + + if (maskUrl && (model === "flux-pro-1.0-fill" || model === "flux-kontext-pro")) { + upstreamBody.mask = (await resolveImageSource(maskUrl)).base64; + } + + if (model === "flux-kontext-pro" || model === "flux-kontext-max") { + upstreamBody.aspect_ratio = body.aspect_ratio || mapImageSize(body.size); + } else if (typeof body.size === "string" && body.size.includes("x")) { + const { width, height } = parseSizeToDimensions(body.size, 1024); + upstreamBody.width = width; + upstreamBody.height = height; + } + + if (body.seed !== undefined) upstreamBody.seed = body.seed; + if (body.n !== undefined && model.includes("ultra")) + upstreamBody.num_images = Number(body.n) || 1; + if (body.quality === "hd" && model.includes("ultra")) upstreamBody.raw = true; + if (body.left !== undefined) upstreamBody.left = body.left; + if (body.right !== undefined) upstreamBody.right = body.right; + if (body.top !== undefined) upstreamBody.top = body.top; + if (body.bottom !== undefined) upstreamBody.bottom = body.bottom; + if (body.steps !== undefined) upstreamBody.steps = body.steps; + if (body.guidance !== undefined) upstreamBody.guidance = body.guidance; + if (body.grow_mask !== undefined) upstreamBody.grow_mask = body.grow_mask; + if (body.safety_tolerance !== undefined) upstreamBody.safety_tolerance = body.safety_tolerance; + + if (log) { + const promptPreview = String(body.prompt ?? "").slice(0, 60); + log.info("IMAGE", `${provider}/${model} (black-forest-labs) | prompt: "${promptPreview}..."`); + } + + const response = await fetch(`${providerConfig.baseUrl.replace(/\/$/, "")}${endpoint}`, { + method: "POST", + headers: { + "Content-Type": "application/json", + Accept: "application/json", + "x-key": token, + }, + body: JSON.stringify(upstreamBody), + }); + + if (!response.ok) { + const errorText = await response.text(); + if (log) + log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); + return saveImageErrorResult({ + provider, + model, + status: response.status, + startTime, + error: errorText, + requestBody: upstreamBody, + }); + } + + const initialPayload = await response.json(); + const finalPayload = initialPayload.polling_url + ? await pollBlackForestLabsResult({ + pollingUrl: initialPayload.polling_url, + token, + body, + log, + }) + : initialPayload; + + const images = await normalizeProviderImagePayload(finalPayload, body, log); + return saveImageSuccessResult({ + provider, + model, + startTime, + requestBody: upstreamBody, + responseBody: { images_count: images.length }, + created: finalPayload.created, + images, + }); + } catch (err) { + if (log) log.error("IMAGE", `${provider} fetch error: ${err.message}`); + return saveImageErrorResult({ + provider, + model, + status: 502, + startTime, + error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, + }); + } +} + + + +async function pollBlackForestLabsResult({ pollingUrl, token, body, log }) { + const timeoutMs = normalizePositiveNumber(body.timeout_ms, 300000); + const pollIntervalMs = normalizePositiveNumber(body.poll_interval_ms, 1500); + const deadline = Date.now() + timeoutMs; + + while (Date.now() < deadline) { + const response = await fetch(pollingUrl, { + method: "GET", + headers: { + "x-key": token, + }, + }); + + if (!response.ok) { + const errorText = await response.text(); + throw new Error(`BFL polling failed (${response.status}): ${errorText}`); + } + + const payload = await response.json(); + const status = payload?.status; + + if (status === "Ready") { + return payload; + } + + if (BFL_FAILURE_STATUSES.has(status)) { + throw new Error(`BFL image generation failed: ${status}`); + } + + if (log) { + log.info("IMAGE", `black-forest-labs polling status: ${String(status || "Pending")}`); + } + + await sleep(pollIntervalMs); + } + + throw new Error(`BFL polling timed out after ${timeoutMs}ms`); +} + diff --git a/open-sse/handlers/imageGeneration/chatgptWeb.ts b/open-sse/handlers/imageGeneration/chatgptWeb.ts new file mode 100644 index 00000000000..24302aa1b37 --- /dev/null +++ b/open-sse/handlers/imageGeneration/chatgptWeb.ts @@ -0,0 +1,207 @@ + +/** + * Image Generation Handler + * + * Handles POST /v1/images/generations requests. + * Proxies to upstream image generation providers using OpenAI-compatible format. + * + * Request format (OpenAI-compatible): + * { + * "model": "openai/gpt-image-2", + * "prompt": "a beautiful sunset over mountains", + * "n": 1, + * "size": "1024x1024", + * "quality": "standard", // optional: "standard" | "hd" + * "response_format": "url" // optional: "url" | "b64_json" + * } + */ + +import { getImageProvider, parseImageModel } from "../../config/imageRegistry.ts"; + +import { mapImageSize } from "../../translator/image/sizeMapper.ts"; + +import { ChatGptWebExecutor } from "../../executors/chatgpt-web.ts"; + +import { getChatGptImage, findChatGptImageBySha256 } from "../../services/chatgptImageCache.ts"; + +import { sanitizeErrorMessage, sanitizeUpstreamDetails } from "../../utils/error.ts"; + +import { saveImageErrorResult, saveImageSuccessResult } from "./logging.ts"; + + + +const CHATGPT_WEB_IMAGE_MARKDOWN_RE = /!\[[^\]]*\]\(([^)\s]+)\)/g; + + +export const CHATGPT_WEB_IMAGE_ID_RE = /\/v1\/chatgpt-web\/image\/([a-f0-9]{16,64})(?=[?\s"'<>)]|$)/i; + + + +export function extractMarkdownImageUrls(text: string): string[] { + const urls: string[] = []; + // String.prototype.matchAll consumes a fresh iterator and ignores the + // regex's lastIndex, so no manual reset is required. + for (const match of text.matchAll(CHATGPT_WEB_IMAGE_MARKDOWN_RE)) { + if (match[1]) urls.push(match[1]); + } + return urls; +} + + + +function buildChatGptWebImagePrompt(body): string { + const prompt = String(body.prompt || "").trim(); + const details: string[] = [`Create an image for this prompt: ${prompt}`]; + if (typeof body.size === "string" && body.size.trim()) { + details.push(`Requested size: ${body.size.trim()}.`); + } + if (typeof body.quality === "string" && body.quality.trim()) { + details.push(`Requested quality: ${body.quality.trim()}.`); + } + if (typeof body.style === "string" && body.style.trim()) { + details.push(`Requested style: ${body.style.trim()}.`); + } + return details.join("\n"); +} + + + +export async function handleChatGptWebImageGeneration({ + model, + provider, + body, + credentials, + log, + signal, + clientHeaders, +}) { + const startTime = Date.now(); + const prompt = typeof body.prompt === "string" ? body.prompt.trim() : ""; + if (!prompt) { + return saveImageErrorResult({ + provider, + model, + status: 400, + startTime, + error: "Prompt is required for ChatGPT Web image generation", + }); + } + + if (!credentials?.apiKey) { + return saveImageErrorResult({ + provider, + model, + status: 401, + startTime, + error: "ChatGPT Web credentials missing session cookie", + }); + } + + // Each image is one chatgpt.com chat turn (~30s). Cap at 4 (matches OpenAI's + // own limit for GPT Image models) so a stray n=1000 doesn't pin the + // executor for hours before the upstream HTTP timeout fires. + const CHATGPT_WEB_IMAGE_N_MAX = 4; + const rawCount = Number.isInteger(body.n) && (body.n as number) > 0 ? (body.n as number) : 1; + if (rawCount > CHATGPT_WEB_IMAGE_N_MAX) { + return saveImageErrorResult({ + provider, + model, + status: 400, + startTime, + error: `ChatGPT Web image generation supports n=1..${CHATGPT_WEB_IMAGE_N_MAX} (got ${rawCount}); each n is a separate ~30s chat turn.`, + }); + } + const requestedCount = rawCount; + if (log && requestedCount > 1) { + log.warn( + "IMAGE", + `ChatGPT Web returns one image per chat turn; requested n=${requestedCount} will run sequentially` + ); + } + + const wantsBase64 = body.response_format === "b64_json"; + const images: Array<{ url?: string; b64_json?: string }> = []; + const requestBody = { + model, + prompt: prompt.slice(0, 500), + size: body.size || undefined, + quality: body.quality || undefined, + }; + + for (let i = 0; i < requestedCount; i++) { + const executor = new ChatGptWebExecutor(); + const result = await executor.execute({ + model, + body: { + messages: [{ role: "user", content: buildChatGptWebImagePrompt(body) }], + }, + stream: false, + credentials, + signal, + log, + clientHeaders, + }); + + const responseText = await result.response.text(); + if (result.response.status >= 400) { + return saveImageErrorResult({ + provider, + model, + status: result.response.status, + startTime, + error: responseText, + requestBody, + }); + } + + let content = ""; + try { + const json = JSON.parse(responseText); + content = String(json?.choices?.[0]?.message?.content || ""); + } catch { + content = responseText; + } + + const urls = extractMarkdownImageUrls(content); + if (urls.length === 0) { + return saveImageErrorResult({ + provider, + model, + status: 502, + startTime, + error: `ChatGPT Web completed without returning image markdown: ${content.slice(0, 300)}`, + requestBody, + }); + } + + for (const url of urls) { + if (!wantsBase64) { + images.push({ url }); + continue; + } + const id = url.match(CHATGPT_WEB_IMAGE_ID_RE)?.[1]; + const cached = id ? getChatGptImage(id) : null; + if (!cached) { + return saveImageErrorResult({ + provider, + model, + status: 502, + startTime, + error: "ChatGPT Web image bytes expired before b64_json conversion", + requestBody, + }); + } + images.push({ b64_json: cached.bytes.toString("base64") }); + } + } + + return saveImageSuccessResult({ + provider, + model, + startTime, + requestBody, + responseBody: { images_count: images.length }, + images, + }); +} + diff --git a/open-sse/handlers/imageGeneration/codex.ts b/open-sse/handlers/imageGeneration/codex.ts new file mode 100644 index 00000000000..0357c6d7080 --- /dev/null +++ b/open-sse/handlers/imageGeneration/codex.ts @@ -0,0 +1,220 @@ + +/** + * Image Generation Handler + * + * Handles POST /v1/images/generations requests. + * Proxies to upstream image generation providers using OpenAI-compatible format. + * + * Request format (OpenAI-compatible): + * { + * "model": "openai/gpt-image-2", + * "prompt": "a beautiful sunset over mountains", + * "n": 1, + * "size": "1024x1024", + * "quality": "standard", // optional: "standard" | "hd" + * "response_format": "url" // optional: "url" | "b64_json" + * } + */ + +import { getImageProvider, parseImageModel } from "../../config/imageRegistry.ts"; + +import { mapImageSize } from "../../translator/image/sizeMapper.ts"; + +import { getCodexClientVersion, getCodexUserAgent } from "../../config/codexClient.ts"; + +import { ChatGptWebExecutor } from "../../executors/chatgpt-web.ts"; + +import { sanitizeErrorMessage, sanitizeUpstreamDetails } from "../../utils/error.ts"; + +import { saveImageErrorResult, saveImageSuccessResult } from "./logging.ts"; +import { mapLegacyImageQualityToImageTool, extractImageGenerationCalls } from "./utils.ts"; + + + +export async function handleCodexImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, +}) { + const startTime = Date.now(); + const prompt = typeof body.prompt === "string" ? body.prompt : ""; + if (!prompt.trim()) { + return saveImageErrorResult({ + provider, + model, + status: 400, + startTime, + error: "Prompt is required for Codex image generation", + }); + } + + const requestedCount = + Number.isInteger(body.n) && (body.n as number) > 0 ? (body.n as number) : 1; + if (log && requestedCount > 1) { + log.warn( + "IMAGE", + `Codex hosted image_generation returns one image per call; requested n=${requestedCount} will fan out in parallel` + ); + } + + const token = credentials?.accessToken || credentials?.apiKey; + if (!token) { + return saveImageErrorResult({ + provider, + model, + status: 401, + startTime, + error: "Codex credentials missing accessToken — reconnect the Codex provider", + }); + } + + const workspaceId = + credentials?.providerSpecificData && + typeof credentials.providerSpecificData === "object" && + !Array.isArray(credentials.providerSpecificData) + ? (credentials.providerSpecificData as Record).workspaceId + : undefined; + + // Forward size/quality from the GPT-Image-style body into the hosted tool so + // OpenWebUI's size/quality selectors actually take effect. Everything else + // (model, n, background, moderation, output_compression) is left to the + // Codex backend's defaults — today that's `gpt-image-2`. + const toolConfig: Record = { type: "image_generation", output_format: "png" }; + if (typeof body.size === "string" && body.size.trim()) { + toolConfig.size = body.size.trim(); + } + if (typeof body.quality === "string" && body.quality.trim()) { + toolConfig.quality = mapLegacyImageQualityToImageTool(body.quality.trim()); + } + + const upstreamBody: Record = { + model, + instructions: + "You must call the image_generation tool exactly once to fulfill the user's request. Do not add narration.", + input: [ + { + role: "user", + content: [{ type: "input_text", text: prompt }], + }, + ], + tools: [toolConfig], + stream: true, + store: false, + }; + + const headers: Record = { + "Content-Type": "application/json", + Accept: "text/event-stream", + Authorization: `Bearer ${token}`, + Version: getCodexClientVersion(), + "User-Agent": getCodexUserAgent(), + originator: "codex_cli_rs", + }; + if (typeof workspaceId === "string" && workspaceId) { + headers["chatgpt-account-id"] = workspaceId; + headers["session_id"] = workspaceId; + } + + if (log) { + log.info( + "IMAGE", + `${provider}/${model} (codex-responses) | prompt: "${prompt.slice(0, 60)}..."` + ); + } + + const fetchOneImage = async () => { + let response: Response; + try { + response = await fetch(providerConfig.baseUrl, { + method: "POST", + headers, + body: JSON.stringify(upstreamBody), + }); + } catch (err) { + if (log) log.error("IMAGE", `${provider} fetch error: ${(err as Error).message}`); + return { + ok: false as const, + error: { + provider, + model, + status: 502, + startTime, + error: `Image provider error: ${(err as Error).message}`, + requestBody: upstreamBody, + }, + }; + } + + if (!response.ok) { + const errorText = await response.text(); + if (log) + log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); + return { + ok: false as const, + error: { + provider, + model, + status: response.status, + startTime, + error: errorText, + requestBody: upstreamBody, + }, + }; + } + + const rawSSE = await response.text(); + const items = extractImageGenerationCalls(rawSSE); + if (items.length === 0) { + return { + ok: false as const, + error: { + provider, + model, + status: 502, + startTime, + error: + "Codex completed without producing an image_generation_call — the model may have declined the tool", + requestBody: upstreamBody, + }, + }; + } + + return { ok: true as const, items }; + }; + + const imageResults = await Promise.all( + Array.from({ length: requestedCount }, () => fetchOneImage()) + ); + + const collected: Array<{ b64_json: string; revised_prompt?: string }> = []; + for (const imageResult of imageResults) { + if (!imageResult.ok) return saveImageErrorResult(imageResult.error); + for (const item of imageResult.items) { + collected.push({ + b64_json: item.b64, + ...(item.revisedPrompt ? { revised_prompt: item.revisedPrompt } : {}), + }); + } + } + + const wantsUrl = body.response_format !== "b64_json"; + const data = wantsUrl + ? collected.map((item) => ({ + url: `data:image/png;base64,${item.b64_json}`, + ...(item.revised_prompt ? { revised_prompt: item.revised_prompt } : {}), + })) + : collected; + + return saveImageSuccessResult({ + provider, + model, + startTime, + requestBody: upstreamBody, + responseBody: { images_count: data.length }, + images: data, + }); +} + diff --git a/open-sse/handlers/imageGeneration/core.ts b/open-sse/handlers/imageGeneration/core.ts new file mode 100644 index 00000000000..ce9b38b951d --- /dev/null +++ b/open-sse/handlers/imageGeneration/core.ts @@ -0,0 +1,316 @@ + +/** + * Image Generation Handler + * + * Handles POST /v1/images/generations requests. + * Proxies to upstream image generation providers using OpenAI-compatible format. + * + * Request format (OpenAI-compatible): + * { + * "model": "openai/gpt-image-2", + * "prompt": "a beautiful sunset over mountains", + * "n": 1, + * "size": "1024x1024", + * "quality": "standard", // optional: "standard" | "hd" + * "response_format": "url" // optional: "url" | "b64_json" + * } + */ + +import { getImageProvider, parseImageModel } from "../../config/imageRegistry.ts"; + +import { HTTP_STATUS } from "../../config/constants.ts"; + +import { applyAntigravityClientProfileHeaders } from "../../services/antigravityClientProfile.ts"; + +import { getAntigravityEnvelopeUserAgent } from "../../services/antigravityIdentity.ts"; + +import { kieExecutor } from "../../executors/kie.ts"; + +import { mapImageSize } from "../../translator/image/sizeMapper.ts"; + +import { getCodexClientVersion, getCodexUserAgent } from "../../config/codexClient.ts"; + +import { ChatGptWebExecutor } from "../../executors/chatgpt-web.ts"; + +import { getChatGptImage, findChatGptImageBySha256 } from "../../services/chatgptImageCache.ts"; + +import { createHash } from "node:crypto"; + +import { sleep } from "../../utils/sleep.ts"; + +import { + getKieErrorMessage, + getKieErrorStatus, + isJsonObject, + parseKieResultJson, +} from "../../utils/kieTask.ts"; + +import { + submitComfyWorkflow, + pollComfyResult, + fetchComfyOutput, + extractComfyOutputFiles, +} from "../../utils/comfyuiClient.ts"; + +import { sanitizeErrorMessage, sanitizeUpstreamDetails } from "../../utils/error.ts"; + +import { resolveImageBaseUrl } from "./utils.ts"; +import { handleOpenAIImageGeneration } from "./openai.ts"; +import { handleGeminiImageGeneration } from "./gemini.ts"; +import { handleImagen3ImageGeneration } from "./imagen3.ts"; +import { handleHyperbolicImageGeneration, handleRecraftImageGeneration, handleTopazImageGeneration, handleNanoBananaImageGeneration, handleSDWebUIImageGeneration, handleComfyUIImageGeneration, handleHaiperImageGeneration, handleLeonardoImageGeneration, handleIdeogramImageGeneration } from "./specialty.ts"; +import { handleFalAIImageGeneration } from "./fal.ts"; +import { handleStabilityAIImageGeneration } from "./stability.ts"; +import { handleBlackForestLabsImageGeneration } from "./blackForestLabs.ts"; +import { handleChatGptWebImageGeneration } from "./chatgptWeb.ts"; +import { handleKieImageGeneration } from "./kie.ts"; +import { handleCodexImageGeneration } from "./codex.ts"; + + + +/** + * Handle image generation request + * @param {object} options + * @param {object} options.body - Request body + * @param {object} options.credentials - Provider credentials { apiKey, accessToken } + * @param {object} options.log - Logger + * @param {string} [options.resolvedProvider] - Pre-resolved provider ID (from route layer custom model resolution) + */ +export async function handleImageGeneration({ + body, + credentials, + log, + resolvedProvider = null, + signal = null, + clientHeaders = null, +}) { + let provider, model; + + if (resolvedProvider) { + // Provider was already resolved by the route layer (custom model from DB) + // Extract model name from the full "provider/model" string + provider = resolvedProvider; + const modelStr = body.model || ""; + model = modelStr.startsWith(provider + "/") ? modelStr.slice(provider.length + 1) : modelStr; + } else { + // Standard path: resolve from built-in image registry + const parsed = parseImageModel(body.model); + provider = parsed.provider; + model = parsed.model; + } + + if (!provider) { + return { + success: false, + status: 400, + error: `Invalid image model: ${body.model}. Use format: provider/model`, + }; + } + + const providerConfig = getImageProvider(provider); + + // For custom models without a built-in provider config, use OpenAI-compatible handler + // with a synthetic config based on the provider's credentials + if (!providerConfig) { + if (!resolvedProvider) { + return { + success: false, + status: 400, + error: `Unknown image provider: ${provider}`, + }; + } + + // Custom model: use OpenAI-compatible format with provider's base URL + // The credentials were already resolved by the route layer + if (log) { + log.info("IMAGE", `Custom model ${provider}/${model} — using OpenAI-compatible handler`); + } + + const syntheticConfig = { + id: provider, + // #3205: custom OpenAI-compatible nodes store their base URL in + // credentials.providerSpecificData.baseUrl (same as the chat path — + // see executors/default.ts:buildUrl / services/provider.ts:buildProviderUrl). + // Previously only the (always-absent) top-level credentials.baseUrl was + // read, so every custom image node fell back to the Gemini endpoint and + // returned "Please pass a valid API key". + baseUrl: resolveImageBaseUrl( + credentials, + `https://generativelanguage.googleapis.com/v1beta/openai/images/generations` + ), + authType: "apikey", + authHeader: "bearer", + format: "openai", + }; + + return handleOpenAIImageGeneration({ + model, + provider, + providerConfig: syntheticConfig, + body, + credentials, + log, + }); + } + + if (providerConfig.format === "gemini-image") { + return handleGeminiImageGeneration({ model, providerConfig, body, credentials, log }); + } + + if (providerConfig.format === "imagen3") { + return handleImagen3ImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, + }); + } + + if (providerConfig.format === "hyperbolic") { + return handleHyperbolicImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, + }); + } + + if (providerConfig.format === "fal-ai") { + return handleFalAIImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, + }); + } + + if (providerConfig.format === "stability-ai") { + return handleStabilityAIImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, + }); + } + + if (providerConfig.format === "black-forest-labs") { + return handleBlackForestLabsImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, + }); + } + + if (providerConfig.format === "recraft") { + return handleRecraftImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, + }); + } + + if (providerConfig.format === "topaz") { + return handleTopazImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, + }); + } + + if (providerConfig.format === "chatgpt-web") { + return handleChatGptWebImageGeneration({ + model, + provider, + body, + credentials, + log, + signal, + clientHeaders, + }); + } + + if (providerConfig.format === "nanobanana") { + return handleNanoBananaImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, + }); + } + + if (providerConfig.format === "kie-image") { + return handleKieImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, + }); + } + + if (providerConfig.format === "sdwebui") { + return handleSDWebUIImageGeneration({ model, provider, providerConfig, body, log }); + } + + if (providerConfig.format === "comfyui") { + return handleComfyUIImageGeneration({ model, provider, providerConfig, body, log }); + } + + if (providerConfig.format === "codex-responses") { + return handleCodexImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, + }); + } + + if (providerConfig.format === "haiper-image") { + return handleHaiperImageGeneration({ model, provider, providerConfig, body, credentials, log }); + } + if (providerConfig.format === "leonardo-image") { + return handleLeonardoImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, + }); + } + if (providerConfig.format === "ideogram-image") { + return handleIdeogramImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, + }); + } + + return handleOpenAIImageGeneration({ model, provider, providerConfig, body, credentials, log }); +} + diff --git a/open-sse/handlers/imageGeneration/fal.ts b/open-sse/handlers/imageGeneration/fal.ts new file mode 100644 index 00000000000..1721acd67f0 --- /dev/null +++ b/open-sse/handlers/imageGeneration/fal.ts @@ -0,0 +1,165 @@ + +/** + * Image Generation Handler + * + * Handles POST /v1/images/generations requests. + * Proxies to upstream image generation providers using OpenAI-compatible format. + * + * Request format (OpenAI-compatible): + * { + * "model": "openai/gpt-image-2", + * "prompt": "a beautiful sunset over mountains", + * "n": 1, + * "size": "1024x1024", + * "quality": "standard", // optional: "standard" | "hd" + * "response_format": "url" // optional: "url" | "b64_json" + * } + */ + +import { getImageProvider, parseImageModel } from "../../config/imageRegistry.ts"; + +import { mapImageSize } from "../../translator/image/sizeMapper.ts"; + +import { sanitizeErrorMessage, sanitizeUpstreamDetails } from "../../utils/error.ts"; + +import { extractImageInputs, normalizeRequestedImageFormat, normalizeProviderImagePayload, parseSizeToDimensions } from "./utils.ts"; +import { normalizeRecraftStyle } from "./specialty.ts"; +import { saveImageErrorResult, saveImageSuccessResult } from "./logging.ts"; + + + +const FAL_PRESET_SIZES = { + "1024x1024": "square_hd", + "512x512": "square", + "1792x1024": "landscape_16_9", + "1024x1792": "portrait_16_9", + "1024x768": "landscape_4_3", + "768x1024": "portrait_4_3", + "1536x1024": "landscape_3_2", + "1024x1536": "portrait_3_2", + "576x1024": "portrait_16_9", + "1024x576": "landscape_16_9", +}; + + + +export async function handleFalAIImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, +}) { + const startTime = Date.now(); + const token = credentials.apiKey || credentials.accessToken; + const { imageUrl, imageUrls } = extractImageInputs(body); + const upstreamBody: Record = { + prompt: body.prompt, + sync_mode: body.sync_mode ?? true, + }; + + if (body.n !== undefined) upstreamBody.num_images = Number(body.n) || 1; + if (body.negative_prompt) upstreamBody.negative_prompt = body.negative_prompt; + if (body.seed !== undefined) upstreamBody.seed = body.seed; + if (body.style) upstreamBody.style = normalizeRecraftStyle(body.style); + + const outputFormat = normalizeRequestedImageFormat(body, "png"); + if (outputFormat) upstreamBody.output_format = outputFormat; + + if (model.includes("flux-pro/v1.1") && !model.includes("ultra")) { + upstreamBody.image_size = mapFalImageSize(body.size, "landscape_4_3"); + } else if ( + model.includes("bytedance/") || + model.includes("stable-diffusion") || + model.includes("ideogram") || + model.includes("recraft/v3") + ) { + upstreamBody.image_size = mapFalImageSize(body.size, "square_hd"); + } else { + upstreamBody.aspect_ratio = body.aspect_ratio || mapFalAspectRatio(body.size, "1:1"); + } + + if (body.quality === "hd" && model.includes("ultra")) { + upstreamBody.raw = true; + } + + if (imageUrl && model.includes("flux-pro/v1.1-ultra")) { + upstreamBody.image_url = imageUrl; + } + + if (imageUrls.length > 0 && model.includes("ideogram")) { + upstreamBody.image_urls = imageUrls; + } + + if (log) { + const promptPreview = String(body.prompt ?? "").slice(0, 60); + log.info("IMAGE", `${provider}/${model} (fal-ai) | prompt: "${promptPreview}..."`); + } + + try { + const response = await fetch(`${providerConfig.baseUrl.replace(/\/$/, "")}/${model}`, { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: `Key ${token}`, + }, + body: JSON.stringify(upstreamBody), + }); + + if (!response.ok) { + const errorText = await response.text(); + if (log) + log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); + return saveImageErrorResult({ + provider, + model, + status: response.status, + startTime, + error: errorText, + requestBody: upstreamBody, + }); + } + + const payload = await response.json(); + const images = await normalizeProviderImagePayload(payload, body, log); + return saveImageSuccessResult({ + provider, + model, + startTime, + requestBody: upstreamBody, + responseBody: { images_count: images.length }, + created: payload.created, + images, + }); + } catch (err) { + if (log) log.error("IMAGE", `${provider} fetch error: ${err.message}`); + return saveImageErrorResult({ + provider, + model, + status: 502, + startTime, + error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, + }); + } +} + + + +function mapFalImageSize(size, fallback = "square_hd") { + if (typeof size !== "string") return fallback; + if (FAL_PRESET_SIZES[size]) return FAL_PRESET_SIZES[size]; + if (size.includes("x")) { + const { width, height } = parseSizeToDimensions(size, 1024); + return { width, height }; + } + return fallback; +} + + + +function mapFalAspectRatio(size, fallback = "1:1") { + if (!size) return fallback; + return mapImageSize(size); +} + diff --git a/open-sse/handlers/imageGeneration/gemini.ts b/open-sse/handlers/imageGeneration/gemini.ts new file mode 100644 index 00000000000..52424084b1f --- /dev/null +++ b/open-sse/handlers/imageGeneration/gemini.ts @@ -0,0 +1,207 @@ +import { randomUUID } from "crypto"; + +/** + * Image Generation Handler + * + * Handles POST /v1/images/generations requests. + * Proxies to upstream image generation providers using OpenAI-compatible format. + * + * Request format (OpenAI-compatible): + * { + * "model": "openai/gpt-image-2", + * "prompt": "a beautiful sunset over mountains", + * "n": 1, + * "size": "1024x1024", + * "quality": "standard", // optional: "standard" | "hd" + * "response_format": "url" // optional: "url" | "b64_json" + * } + */ + +import { getImageProvider, parseImageModel } from "../../config/imageRegistry.ts"; + +import { applyAntigravityClientProfileHeaders } from "../../services/antigravityClientProfile.ts"; + +import { getAntigravityEnvelopeUserAgent } from "../../services/antigravityIdentity.ts"; + +import { mapImageSize } from "../../translator/image/sizeMapper.ts"; + +import { saveCallLog } from "@/lib/usageDb"; + +import { sanitizeErrorMessage, sanitizeUpstreamDetails } from "../../utils/error.ts"; + +import { saveImageErrorResult } from "./logging.ts"; +import { normalizeImageAspectRatio, sanitizeImageProviderError } from "./utils.ts"; + + +/** + * Handle Gemini-format image generation (Antigravity / Nano Banana) + * Uses Gemini's generateContent API with responseModalities: ["TEXT", "IMAGE"] + */ +export async function handleGeminiImageGeneration({ model, providerConfig, body, credentials, log }) { + const startTime = Date.now(); + const url = providerConfig.baseUrl; + const provider = "antigravity"; + const credentialRecord = credentials || {}; + const token = credentialRecord.accessToken || credentialRecord.apiKey; + const providerSpecificData = credentialRecord.providerSpecificData; + const providerSpecificProjectId = + providerSpecificData && typeof providerSpecificData === "object" + ? (providerSpecificData as Record).projectId + : null; + const credentialProjectId = + typeof credentialRecord.projectId === "string" ? credentialRecord.projectId.trim() : ""; + const providerProjectId = + typeof providerSpecificProjectId === "string" ? providerSpecificProjectId.trim() : ""; + const projectId = credentialProjectId || providerProjectId || null; + const candidateCount = + typeof body.n === "number" && Number.isFinite(body.n) && body.n > 0 ? Math.floor(body.n) : 1; + const promptText = typeof body.prompt === "string" ? body.prompt : String(body.prompt ?? ""); + + // Summarized request for call log + const logRequestBody = { + model: body.model, + prompt: promptText.slice(0, 200), + size: body.size || "default", + n: candidateCount, + }; + + if (!projectId || typeof projectId !== "string") { + return saveImageErrorResult({ + provider, + model, + status: 400, + startTime, + error: + "Missing Google projectId for Antigravity account. Please reconnect OAuth in Providers so OmniRoute can fetch your Cloud Code project.", + requestBody: logRequestBody, + }); + } + + const antigravityBody = { + project: projectId, + requestId: `image_gen/${Date.now()}/${randomUUID()}/0`, + request: { + contents: [ + { + role: "user", + parts: [{ text: promptText }], + }, + ], + generationConfig: { + candidateCount, + imageConfig: { + aspectRatio: normalizeImageAspectRatio(body.aspect_ratio, body.size), + }, + }, + }, + model, + userAgent: getAntigravityEnvelopeUserAgent(credentialRecord), + requestType: "image_gen", + }; + + const headers = { + "Content-Type": "application/json", + Authorization: `Bearer ${token}`, + }; + applyAntigravityClientProfileHeaders(headers, credentialRecord, antigravityBody); + delete headers["x-goog-user-project"]; + + if (log) { + const promptPreview = promptText.slice(0, 60); + log.info( + "IMAGE", + `antigravity/${model} (gemini) | prompt: "${promptPreview}..." | format: gemini-image` + ); + } + + try { + const response = await fetch(url, { + method: "POST", + headers, + body: JSON.stringify(antigravityBody), + }); + + if (!response.ok) { + const errorText = await response.text(); + const safeError = sanitizeImageProviderError(errorText); + const safeErrorLog = + typeof safeError === "string" ? safeError : JSON.stringify(safeError ?? {}); + if (log) { + log.error("IMAGE", `antigravity error ${response.status}: ${safeErrorLog.slice(0, 200)}`); + } + + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: response.status, + model: `antigravity/${model}`, + provider, + duration: Date.now() - startTime, + error: safeErrorLog.slice(0, 500), + requestBody: logRequestBody, + }).catch(() => {}); + + return { success: false, status: response.status, error: safeError }; + } + + const data = await response.json(); + const responseBody = data.response || data; + + // Extract image data from Antigravity's wrapped Gemini response. + const images = []; + const candidates = responseBody.candidates || []; + for (const candidate of candidates) { + const parts = candidate.content?.parts || []; + for (const part of parts) { + if (part.inlineData) { + images.push({ + b64_json: part.inlineData.data, + revised_prompt: parts.find((p) => p.text)?.text || promptText, + }); + } + } + } + + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 200, + model: `antigravity/${model}`, + provider, + duration: Date.now() - startTime, + tokens: { prompt_tokens: 0, completion_tokens: 0 }, + requestBody: logRequestBody, + responseBody: { images_count: images.length }, + }).catch(() => {}); + + return { + success: true, + data: { + created: Math.floor(Date.now() / 1000), + data: images, + }, + }; + } catch (err) { + if (log) { + log.error("IMAGE", `antigravity fetch error: ${err.message}`); + } + + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 502, + model: `antigravity/${model}`, + provider, + duration: Date.now() - startTime, + error: err.message, + requestBody: logRequestBody, + }).catch(() => {}); + + return { + success: false, + status: 502, + error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, + }; + } +} + diff --git a/open-sse/handlers/imageGeneration/imagen3.ts b/open-sse/handlers/imageGeneration/imagen3.ts new file mode 100644 index 00000000000..28aebeb0157 --- /dev/null +++ b/open-sse/handlers/imageGeneration/imagen3.ts @@ -0,0 +1,162 @@ + +/** + * Image Generation Handler + * + * Handles POST /v1/images/generations requests. + * Proxies to upstream image generation providers using OpenAI-compatible format. + * + * Request format (OpenAI-compatible): + * { + * "model": "openai/gpt-image-2", + * "prompt": "a beautiful sunset over mountains", + * "n": 1, + * "size": "1024x1024", + * "quality": "standard", // optional: "standard" | "hd" + * "response_format": "url" // optional: "url" | "b64_json" + * } + */ + +import { getImageProvider, parseImageModel } from "../../config/imageRegistry.ts"; + +import { mapImageSize } from "../../translator/image/sizeMapper.ts"; + +import { saveCallLog } from "@/lib/usageDb"; + +import { sanitizeErrorMessage, sanitizeUpstreamDetails } from "../../utils/error.ts"; + + + + +type Imagen3ImageGenArgs = { + model: string; + provider: string; + providerConfig: { baseUrl: string }; + body: { prompt?: string; size?: string; n?: number }; + credentials: { apiKey?: string; accessToken?: string }; + log?: { + info?: (tag: string, msg: string) => void; + error?: (tag: string, msg: string) => void; + } | null; +}; + + + +type Imagen3NormalizedImage = { + b64_json?: unknown; + url?: unknown; + revised_prompt?: string; +}; + + + +/** + * Handle Imagen 3 image generation + */ +export async function handleImagen3ImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, +}: Imagen3ImageGenArgs) { + const startTime = Date.now(); + const token = credentials.apiKey || credentials.accessToken; + const aspectRatio = mapImageSize(body.size); + + const upstreamBody = { + prompt: body.prompt, + aspect_ratio: aspectRatio, + number_of_images: body.n ?? 1, + }; + + if (log) { + const promptPreview = String(body.prompt ?? "").slice(0, 60); + log.info( + "IMAGE", + `${provider}/${model} (imagen3) | prompt: "${promptPreview}..." | aspect_ratio: ${aspectRatio}` + ); + } + + try { + const response = await fetch(providerConfig.baseUrl, { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${token}`, + }, + body: JSON.stringify(upstreamBody), + }); + + if (!response.ok) { + const errorText = await response.text(); + if (log) + log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); + + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: response.status, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: errorText.slice(0, 500), + requestBody: upstreamBody, + }).catch(() => {}); + + return { success: false, status: response.status, error: errorText }; + } + + const data = await response.json(); + + // Normalize response to OpenAI format + const images: Imagen3NormalizedImage[] = []; + if (Array.isArray(data.images)) { + images.push( + ...data.images.map((img: Record) => ({ + b64_json: img.image ?? img.b64_json ?? img.url ?? img, + revised_prompt: body.prompt, + })) + ); + } else if (Array.isArray(data.data)) { + images.push(...data.data); + } else if (data.url || data.b64_json || data.image) { + images.push({ + b64_json: data.image || data.b64_json || data.url, + url: data.url, + revised_prompt: body.prompt, + }); + } + + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 200, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + responseBody: { images_count: images.length }, + }).catch(() => {}); + + return { + success: true, + data: { created: data.created || Math.floor(Date.now() / 1000), data: images }, + }; + } catch (err: unknown) { + const errMsg = err instanceof Error ? err.message : String(err); + if (log) log.error("IMAGE", `${provider} fetch error: ${errMsg}`); + + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 502, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: errMsg, + }).catch(() => {}); + + return { success: false, status: 502, error: `Image provider error: ${errMsg}` }; + } +} + diff --git a/open-sse/handlers/imageGeneration/index.ts b/open-sse/handlers/imageGeneration/index.ts new file mode 100644 index 00000000000..f191ceeb7df --- /dev/null +++ b/open-sse/handlers/imageGeneration/index.ts @@ -0,0 +1,14 @@ +// Re-export everything to maintain backward compatibility +export * from "./utils.ts"; +export * from "./logging.ts"; +export * from "./chatgptWeb.ts"; +export * from "./blackForestLabs.ts"; +export * from "./stability.ts"; +export * from "./fal.ts"; +export * from "./gemini.ts"; +export * from "./openai.ts"; +export * from "./kie.ts"; +export * from "./imagen3.ts"; +export * from "./codex.ts"; +export * from "./specialty.ts"; +export * from "./core.ts"; diff --git a/open-sse/handlers/imageGeneration/kie.ts b/open-sse/handlers/imageGeneration/kie.ts new file mode 100644 index 00000000000..878e9d7933b --- /dev/null +++ b/open-sse/handlers/imageGeneration/kie.ts @@ -0,0 +1,263 @@ + +/** + * Image Generation Handler + * + * Handles POST /v1/images/generations requests. + * Proxies to upstream image generation providers using OpenAI-compatible format. + * + * Request format (OpenAI-compatible): + * { + * "model": "openai/gpt-image-2", + * "prompt": "a beautiful sunset over mountains", + * "n": 1, + * "size": "1024x1024", + * "quality": "standard", // optional: "standard" | "hd" + * "response_format": "url" // optional: "url" | "b64_json" + * } + */ + +import { getImageProvider, parseImageModel } from "../../config/imageRegistry.ts"; + +import { kieExecutor } from "../../executors/kie.ts"; + +import { mapImageSize } from "../../translator/image/sizeMapper.ts"; + +import { + getKieErrorMessage, + getKieErrorStatus, + isJsonObject, + parseKieResultJson, +} from "../../utils/kieTask.ts"; + +import { sanitizeErrorMessage, sanitizeUpstreamDetails } from "../../utils/error.ts"; + +import { normalizePositiveNumber, extractImageInputs } from "./utils.ts"; +import { saveImageErrorResult, saveImageSuccessResult } from "./logging.ts"; + + + +interface KieImageOptions { + model: string; + provider: string; + providerConfig: { + baseUrl: string; + statusUrl?: string; + }; + body: Record & { + prompt?: unknown; + size?: unknown; + n?: unknown; + timeout_ms?: unknown; + poll_interval_ms?: unknown; + }; + credentials?: { + apiKey?: string; + accessToken?: string; + } | null; + log?: { + info: (scope: string, message: string) => void; + error: (scope: string, message: string) => void; + } | null; +} + + + +function normalizeKieImageResult(recordData: unknown): string[] { + const record = isJsonObject(recordData) ? recordData : {}; + const data = isJsonObject(record.data) ? record.data : {}; + const response = isJsonObject(data.response) ? data.response : {}; + const resultJson = parseKieResultJson(recordData); + const urls = new Set(); + + const add = (val: unknown) => { + if (typeof val === "string" && val.startsWith("http")) urls.add(val); + if (Array.isArray(val)) { + val.forEach((v) => { + if (typeof v === "string" && v.startsWith("http")) urls.add(v); + }); + } + }; + + // Check resultJson (common in Market API) + add(resultJson?.resultUrls); + add(resultJson?.imageUrls); + add(resultJson?.resultUrl); + add(resultJson?.imageUrl); + + // Check data.response (common in 4o-image API) + add(response.resultUrls); + add(response.resultUrl); + + // Check direct data fields + add(data.resultImageUrls); + add(data.resultImageUrl); + add(data.url); + + return Array.from(urls); +} + + + +export async function handleKieImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, +}: KieImageOptions) { + const startTime = Date.now(); + const token = credentials?.apiKey || credentials?.accessToken; + const timeoutMs = normalizePositiveNumber(body.timeout_ms, 300000); + const pollIntervalMs = normalizePositiveNumber(body.poll_interval_ms, 2500); + const prompt = typeof body.prompt === "string" ? body.prompt : String(body.prompt ?? ""); + const size = typeof body.size === "string" ? body.size : undefined; + + if (!token) { + return saveImageErrorResult({ + provider, + model, + status: 401, + startTime, + error: "KIE API key is required", + }); + } + + // Check if model is a Market model (unified API) + const fullRegistry = getImageProvider(provider); + const modelEntry = fullRegistry?.models?.find((m) => m.id === model); + const isMarket = modelEntry?.isMarket || model.includes("/"); + + const { imageUrl } = extractImageInputs(body); + let baseUrl = ""; + let payload: Record = {}; + + if (isMarket) { + // Unified Market API endpoint + baseUrl = `${providerConfig.baseUrl.replace(/\/$/, "")}/api/v1/jobs/createTask`; + const input: Record = { + prompt, + aspect_ratio: mapImageSize(size, "1:1"), + }; + if (imageUrl) { + input.image_url = imageUrl; + } + payload = { + model, + input, + }; + } else { + // Legacy/Direct endpoint + const modelPath = model.replace("-t2i", "").replace("-i2i", ""); + baseUrl = providerConfig.baseUrl.includes(model) + ? providerConfig.baseUrl + : `https://api.kie.ai/api/v1/${modelPath}/generate`; + + payload = { + prompt, + size: mapImageSize(size, "1:1"), + nVariants: body.n || 1, + }; + } + + if (log) { + const promptPreview = String(body.prompt ?? "").slice(0, 60); + log.info( + "IMAGE", + `${provider}/${model} (${isMarket ? "market" : "direct"}) | prompt: "${promptPreview}..."` + ); + } + + try { + const endpoint = isMarket ? "/api/v1/jobs/createTask" : new URL(baseUrl).pathname; + const createBaseUrl = isMarket ? providerConfig.baseUrl : baseUrl.replace(endpoint, ""); + const createData = await kieExecutor.createTask({ + baseUrl: createBaseUrl, + token, + payload, + endpoint, + }); + const taskId = createData?.data?.taskId || createData?.taskId; + + if (!taskId) { + const errorMessage = + createData?.msg || + createData?.message || + createData?.error || + "KIE image generation did not return taskId"; + if (log) { + log.error("IMAGE", `KIE createTask failed: ${JSON.stringify(createData)}`); + } + return saveImageErrorResult({ + provider, + model, + status: 502, + startTime, + error: errorMessage, + requestBody: payload, + }); + } + + // Use statusUrl from providerConfig if available, fallback to dynamic derivation + const statusUrl = isMarket + ? `${providerConfig.baseUrl.replace(/\/$/, "")}/api/v1/jobs/recordInfo` + : providerConfig.statusUrl && !providerConfig.statusUrl.includes("jobs/recordInfo") + ? providerConfig.statusUrl + : baseUrl.replace(/\/generate$/, "/record-info"); + + const { data: recordData, state } = await kieExecutor.pollTask({ + statusUrl, + taskId: String(taskId), + token, + timeoutMs, + pollIntervalMs, + }); + + if (state === "success") { + if (log) { + log.info("IMAGE", `KIE poll success for task ${taskId}`); + } + const urls = normalizeKieImageResult(recordData); + const images = urls.map((url: string) => ({ url, revised_prompt: prompt })); + + return saveImageSuccessResult({ + provider, + model, + startTime, + requestBody: payload, + responseBody: { images_count: images.length }, + images, + }); + } + + const record = isJsonObject(recordData) ? recordData : {}; + const recordDataBody = isJsonObject(record.data) ? record.data : {}; + const errorMessage = + recordDataBody.errorMessage || + recordDataBody.failMsg || + record.msg || + "KIE image task failed"; + + if (log) { + log.error("IMAGE", `KIE poll failed for task ${taskId}: ${JSON.stringify(recordData)}`); + } + + return saveImageErrorResult({ + provider, + model, + status: 502, + startTime, + error: String(errorMessage), + requestBody: payload, + }); + } catch (err: unknown) { + return saveImageErrorResult({ + provider, + model, + status: getKieErrorStatus(err, 502), + startTime, + error: `Image provider error: ${getKieErrorMessage(err, "KIE image generation failed")}`, + }); + } +} + diff --git a/open-sse/handlers/imageGeneration/logging.ts b/open-sse/handlers/imageGeneration/logging.ts new file mode 100644 index 00000000000..2310dbcb69b --- /dev/null +++ b/open-sse/handlers/imageGeneration/logging.ts @@ -0,0 +1,77 @@ + +/** + * Image Generation Handler + * + * Handles POST /v1/images/generations requests. + * Proxies to upstream image generation providers using OpenAI-compatible format. + * + * Request format (OpenAI-compatible): + * { + * "model": "openai/gpt-image-2", + * "prompt": "a beautiful sunset over mountains", + * "n": 1, + * "size": "1024x1024", + * "quality": "standard", // optional: "standard" | "hd" + * "response_format": "url" // optional: "url" | "b64_json" + * } + */ + +import { getImageProvider, parseImageModel } from "../../config/imageRegistry.ts"; + +import { saveCallLog } from "@/lib/usageDb"; + +import { sanitizeErrorMessage, sanitizeUpstreamDetails } from "../../utils/error.ts"; + + + + +export function saveImageSuccessResult({ + provider, + model, + startTime, + requestBody = null, + responseBody = null, + created = null, + images, +}) { + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 200, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + requestBody, + responseBody, + }).catch(() => {}); + + return { + success: true, + data: { + created: created || Math.floor(Date.now() / 1000), + data: images, + }, + }; +} + + + +export function saveImageErrorResult({ provider, model, status, startTime, error, requestBody = null }) { + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: typeof error === "string" ? error.slice(0, 500) : String(error).slice(0, 500), + requestBody, + }).catch(() => {}); + + return { + success: false, + status, + error, + }; +} + diff --git a/open-sse/handlers/imageGeneration/openai.ts b/open-sse/handlers/imageGeneration/openai.ts new file mode 100644 index 00000000000..bffe38e35eb --- /dev/null +++ b/open-sse/handlers/imageGeneration/openai.ts @@ -0,0 +1,490 @@ +import { randomUUID } from "crypto"; + +/** + * Image Generation Handler + * + * Handles POST /v1/images/generations requests. + * Proxies to upstream image generation providers using OpenAI-compatible format. + * + * Request format (OpenAI-compatible): + * { + * "model": "openai/gpt-image-2", + * "prompt": "a beautiful sunset over mountains", + * "n": 1, + * "size": "1024x1024", + * "quality": "standard", // optional: "standard" | "hd" + * "response_format": "url" // optional: "url" | "b64_json" + * } + */ + +import { getImageProvider, parseImageModel } from "../../config/imageRegistry.ts"; + +import { mapImageSize } from "../../translator/image/sizeMapper.ts"; + +import { ChatGptWebExecutor } from "../../executors/chatgpt-web.ts"; + +import { getChatGptImage, findChatGptImageBySha256 } from "../../services/chatgptImageCache.ts"; + +import { createHash } from "node:crypto"; + +import { saveCallLog } from "@/lib/usageDb"; + +import { sanitizeErrorMessage, sanitizeUpstreamDetails } from "../../utils/error.ts"; + +import { extractImageInputs, fetchImageEndpoint, resolveImageBaseUrl } from "./utils.ts"; +import { saveImageErrorResult, saveImageSuccessResult } from "./logging.ts"; +import { extractMarkdownImageUrls, CHATGPT_WEB_IMAGE_ID_RE } from "./chatgptWeb.ts"; + + + +const OPENAI_IMAGE_TO_IMAGE_MODELS = new Set([ + "black-forest-labs/FLUX.2-max", + "black-forest-labs/FLUX.2-pro", + "black-forest-labs/FLUX.2-flex", + "black-forest-labs/FLUX.2-dev", + "openai/gpt-image-1.5", + "Wan-AI/Wan2.6-image", + "Qwen/Qwen-Image-2.0-Pro", + "Qwen/Qwen-Image-2.0", + "google/flash-image-3.1", + "google/gemini-3-pro-image", + "flux-kontext-max", + "flux-kontext", + "flux-kontext-pro", + "qwen-image", +]); + + + +/** + * Handle OpenAI-compatible image generation (standard providers + Nebius fallback) + */ +export async function handleOpenAIImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, +}) { + const startTime = Date.now(); + + // Summarized request for call log + const logRequestBody = { + model: body.model, + prompt: + typeof body.prompt === "string" + ? body.prompt.slice(0, 200) + : String(body.prompt ?? "").slice(0, 200), + size: body.size || "default", + n: body.n || 1, + quality: body.quality || undefined, + }; + + // Build upstream request (OpenAI-compatible format) + const upstreamBody: Record = { + model: model, + prompt: body.prompt, + }; + + // Pass optional parameters + if (body.n !== undefined) upstreamBody.n = body.n; + if (body.size !== undefined) upstreamBody.size = body.size; + if (body.quality !== undefined) upstreamBody.quality = body.quality; + if (body.response_format !== undefined) upstreamBody.response_format = body.response_format; + if (body.style !== undefined) upstreamBody.style = body.style; + + const { imageUrl } = extractImageInputs(body); + if (imageUrl && OPENAI_IMAGE_TO_IMAGE_MODELS.has(model)) { + upstreamBody.image_url = imageUrl; + } + + // Build headers + const headers = { + "Content-Type": "application/json", + }; + + const token = credentials.apiKey || credentials.accessToken; + if (providerConfig.authHeader === "bearer") { + headers["Authorization"] = `Bearer ${token}`; + } else if (providerConfig.authHeader === "x-api-key") { + headers["x-api-key"] = token; + } + + if (log) { + const promptPreview = + typeof body.prompt === "string" + ? body.prompt.slice(0, 60) + : String(body.prompt ?? "").slice(0, 60); + log.info( + "IMAGE", + `${provider}/${model} | prompt: "${promptPreview}..." | size: ${body.size || "default"}` + ); + } + + const requestBody = JSON.stringify(upstreamBody); + + // Try primary URL + let result = await fetchImageEndpoint( + providerConfig.baseUrl, + headers, + requestBody, + provider, + log + ); + + // Fallback for providers with fallbackUrl (e.g., Nebius) + if ( + !result.success && + providerConfig.fallbackUrl && + [404, 410, 502, 503].includes(result.status) + ) { + if (log) { + log.info("IMAGE", `${provider}: primary URL failed (${result.status}), trying fallback...`); + } + result = await fetchImageEndpoint( + providerConfig.fallbackUrl, + headers, + requestBody, + provider, + log + ); + } + + // Save call log after result is determined + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: result.status || (result.success ? 200 : 502), + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + tokens: { prompt_tokens: 0, completion_tokens: 0 }, + error: result.success + ? null + : typeof result.error === "string" + ? result.error.slice(0, 500) + : null, + requestBody: logRequestBody, + responseBody: result.success ? { images_count: result.data?.data?.length || 0 } : null, + }).catch(() => {}); + + return result; +} + + + +/** + * OpenAI-compatible image *edit* forwarder for custom providers (#3214 / #3215). + * + * Mirrors `handleOpenAIImageGeneration` but posts multipart/form-data to the node's + * `/images/edits` endpoint and returns the upstream OpenAI-compatible response. Kept + * separate from the chatgpt-web edit flow, which continues a saved conversation node + * rather than forwarding a stateless edit. The fetch helper leaves Content-Type unset so + * `fetch` derives the multipart boundary from the FormData body. + */ +export async function handleOpenAIImageEdit({ + model, + provider, + credentials, + prompt, + imageBytes, + imageMime, + size, + responseFormat, + n = 1, + log, +}: { + model: string; + provider: string; + credentials: + | { + apiKey?: string; + accessToken?: string; + baseUrl?: unknown; + providerSpecificData?: { baseUrl?: unknown } | null; + } + | null + | undefined; + prompt: string; + imageBytes: Buffer; + imageMime?: string | null; + size?: string | null; + responseFormat?: string | null; + n?: number; + log?: { info: (tag: string, message: string) => void } | null; +}) { + const startTime = Date.now(); + const url = resolveImageBaseUrl( + credentials, + `https://generativelanguage.googleapis.com/v1beta/openai/images/edits`, + "edits" + ); + + // Build the multipart body as a Buffer with an explicit boundary instead of a global + // `FormData`. In production `globalThis.fetch` is patched with node_modules/undici's fetch, + // whose `FormData` class differs from `globalThis.FormData` — passing a native FormData + // makes undici serialize it as the string "[object FormData]" (text/plain), dropping every + // field (including `model`, which reaches the upstream empty). A Buffer body is accepted + // verbatim by any fetch implementation. (#3273) + const boundary = `----OmniRouteImageEdit${randomUUID().replace(/-/g, "")}`; + const CRLF = "\r\n"; + const partBuffers: Buffer[] = []; + const appendField = (name: string, value: string) => { + partBuffers.push( + Buffer.from( + `--${boundary}${CRLF}Content-Disposition: form-data; name="${name}"${CRLF}${CRLF}${value}${CRLF}` + ) + ); + }; + appendField("model", model); + appendField("prompt", prompt); + if (size) appendField("size", size); + if (responseFormat) appendField("response_format", responseFormat); + appendField("n", String(n || 1)); + partBuffers.push( + Buffer.from( + `--${boundary}${CRLF}Content-Disposition: form-data; name="image"; filename="image.png"${CRLF}` + + `Content-Type: ${imageMime || "image/png"}${CRLF}${CRLF}` + ) + ); + partBuffers.push(imageBytes); + partBuffers.push(Buffer.from(`${CRLF}--${boundary}--${CRLF}`)); + const multipartBody = Buffer.concat(partBuffers); + + const headers: Record = { + "Content-Type": `multipart/form-data; boundary=${boundary}`, + }; + const token = credentials?.apiKey || credentials?.accessToken; + if (token) headers["Authorization"] = `Bearer ${token}`; + + if (log) { + log.info("IMAGE", `${provider}/${model} (edit) | prompt: "${prompt.slice(0, 60)}..." -> ${url}`); + } + + const result = await fetchImageEndpoint( + url, + headers, + multipartBody as unknown as BodyInit, + provider, + log + ); + + saveCallLog({ + method: "POST", + path: "/v1/images/edits", + status: result.status || (result.success ? 200 : 502), + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + tokens: { prompt_tokens: 0, completion_tokens: 0 }, + error: result.success + ? null + : typeof result.error === "string" + ? result.error.slice(0, 500) + : null, + requestBody: { model, prompt: prompt.slice(0, 200), size: size || "default", n: n || 1 }, + responseBody: result.success ? { images_count: result.data?.data?.length || 0 } : null, + }).catch(() => {}); + + return result; +} + + + +/** + * Handle a multipart /v1/images/edits request for chatgpt-web. Open WebUI + * uploads the prior image's bytes; we hash them and look up our cache. + * + * The hash match is reliable because Open WebUI's image-gen pipeline + * downloads our /v1/chatgpt-web/image/ URL byte-for-byte and re-serves + * those exact bytes through its own file store. When the user asks to edit + * the image, OWUI uploads the same bytes back to us via multipart — same + * hash, we find the conversation context, and drive the executor with a + * synthetic chat thread that triggers continuation mode. + * + * No-match cases (cache evicted by TTL, or the user uploaded a foreign + * image) get a clear 400. We can't actually edit an image we don't have a + * conversation context for — chatgpt.com's image_gen tool needs the + * original conversation node, and we don't have a path to upload bytes + * directly. + */ +export async function handleImageEdit({ + provider, + model, + body, + imageBytes, + credentials, + log, + signal = null, + clientHeaders = null, +}: { + provider: string; + model: string; + body: Record; + imageBytes: Buffer; + imageMime?: string; // accepted for symmetry with route layer; not used + credentials: any; + log: any; + signal?: AbortSignal | null; + clientHeaders?: Record | null; +}) { + const startTime = Date.now(); + const prompt = typeof body.prompt === "string" ? body.prompt.trim() : ""; + if (!prompt) { + return saveImageErrorResult({ + provider, + model, + status: 400, + startTime, + error: "Prompt is required for image edit", + }); + } + + if (!credentials?.apiKey) { + return saveImageErrorResult({ + provider, + model, + status: 401, + startTime, + error: "ChatGPT Web credentials missing session cookie", + }); + } + + const imageHash = createHash("sha256").update(imageBytes).digest("hex"); + const cached = findChatGptImageBySha256(imageHash); + + const wantsBase64 = body.response_format === "b64_json"; + const requestBody = { + model, + prompt: prompt.slice(0, 500), + size: body.size || undefined, + image_hash: imageHash.slice(0, 16), + image_bytes: imageBytes.length, + cached_match: Boolean(cached?.entry.context), + }; + + if (!cached?.entry.context) { + // chatgpt-web's image_gen tool can only edit an image when we continue + // the original conversation node. If we never generated this image (or + // its 30-minute TTL elapsed), there's no node to continue. Return a + // clear, actionable error — much better than silently spawning an + // unrelated image and confusing the user. + log?.warn?.( + "IMAGE", + `chatgpt-web edit: no cached match for sha256=${imageHash.slice(0, 16)} (bytes=${imageBytes.length}); returning 400` + ); + return saveImageErrorResult({ + provider, + model, + status: 400, + startTime, + error: + "chatgpt-web image edit only works for images recently generated through this OmniRoute instance " + + "(cache window: 30 minutes). Re-generate the image and try the edit immediately, or disable image-edit " + + "in your client to use plain chat-completion edit prompts instead.", + requestBody, + }); + } + + // Build a synthetic chat thread that surfaces the cached image URL on + // the assistant turn. The executor's parseOpenAIMessages picks up the + // URL, findCachedImageContext resolves it to {conversationId, + // parentMessageId}, and looksLikeImageEditRequest fires on the user + // prompt — together producing a continuation request that actually + // edits the saved image. + // + // The synthetic user prompt is anchored with both an edit verb AND an + // image-gen verb so the executor's heuristics fire regardless of what + // wording the caller used ("now make it brighter", "tweak this", ...): + // - looksLikeImageEditRequest: matches "edit" + "image" within 120 chars + // - looksLikeImageGenRequest: matches "generate" + "image" within 40 chars + // Either match alone would set forImageGen, but covering both is cheap + // insurance for prompts that don't fit common phrasings. + const messages: Array<{ role: string; content: string }> = [ + { + role: "assistant", + // The base URL is irrelevant — only the path is parsed by + // CACHED_IMAGE_URL_RE in the executor's findCachedImageContext. + content: `![image](http://internal/v1/chatgpt-web/image/${cached.id})`, + }, + { + role: "user", + content: `Edit the image and generate the new image: ${prompt}`, + }, + ]; + + const executor = new ChatGptWebExecutor(); + const result = await executor.execute({ + model, + body: { messages }, + stream: false, + credentials, + signal, + log, + clientHeaders, + }); + + const responseText = await result.response.text(); + if (result.response.status >= 400) { + return saveImageErrorResult({ + provider, + model, + status: result.response.status, + startTime, + error: responseText, + requestBody, + }); + } + + let content = ""; + try { + const json = JSON.parse(responseText); + content = String(json?.choices?.[0]?.message?.content || ""); + } catch { + content = responseText; + } + + const urls = extractMarkdownImageUrls(content); + if (urls.length === 0) { + return saveImageErrorResult({ + provider, + model, + status: 502, + startTime, + error: `ChatGPT Web edit completed without returning image markdown: ${content.slice(0, 300)}`, + requestBody, + }); + } + + const images: Array<{ url?: string; b64_json?: string }> = []; + for (const url of urls) { + if (!wantsBase64) { + images.push({ url }); + continue; + } + const id = url.match(CHATGPT_WEB_IMAGE_ID_RE)?.[1]; + const cachedNew = id ? getChatGptImage(id) : null; + if (!cachedNew) { + return saveImageErrorResult({ + provider, + model, + status: 502, + startTime, + error: "ChatGPT Web image bytes expired before b64_json conversion", + requestBody, + }); + } + images.push({ b64_json: cachedNew.bytes.toString("base64") }); + } + + return saveImageSuccessResult({ + provider, + model, + startTime, + requestBody, + responseBody: { images_count: images.length, edit_match: Boolean(cached?.entry.context) }, + images, + }); +} + diff --git a/open-sse/handlers/imageGeneration/specialty.ts b/open-sse/handlers/imageGeneration/specialty.ts new file mode 100644 index 00000000000..5fbb8a3b9fd --- /dev/null +++ b/open-sse/handlers/imageGeneration/specialty.ts @@ -0,0 +1,1207 @@ +import { randomUUID } from "crypto"; + +/** + * Image Generation Handler + * + * Handles POST /v1/images/generations requests. + * Proxies to upstream image generation providers using OpenAI-compatible format. + * + * Request format (OpenAI-compatible): + * { + * "model": "openai/gpt-image-2", + * "prompt": "a beautiful sunset over mountains", + * "n": 1, + * "size": "1024x1024", + * "quality": "standard", // optional: "standard" | "hd" + * "response_format": "url" // optional: "url" | "b64_json" + * } + */ + +import { getImageProvider, parseImageModel } from "../../config/imageRegistry.ts"; + +import { mapImageSize } from "../../translator/image/sizeMapper.ts"; + +import { saveCallLog } from "@/lib/usageDb"; + +import { sleep } from "../../utils/sleep.ts"; + +import { + submitComfyWorkflow, + pollComfyResult, + fetchComfyOutput, + extractComfyOutputFiles, +} from "../../utils/comfyuiClient.ts"; + +import { fetchRemoteImage } from "@/shared/network/remoteImageFetch"; + +import { sanitizeErrorMessage, sanitizeUpstreamDetails } from "../../utils/error.ts"; + +import { saveImageErrorResult, saveImageSuccessResult } from "./logging.ts"; +import { normalizeProviderImagePayload, extractImageInputs, resolveImageSource, parseSizeToDimensions, inferResolutionFromSize, normalizePositiveNumber } from "./utils.ts"; + + + +export async function handleRecraftImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, +}) { + const startTime = Date.now(); + const token = credentials.apiKey || credentials.accessToken; + const upstreamBody: Record = { + model, + prompt: body.prompt, + }; + + if (body.n !== undefined) upstreamBody.n = body.n; + if (body.size !== undefined) upstreamBody.size = body.size; + if (body.response_format !== undefined) upstreamBody.response_format = body.response_format; + if (body.style !== undefined) upstreamBody.style = body.style; + + if (log) { + const promptPreview = String(body.prompt ?? "").slice(0, 60); + log.info("IMAGE", `${provider}/${model} (recraft) | prompt: "${promptPreview}..."`); + } + + try { + const response = await fetch( + `${providerConfig.baseUrl.replace(/\/$/, "")}/v1/images/generations`, + { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${token}`, + }, + body: JSON.stringify(upstreamBody), + } + ); + + if (!response.ok) { + const errorText = await response.text(); + if (log) + log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); + return saveImageErrorResult({ + provider, + model, + status: response.status, + startTime, + error: errorText, + requestBody: upstreamBody, + }); + } + + const payload = await response.json(); + const images = await normalizeProviderImagePayload(payload, body, log); + return saveImageSuccessResult({ + provider, + model, + startTime, + requestBody: upstreamBody, + responseBody: { images_count: images.length }, + created: payload.created, + images, + }); + } catch (err) { + if (log) log.error("IMAGE", `${provider} fetch error: ${err.message}`); + return saveImageErrorResult({ + provider, + model, + status: 502, + startTime, + error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, + }); + } +} + + + +export async function handleTopazImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, +}) { + const startTime = Date.now(); + const token = credentials.apiKey || credentials.accessToken; + const { imageUrl } = extractImageInputs(body); + + if (!imageUrl) { + return { + success: false, + status: 400, + error: `Topaz model ${model} requires an input image`, + }; + } + + try { + const imageSource = await resolveImageSource(imageUrl); + const formData = new FormData(); + const blob = new Blob([imageSource.buffer], { type: imageSource.contentType || "image/png" }); + formData.append("image", blob, "image.png"); + + if (typeof body.size === "string" && body.size.includes("x")) { + const { width, height } = parseSizeToDimensions(body.size, 1024); + formData.append("output_width", String(width)); + formData.append("output_height", String(height)); + } + + if (log) { + const promptPreview = String(body.prompt ?? "enhance image").slice(0, 60); + log.info("IMAGE", `${provider}/${model} (topaz) | prompt: "${promptPreview}..."`); + } + + const response = await fetch(`${providerConfig.baseUrl.replace(/\/$/, "")}/image/v1/enhance`, { + method: "POST", + headers: { + Accept: "image/jpeg", + "X-API-Key": token, + }, + body: formData, + }); + + if (!response.ok) { + const errorText = await response.text(); + if (log) + log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); + return saveImageErrorResult({ + provider, + model, + status: response.status, + startTime, + error: errorText, + }); + } + + const contentType = response.headers.get("content-type") || "image/jpeg"; + const buffer = Buffer.from(await response.arrayBuffer()); + const base64 = buffer.toString("base64"); + const wantsBase64 = body.response_format === "b64_json"; + const images = [ + wantsBase64 + ? { b64_json: base64, revised_prompt: body.prompt } + : { url: `data:${contentType};base64,${base64}`, revised_prompt: body.prompt }, + ]; + + return saveImageSuccessResult({ + provider, + model, + startTime, + responseBody: { images_count: images.length }, + images, + }); + } catch (err) { + if (log) log.error("IMAGE", `${provider} fetch error: ${err.message}`); + return saveImageErrorResult({ + provider, + model, + status: 502, + startTime, + error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, + }); + } +} + + + +export function normalizeRecraftStyle(style) { + if (style === "vivid") return "digital_illustration"; + if (style === "natural") return "realistic_image"; + return style; +} + + + +/** + * Handle Hyperbolic image generation + * Uses { model_name, prompt, height, width } and returns { images: [{ image: base64 }] } + */ +export async function handleHyperbolicImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, +}) { + const startTime = Date.now(); + const token = credentials.apiKey || credentials.accessToken; + + const [width, height] = (body.size || "1024x1024").split("x").map(Number); + + const upstreamBody = { + model_name: model, + prompt: body.prompt, + height: height || 1024, + width: width || 1024, + backend: "auto", + }; + + if (log) { + const promptPreview = String(body.prompt ?? "").slice(0, 60); + log.info("IMAGE", `${provider}/${model} (hyperbolic) | prompt: "${promptPreview}..."`); + } + + try { + const response = await fetch(providerConfig.baseUrl, { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${token}`, + }, + body: JSON.stringify(upstreamBody), + }); + + if (!response.ok) { + const errorText = await response.text(); + if (log) + log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); + + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: response.status, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: errorText.slice(0, 500), + }).catch(() => {}); + + return { success: false, status: response.status, error: errorText }; + } + + const data = await response.json(); + // Transform { images: [{ image: base64 }] } → OpenAI format + const images = (data.images || []).map((img) => ({ + b64_json: img.image, + revised_prompt: body.prompt, + })); + + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 200, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + responseBody: { images_count: images.length }, + }).catch(() => {}); + + return { + success: true, + data: { created: Math.floor(Date.now() / 1000), data: images }, + }; + } catch (err) { + if (log) log.error("IMAGE", `${provider} fetch error: ${err.message}`); + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 502, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: err.message, + }).catch(() => {}); + return { + success: false, + status: 502, + error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, + }; + } +} + + + +/** + * Handle NanoBanana image generation + * NanoBanana is async (submit task -> poll status -> return final image URL/base64) + */ +export async function handleNanoBananaImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, +}) { + const startTime = Date.now(); + const token = credentials.apiKey || credentials.accessToken; + + // Route to pro URL for "nanobanana-pro" model + const isPro = model === "nanobanana-pro"; + const submitUrl = isPro && providerConfig.proUrl ? providerConfig.proUrl : providerConfig.baseUrl; + const statusUrl = providerConfig.statusUrl; + + const aspectRatio = + typeof body.aspectRatio === "string" + ? body.aspectRatio + : typeof body.aspect_ratio === "string" + ? body.aspect_ratio + : mapImageSize(body.size); + + let resolution = + typeof body.resolution === "string" + ? body.resolution + : inferResolutionFromSize(body.size) || "1K"; + if (body.quality === "hd" && resolution === "1K") { + resolution = "2K"; + } + + const upstreamBody = isPro + ? { + prompt: body.prompt, + resolution, + aspectRatio, + ...(Array.isArray(body.imageUrls) ? { imageUrls: body.imageUrls } : {}), + } + : { + prompt: body.prompt, + type: + Array.isArray(body.imageUrls) && body.imageUrls.length > 0 + ? "IMAGETOIAMGE" + : "TEXTTOIAMGE", + numImages: Number.isFinite(body.n) ? Math.max(1, Number(body.n)) : 1, + image_size: aspectRatio, + ...(Array.isArray(body.imageUrls) ? { imageUrls: body.imageUrls } : {}), + }; + + if (log) { + const promptPreview = String(body.prompt ?? "").slice(0, 60); + log.info( + "IMAGE", + `${provider}/${model} (nanobanana ${isPro ? "pro" : "flash"}) | prompt: "${promptPreview}..."` + ); + } + + try { + const submitResp = await fetch(submitUrl, { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${token}`, + }, + body: JSON.stringify(upstreamBody), + }); + + if (!submitResp.ok) { + const errorText = await submitResp.text(); + if (log) { + log.error( + "IMAGE", + `${provider} submit error ${submitResp.status}: ${errorText.slice(0, 200)}` + ); + } + + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: submitResp.status, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: errorText.slice(0, 500), + }).catch(() => {}); + + return { success: false, status: submitResp.status, error: errorText }; + } + + const submitData = await submitResp.json(); + + // Backward compatibility: handle providers returning image payload synchronously + const hasSyncPayload = + Boolean(submitData?.image) || + Array.isArray(submitData?.images) || + Array.isArray(submitData?.data) || + Boolean(submitData?.data?.[0]?.url) || + Boolean(submitData?.data?.[0]?.b64_json); + + if (hasSyncPayload) { + const syncResult = normalizeNanoBananaSyncPayload(submitData, body.prompt); + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 200, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + responseBody: { images_count: syncResult.data?.length || 0, mode: "sync" }, + }).catch(() => {}); + return { + success: true, + data: { created: Math.floor(Date.now() / 1000), data: syncResult.data }, + }; + } + + const taskId = submitData?.data?.taskId || submitData?.taskId; + if (!taskId) { + const errorText = `NanoBanana submit did not return taskId: ${JSON.stringify(submitData).slice(0, 400)}`; + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 502, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: errorText, + }).catch(() => {}); + return { success: false, status: 502, error: errorText }; + } + + if (!statusUrl) { + const errorText = "NanoBanana statusUrl is not configured"; + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 500, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: errorText, + }).catch(() => {}); + return { success: false, status: 500, error: errorText }; + } + + const timeoutMs = normalizePositiveNumber( + body.timeout_ms, + normalizePositiveNumber(process.env.NANOBANANA_POLL_TIMEOUT_MS, 120000) + ); + const pollIntervalMs = normalizePositiveNumber( + body.poll_interval_ms, + normalizePositiveNumber(process.env.NANOBANANA_POLL_INTERVAL_MS, 2500) + ); + + let lastTaskData = null; + const deadline = Date.now() + timeoutMs; + + while (Date.now() < deadline) { + const pollResp = await fetch(`${statusUrl}?taskId=${encodeURIComponent(taskId)}`, { + method: "GET", + headers: { Authorization: `Bearer ${token}` }, + }); + + if (!pollResp.ok) { + const errorText = await pollResp.text(); + if (log) { + log.error( + "IMAGE", + `${provider} poll error ${pollResp.status}: ${errorText.slice(0, 200)}` + ); + } + return { success: false, status: pollResp.status, error: errorText }; + } + + const pollData = await pollResp.json(); + const taskData = pollData?.data || pollData; + lastTaskData = taskData; + + const successFlag = Number(taskData?.successFlag); + if (successFlag === 1) { + const normalized = await normalizeNanoBananaTaskResult(taskData, body, log); + + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 200, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + responseBody: { images_count: normalized.length, mode: "async", taskId }, + }).catch(() => {}); + + return { + success: true, + data: { + created: Math.floor(Date.now() / 1000), + data: normalized, + }, + }; + } + + if (successFlag === 2 || successFlag === 3) { + const errorText = + taskData?.errorMessage || `NanoBanana task failed (successFlag=${String(successFlag)})`; + + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 502, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: errorText.slice(0, 500), + responseBody: { taskId, successFlag, errorCode: taskData?.errorCode ?? null }, + }).catch(() => {}); + + return { success: false, status: 502, error: errorText }; + } + + await sleep(pollIntervalMs); + } + + const timeoutError = `NanoBanana task timeout after ${timeoutMs}ms (taskId=${taskId}, successFlag=${String(lastTaskData?.successFlag ?? "unknown")})`; + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 504, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: timeoutError, + responseBody: { taskId, lastSuccessFlag: lastTaskData?.successFlag ?? null }, + }).catch(() => {}); + + return { success: false, status: 504, error: timeoutError }; + } catch (err) { + if (log) log.error("IMAGE", `${provider} fetch error: ${err.message}`); + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 502, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: err.message, + }).catch(() => {}); + return { + success: false, + status: 502, + error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, + }; + } +} + + + +function normalizeNanoBananaSyncPayload(data, prompt) { + const images = []; + + if (data.image) { + images.push({ b64_json: data.image, revised_prompt: prompt }); + } else if (Array.isArray(data.images)) { + for (const img of data.images) { + images.push({ + b64_json: typeof img === "string" ? img : img?.image || img?.data, + revised_prompt: prompt, + }); + } + } else if (Array.isArray(data.data)) { + for (const img of data.data) { + if (!img) continue; + images.push(img); + } + } + + return { data: images.filter(Boolean) }; +} + + + +async function normalizeNanoBananaTaskResult(taskData, body, log) { + const response = taskData?.response || {}; + + const urlCandidates = [ + response?.resultImageUrl, + response?.originImageUrl, + taskData?.resultImageUrl, + taskData?.originImageUrl, + ].filter((v) => typeof v === "string" && v.length > 0); + + if (Array.isArray(response?.resultImageUrls)) { + for (const u of response.resultImageUrls) { + if (typeof u === "string" && u.length > 0) urlCandidates.push(u); + } + } + + const b64Candidates = [ + response?.resultImageBase64, + response?.resultImage, + taskData?.resultImageBase64, + taskData?.resultImage, + ].filter((v) => typeof v === "string" && v.length > 0); + + if (Array.isArray(response?.resultImageBase64List)) { + for (const b64 of response.resultImageBase64List) { + if (typeof b64 === "string" && b64.length > 0) b64Candidates.push(b64); + } + } + + const wantsBase64 = body.response_format === "b64_json"; + + if (wantsBase64) { + if (b64Candidates.length > 0) { + return b64Candidates.map((b64) => ({ b64_json: b64, revised_prompt: body.prompt })); + } + + if (urlCandidates.length > 0) { + const firstUrl = urlCandidates[0]; + const remoteImage = await fetchRemoteImage(firstUrl); + const base64 = remoteImage.buffer.toString("base64"); + return [{ b64_json: base64, revised_prompt: body.prompt }]; + } + } + + if (urlCandidates.length > 0) { + return urlCandidates.map((url) => ({ url, revised_prompt: body.prompt })); + } + + if (b64Candidates.length > 0) { + return b64Candidates.map((b64) => ({ b64_json: b64, revised_prompt: body.prompt })); + } + + if (log) { + log.warn( + "IMAGE", + `NanoBanana task completed without image payload: ${JSON.stringify(taskData).slice(0, 240)}` + ); + } + + return []; +} + + + +/** + * Handle SD WebUI image generation (local, no auth) + * POST {baseUrl} with { prompt, negative_prompt, width, height, steps } + * Response: { images: ["base64..."] } + */ +export async function handleSDWebUIImageGeneration({ model, provider, providerConfig, body, log }) { + const startTime = Date.now(); + const [width, height] = (body.size || "512x512").split("x").map(Number); + + const upstreamBody = { + prompt: body.prompt, + negative_prompt: body.negative_prompt || "", + width: width || 512, + height: height || 512, + steps: body.steps || 20, + cfg_scale: body.cfg_scale || 7, + sampler_name: body.sampler || "Euler a", + batch_size: body.n || 1, + override_settings: { + sd_model_checkpoint: model, + }, + }; + + if (log) { + const promptPreview = String(body.prompt ?? "").slice(0, 60); + log.info("IMAGE", `${provider}/${model} (sdwebui) | prompt: "${promptPreview}..."`); + } + + try { + const response = await fetch(providerConfig.baseUrl, { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify(upstreamBody), + }); + + if (!response.ok) { + const errorText = await response.text(); + if (log) + log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); + + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: response.status, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: errorText.slice(0, 500), + }).catch(() => {}); + + return { success: false, status: response.status, error: errorText }; + } + + const data = await response.json(); + // SD WebUI returns { images: ["base64...", ...] } + const images = (data.images || []).map((b64) => ({ + b64_json: b64, + revised_prompt: body.prompt, + })); + + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 200, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + responseBody: { images_count: images.length }, + }).catch(() => {}); + + return { + success: true, + data: { created: Math.floor(Date.now() / 1000), data: images }, + }; + } catch (err) { + if (log) log.error("IMAGE", `${provider} sdwebui error: ${err.message}`); + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 502, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: err.message, + }).catch(() => {}); + return { + success: false, + status: 502, + error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, + }; + } +} + + + +/** + * Handle ComfyUI image generation (local, no auth) + * Submits a txt2img workflow, polls for completion, fetches output + */ +export async function handleComfyUIImageGeneration({ model, provider, providerConfig, body, log }) { + const startTime = Date.now(); + const [width, height] = (body.size || "1024x1024").split("x").map(Number); + + // Default txt2img workflow template for ComfyUI + const workflow = { + "3": { + class_type: "KSampler", + inputs: { + seed: parseInt(randomUUID().replace(/-/g, "").substring(0, 8), 16) % 2 ** 32, + steps: body.steps || 20, + cfg: body.cfg_scale || 7, + sampler_name: "euler", + scheduler: "normal", + denoise: 1, + model: ["4", 0], + positive: ["6", 0], + negative: ["7", 0], + latent_image: ["5", 0], + }, + }, + "4": { + class_type: "CheckpointLoaderSimple", + inputs: { ckpt_name: model }, + }, + "5": { + class_type: "EmptyLatentImage", + inputs: { width: width || 1024, height: height || 1024, batch_size: body.n || 1 }, + }, + "6": { + class_type: "CLIPTextEncode", + inputs: { text: body.prompt, clip: ["4", 1] }, + }, + "7": { + class_type: "CLIPTextEncode", + inputs: { text: body.negative_prompt || "", clip: ["4", 1] }, + }, + "8": { + class_type: "VAEDecode", + inputs: { samples: ["3", 0], vae: ["4", 2] }, + }, + "9": { + class_type: "SaveImage", + inputs: { filename_prefix: "omniroute", images: ["8", 0] }, + }, + }; + + if (log) { + const promptPreview = String(body.prompt ?? "").slice(0, 60); + log.info("IMAGE", `${provider}/${model} (comfyui) | prompt: "${promptPreview}..."`); + } + + try { + const promptId = await submitComfyWorkflow(providerConfig.baseUrl, workflow); + const historyEntry = await pollComfyResult(providerConfig.baseUrl, promptId); + const outputFiles = extractComfyOutputFiles(historyEntry); + + const images = []; + for (const file of outputFiles) { + const buffer = await fetchComfyOutput( + providerConfig.baseUrl, + file.filename, + file.subfolder, + file.type + ); + const base64 = Buffer.from(buffer).toString("base64"); + images.push({ b64_json: base64, revised_prompt: body.prompt }); + } + + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 200, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + responseBody: { images_count: images.length }, + }).catch(() => {}); + + return { + success: true, + data: { created: Math.floor(Date.now() / 1000), data: images }, + }; + } catch (err) { + if (log) log.error("IMAGE", `${provider} comfyui error: ${err.message}`); + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 502, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: err.message, + }).catch(() => {}); + return { + success: false, + status: 502, + error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, + }; + } +} + + + +export async function handleHaiperImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, +}) { + const startTime = Date.now(); + const token = credentials?.apiKey || ""; + const prompt = typeof body.prompt === "string" ? body.prompt : String(body.prompt ?? ""); + if (log) { + log.info("IMAGE", `${provider}/${model} (haiper) | prompt: "${prompt.slice(0, 60)}..."`); + } + try { + const res = await fetch(providerConfig.baseUrl, { + method: "POST", + headers: { "Content-Type": "application/json", HAIPER_KEY: token }, + body: JSON.stringify({ prompt, aspect_ratio: body.aspect_ratio || "16:9" }), + }); + if (!res.ok) { + const errorText = await res.text(); + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: res.status, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: errorText.slice(0, 500), + }).catch(() => {}); + return { success: false, status: res.status, error: errorText }; + } + const { job_id } = await res.json(); + const deadline = Date.now() + 300000; + while (Date.now() < deadline) { + await sleep(5000); + const statusRes = await fetch(`${providerConfig.statusUrl}/${job_id}`, { + headers: { HAIPER_KEY: token }, + }); + const status = await statusRes.json(); + if (status.status === "completed" || status.status === "succeeded") { + const imgUrl = status.creation_url || status.output?.image_url; + if (imgUrl) { + const imgRes = await fetch(imgUrl); + if (!imgRes.ok) { + return { + success: false, + status: imgRes.status, + error: `Failed to download image: ${imgRes.status}`, + }; + } + const buf = await imgRes.arrayBuffer(); + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 200, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + }).catch(() => {}); + return { + success: true, + data: { + created: Math.floor(Date.now() / 1000), + data: [{ b64_json: Buffer.from(buf).toString("base64") }], + }, + }; + } + } + if (status.status === "failed") { + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 502, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: "Haiper image generation failed", + }).catch(() => {}); + return { success: false, status: 502, error: "Haiper image generation failed" }; + } + } + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 504, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: "Haiper image generation timed out", + }).catch(() => {}); + return { success: false, status: 504, error: "Haiper image generation timed out" }; + } catch (err) { + if (log) log.error("IMAGE", `${provider} haiper error: ${err.message}`); + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 502, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: err.message, + }).catch(() => {}); + return { + success: false, + status: 502, + error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, + }; + } +} + + + +export async function handleLeonardoImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, +}) { + const startTime = Date.now(); + const token = credentials?.apiKey || ""; + const prompt = typeof body.prompt === "string" ? body.prompt : String(body.prompt ?? ""); + if (log) { + log.info("IMAGE", `${provider}/${model} (leonardo) | prompt: "${prompt.slice(0, 60)}..."`); + } + try { + const res = await fetch(providerConfig.baseUrl, { + method: "POST", + headers: { "Content-Type": "application/json", Authorization: `Bearer ${token}` }, + body: JSON.stringify({ + modelId: model || "phoenix", + prompt, + width: body.width || 1024, + height: body.height || 1024, + num_images: 1, + }), + }); + if (!res.ok) { + const errorText = await res.text(); + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: res.status, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: errorText.slice(0, 500), + }).catch(() => {}); + return { success: false, status: res.status, error: errorText }; + } + const { sdGenerationJob } = await res.json(); + const genId = sdGenerationJob?.generationId; + if (!genId) { + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 502, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: "No generation ID returned", + }).catch(() => {}); + return { success: false, status: 502, error: "No generation ID returned" }; + } + const deadline = Date.now() + 300000; + while (Date.now() < deadline) { + await sleep(5000); + const statusRes = await fetch(`${providerConfig.baseUrl}/${genId}`, { + headers: { Authorization: `Bearer ${token}` }, + }); + const status = await statusRes.json(); + const gen = status.generations_by_pk || status; + if (gen.status === "COMPLETE") { + const imgUrl = gen.generated_images?.[0]?.url; + if (imgUrl) { + const imgRes = await fetch(imgUrl); + if (!imgRes.ok) { + return { + success: false, + status: imgRes.status, + error: `Failed to download image: ${imgRes.status}`, + }; + } + const buf = await imgRes.arrayBuffer(); + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 200, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + }).catch(() => {}); + return { + success: true, + data: { + created: Math.floor(Date.now() / 1000), + data: [{ b64_json: Buffer.from(buf).toString("base64") }], + }, + }; + } + } + if (gen.status === "FAILED") { + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 502, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: "Leonardo image generation failed", + }).catch(() => {}); + return { success: false, status: 502, error: "Leonardo image generation failed" }; + } + } + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 504, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: "Leonardo image generation timed out", + }).catch(() => {}); + return { success: false, status: 504, error: "Leonardo image generation timed out" }; + } catch (err) { + if (log) log.error("IMAGE", `${provider} leonardo error: ${err.message}`); + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 502, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: err.message, + }).catch(() => {}); + return { + success: false, + status: 502, + error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, + }; + } +} + + + +export async function handleIdeogramImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, +}) { + const startTime = Date.now(); + const token = credentials?.apiKey || ""; + const prompt = typeof body.prompt === "string" ? body.prompt : String(body.prompt ?? ""); + if (log) { + log.info("IMAGE", `${provider}/${model} (ideogram) | prompt: "${prompt.slice(0, 60)}..."`); + } + try { + const res = await fetch(providerConfig.baseUrl, { + method: "POST", + headers: { "Content-Type": "application/json", "Api-Key": token }, + body: JSON.stringify({ prompt, aspect_ratio: "ASPECT_16_9", model: model || "V_3" }), + }); + if (!res.ok) { + const errorText = await res.text(); + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: res.status, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: errorText.slice(0, 500), + }).catch(() => {}); + return { success: false, status: res.status, error: errorText }; + } + const data = await res.json(); + if (data.data && data.data.length > 0) { + const imgUrl = data.data[0].url; + const imgRes = await fetch(imgUrl); + if (!imgRes.ok) { + return { + success: false, + status: imgRes.status, + error: `Failed to download image: ${imgRes.status}`, + }; + } + const buf = await imgRes.arrayBuffer(); + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 200, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + }).catch(() => {}); + return { + success: true, + data: { + created: Math.floor(Date.now() / 1000), + data: [{ b64_json: Buffer.from(buf).toString("base64") }], + }, + }; + } + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 502, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: "No images returned from Ideogram", + }).catch(() => {}); + return { success: false, status: 502, error: "No images returned from Ideogram" }; + } catch (err) { + if (log) log.error("IMAGE", `${provider} ideogram error: ${err.message}`); + saveCallLog({ + method: "POST", + path: "/v1/images/generations", + status: 502, + model: `${provider}/${model}`, + provider, + duration: Date.now() - startTime, + error: err.message, + }).catch(() => {}); + return { + success: false, + status: 502, + error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, + }; + } +} + diff --git a/open-sse/handlers/imageGeneration/stability.ts b/open-sse/handlers/imageGeneration/stability.ts new file mode 100644 index 00000000000..1513e2ab055 --- /dev/null +++ b/open-sse/handlers/imageGeneration/stability.ts @@ -0,0 +1,265 @@ + +/** + * Image Generation Handler + * + * Handles POST /v1/images/generations requests. + * Proxies to upstream image generation providers using OpenAI-compatible format. + * + * Request format (OpenAI-compatible): + * { + * "model": "openai/gpt-image-2", + * "prompt": "a beautiful sunset over mountains", + * "n": 1, + * "size": "1024x1024", + * "quality": "standard", // optional: "standard" | "hd" + * "response_format": "url" // optional: "url" | "b64_json" + * } + */ + +import { getImageProvider, parseImageModel } from "../../config/imageRegistry.ts"; + +import { mapImageSize } from "../../translator/image/sizeMapper.ts"; + +import { sanitizeErrorMessage, sanitizeUpstreamDetails } from "../../utils/error.ts"; + +import { extractImageInputs, normalizeRequestedImageFormat, appendOptionalFormValue, resolveImageSource, appendImageFormValue, normalizeProviderImagePayload } from "./utils.ts"; +import { saveImageErrorResult, saveImageSuccessResult } from "./logging.ts"; + + + +const STABILITY_GENERATION_ENDPOINTS = { + "sd3.5-large": "/v2beta/stable-image/generate/sd3", + "sd3.5-large-turbo": "/v2beta/stable-image/generate/sd3", + "sd3.5-medium": "/v2beta/stable-image/generate/sd3", + "sd3.5-flash": "/v2beta/stable-image/generate/sd3", + "stable-image-ultra": "/v2beta/stable-image/generate/ultra", + "stable-image-core": "/v2beta/stable-image/generate/core", +}; + + + +const STABILITY_EDIT_ENDPOINTS = { + inpaint: "/v2beta/stable-image/edit/inpaint", + outpaint: "/v2beta/stable-image/edit/outpaint", + erase: "/v2beta/stable-image/edit/erase", + "search-and-replace": "/v2beta/stable-image/edit/search-and-replace", + "search-and-recolor": "/v2beta/stable-image/edit/search-and-recolor", + "remove-background": "/v2beta/stable-image/edit/remove-background", + "replace-background-and-relight": "/v2beta/stable-image/edit/replace-background-and-relight", + fast: "/v2beta/stable-image/upscale/fast", + conservative: "/v2beta/stable-image/upscale/conservative", + creative: "/v2beta/stable-image/upscale/creative", + sketch: "/v2beta/stable-image/control/sketch", + structure: "/v2beta/stable-image/control/structure", + style: "/v2beta/stable-image/control/style", + "style-transfer": "/v2beta/stable-image/control/style-transfer", +}; + + + +const STABILITY_CONTROL_MODELS = new Set(["sketch", "structure", "style", "style-transfer"]); + + + +export async function handleStabilityAIImageGeneration({ + model, + provider, + providerConfig, + body, + credentials, + log, +}) { + const startTime = Date.now(); + const token = credentials.apiKey || credentials.accessToken; + const endpoint = STABILITY_GENERATION_ENDPOINTS[model] || STABILITY_EDIT_ENDPOINTS[model]; + + if (!endpoint) { + return { + success: false, + status: 400, + error: `Unsupported Stability AI image model: ${model}`, + }; + } + + const { imageUrl, maskUrl } = extractImageInputs(body); + const upstreamBody: Record = { + output_format: + model === "remove-background" + ? normalizeRequestedImageFormat(body, "png", ["png", "webp"]) + : normalizeRequestedImageFormat(body, "png"), + }; + const formData = new FormData(); + + appendOptionalFormValue(formData, "output_format", upstreamBody.output_format); + if (body.prompt) { + upstreamBody.prompt = body.prompt; + appendOptionalFormValue(formData, "prompt", body.prompt); + } + if (body.negative_prompt) { + upstreamBody.negative_prompt = body.negative_prompt; + appendOptionalFormValue(formData, "negative_prompt", body.negative_prompt); + } + if (body.seed !== undefined) { + upstreamBody.seed = body.seed; + appendOptionalFormValue(formData, "seed", body.seed); + } + + try { + if (STABILITY_GENERATION_ENDPOINTS[model]) { + if (model.startsWith("sd3.5")) { + upstreamBody.model = model; + appendOptionalFormValue(formData, "model", model); + } + + if (imageUrl) { + const imageSource = await resolveImageSource(imageUrl); + upstreamBody.mode = "image-to-image"; + appendOptionalFormValue(formData, "mode", "image-to-image"); + upstreamBody.image = imageSource.base64; + appendImageFormValue(formData, "image", imageSource, "image"); + if (body.strength !== undefined) { + upstreamBody.strength = body.strength; + appendOptionalFormValue(formData, "strength", body.strength); + } + } else { + upstreamBody.mode = "text-to-image"; + appendOptionalFormValue(formData, "mode", "text-to-image"); + } + + if (!model.startsWith("sd3.5") || !imageUrl) { + const aspectRatio = body.aspect_ratio || mapImageSize(body.size); + upstreamBody.aspect_ratio = aspectRatio; + appendOptionalFormValue(formData, "aspect_ratio", aspectRatio); + } + + if (body.style_preset) { + upstreamBody.style_preset = body.style_preset; + appendOptionalFormValue(formData, "style_preset", body.style_preset); + } + } else { + if (imageUrl) { + const imageSource = await resolveImageSource(imageUrl); + upstreamBody.image = imageSource.base64; + appendImageFormValue(formData, "image", imageSource, "image"); + } + + if (maskUrl && shouldIncludeStabilityMask(model)) { + const maskSource = await resolveImageSource(maskUrl); + upstreamBody.mask = maskSource.base64; + appendImageFormValue(formData, "mask", maskSource, "mask"); + } + + if (body.search_prompt) { + upstreamBody.search_prompt = body.search_prompt; + appendOptionalFormValue(formData, "search_prompt", body.search_prompt); + } + if (body.grow_mask !== undefined) { + upstreamBody.grow_mask = body.grow_mask; + appendOptionalFormValue(formData, "grow_mask", body.grow_mask); + } + if (body.control_strength !== undefined) { + upstreamBody.control_strength = body.control_strength; + appendOptionalFormValue(formData, "control_strength", body.control_strength); + } + if (body.creativity !== undefined) { + upstreamBody.creativity = body.creativity; + appendOptionalFormValue(formData, "creativity", body.creativity); + } + if (body.left !== undefined) { + upstreamBody.left = body.left; + appendOptionalFormValue(formData, "left", body.left); + } + if (body.right !== undefined) { + upstreamBody.right = body.right; + appendOptionalFormValue(formData, "right", body.right); + } + if (body.up !== undefined) { + upstreamBody.up = body.up; + appendOptionalFormValue(formData, "up", body.up); + } + if (body.down !== undefined) { + upstreamBody.down = body.down; + appendOptionalFormValue(formData, "down", body.down); + } + if (body.style_preset) { + upstreamBody.style_preset = body.style_preset; + appendOptionalFormValue(formData, "style_preset", body.style_preset); + } + + if (STABILITY_CONTROL_MODELS.has(model) && !upstreamBody.prompt) { + upstreamBody.prompt = body.prompt || ""; + appendOptionalFormValue(formData, "prompt", body.prompt || ""); + } + } + + if (log) { + const promptPreview = String(body.prompt ?? "").slice(0, 60); + log.info("IMAGE", `${provider}/${model} (stability-ai) | prompt: "${promptPreview}..."`); + } + + const response = await fetch(`${providerConfig.baseUrl.replace(/\/$/, "")}${endpoint}`, { + method: "POST", + headers: { + Accept: "application/json", + Authorization: `Bearer ${token}`, + }, + body: formData, + }); + + if (!response.ok) { + const errorText = await response.text(); + if (log) + log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); + return saveImageErrorResult({ + provider, + model, + status: response.status, + startTime, + error: errorText, + requestBody: upstreamBody, + }); + } + + const contentType = response.headers.get("content-type") || ""; + let payload; + if (contentType.includes("application/json")) { + payload = await response.json(); + } else { + const buffer = Buffer.from(await response.arrayBuffer()); + payload = { image: buffer.toString("base64") }; + } + + const images = await normalizeProviderImagePayload(payload, body, log); + return saveImageSuccessResult({ + provider, + model, + startTime, + requestBody: upstreamBody, + responseBody: { images_count: images.length }, + created: payload.created, + images, + }); + } catch (err) { + if (log) log.error("IMAGE", `${provider} fetch error: ${err.message}`); + return saveImageErrorResult({ + provider, + model, + status: 502, + startTime, + error: `Image provider error: ${sanitizeErrorMessage((err as Error).message || err)}`, + }); + } +} + + + +function shouldIncludeStabilityMask(model) { + return new Set([ + "inpaint", + "erase", + "search-and-replace", + "search-and-recolor", + "replace-background-and-relight", + ]).has(model); +} + diff --git a/open-sse/handlers/imageGeneration/utils.ts b/open-sse/handlers/imageGeneration/utils.ts new file mode 100644 index 00000000000..127d293e3cf --- /dev/null +++ b/open-sse/handlers/imageGeneration/utils.ts @@ -0,0 +1,539 @@ + +/** + * Image Generation Handler + * + * Handles POST /v1/images/generations requests. + * Proxies to upstream image generation providers using OpenAI-compatible format. + * + * Request format (OpenAI-compatible): + * { + * "model": "openai/gpt-image-2", + * "prompt": "a beautiful sunset over mountains", + * "n": 1, + * "size": "1024x1024", + * "quality": "standard", // optional: "standard" | "hd" + * "response_format": "url" // optional: "url" | "b64_json" + * } + */ + +import { getImageProvider, parseImageModel } from "../../config/imageRegistry.ts"; + +import { HTTP_STATUS } from "../../config/constants.ts"; + +import { applyAntigravityClientProfileHeaders } from "../../services/antigravityClientProfile.ts"; + +import { getAntigravityEnvelopeUserAgent } from "../../services/antigravityIdentity.ts"; + +import { kieExecutor } from "../../executors/kie.ts"; + +import { mapImageSize } from "../../translator/image/sizeMapper.ts"; + +import { getCodexClientVersion, getCodexUserAgent } from "../../config/codexClient.ts"; + +import { ChatGptWebExecutor } from "../../executors/chatgpt-web.ts"; + +import { getChatGptImage, findChatGptImageBySha256 } from "../../services/chatgptImageCache.ts"; + +import { createHash } from "node:crypto"; + +import { sleep } from "../../utils/sleep.ts"; + +import { + getKieErrorMessage, + getKieErrorStatus, + isJsonObject, + parseKieResultJson, +} from "../../utils/kieTask.ts"; + +import { + submitComfyWorkflow, + pollComfyResult, + fetchComfyOutput, + extractComfyOutputFiles, +} from "../../utils/comfyuiClient.ts"; + +import { fetchRemoteImage } from "@/shared/network/remoteImageFetch"; + +import { FetchTimeoutError, fetchWithTimeout, getConfiguredTimeout } from "@/shared/utils/fetchTimeout"; + +import { sanitizeErrorMessage, sanitizeUpstreamDetails } from "../../utils/error.ts"; + + + + +const IMAGE_ASPECT_RATIO_PATTERN = /^\d+:\d+$/; + + + +/** + * Resolve the upstream images endpoint for a custom (OpenAI-compatible) image + * provider node (#3205). + * + * Custom provider nodes store their base URL the same way the chat path does: + * in `credentials.providerSpecificData.baseUrl` (e.g. `https://example.com/v1`), + * NOT as a top-level `credentials.baseUrl`. Older callers may still pass a + * top-level `baseUrl`, so we honor that as a secondary source. When neither is + * present we fall back to `fallback` (the built-in Gemini OpenAI endpoint). + * + * Resolution order: providerSpecificData.baseUrl → credentials.baseUrl → fallback. + * + * A node base URL like `https://example.com/v1` is normalized and the + * OpenAI-compatible `/images/generations` path appended (mirroring + * `buildOpenAICompatibleUrl` in services/provider.ts). A node URL that already + * ends in `/images/generations` is returned as-is (no double-append). The + * `fallback` value is assumed to already be a complete URL and is returned + * verbatim. + */ +export function resolveImageBaseUrl( + credentials: + | { baseUrl?: unknown; providerSpecificData?: { baseUrl?: unknown } | null } + | null + | undefined, + fallback: string, + endpoint: "generations" | "edits" = "generations" +): string { + const psd = credentials?.providerSpecificData; + const psdBaseUrl = + psd && typeof psd === "object" && typeof psd.baseUrl === "string" && psd.baseUrl.trim() + ? psd.baseUrl.trim() + : null; + const topLevelBaseUrl = + typeof credentials?.baseUrl === "string" && credentials.baseUrl.trim() + ? credentials.baseUrl.trim() + : null; + const nodeBaseUrl = psdBaseUrl || topLevelBaseUrl; + + if (!nodeBaseUrl) return fallback; + + // A single configured node serves both image routes: honor a base URL that already + // points at the requested OpenAI image path, and rewrite one that points at the other + // image endpoint (e.g. `.../images/generations` requested for edits) (#3214/#3215). + const suffix = `/images/${endpoint}`; + // Trim trailing slashes without a backtracking-prone regex (`/\/+$/` is a + // polynomial-ReDoS pattern on long runs of "/" — CodeQL js/polynomial-redos). + let normalized = nodeBaseUrl; + while (normalized.endsWith("/")) normalized = normalized.slice(0, -1); + if (normalized.endsWith(suffix)) return normalized; + const stripped = normalized.replace(/\/images\/(?:generations|edits)$/, ""); + return `${stripped}${suffix}`; +} + + + +export function normalizeImageAspectRatio(value: unknown, fallbackSize: unknown): string { + if (typeof value === "string") { + const trimmedValue = value.trim(); + if (IMAGE_ASPECT_RATIO_PATTERN.test(trimmedValue)) return trimmedValue; + } + return mapImageSize(typeof fallbackSize === "string" ? fallbackSize : null); +} + + + +function parseJsonOrNull(value: string): unknown | null { + try { + return JSON.parse(value); + } catch { + return null; + } +} + + + +export function sanitizeImageProviderError(errorText: string): unknown { + const parsed = parseJsonOrNull(errorText); + if (parsed !== null) { + return sanitizeUpstreamDetails(parsed) || sanitizeErrorMessage(errorText); + } + return sanitizeErrorMessage(errorText); +} + + + +function formatImageProviderError(err) { + const sanitized = sanitizeErrorMessage(err); + const message = (sanitized || "").replace(/^Error:\s*/i, "").trim(); + return message ? `Image provider error: ${message}` : "Image provider error"; +} + + + +export function appendOptionalFormValue(formData, key, value) { + if (value === undefined || value === null || value === "") return; + formData.append(key, String(value)); +} + + + +export function appendImageFormValue(formData, key, source, filename) { + formData.append( + key, + new Blob([source.buffer], { + type: source.contentType || "application/octet-stream", + }), + filename + ); +} + + + +export function extractImageInputs(body) { + const imageUrls = []; + const seen = new Set(); + + const pushCandidate = (candidate) => { + if (typeof candidate !== "string") return; + const trimmed = candidate.trim(); + if (!trimmed || seen.has(trimmed)) return; + seen.add(trimmed); + imageUrls.push(trimmed); + }; + + pushCandidate(body?.image_url); + pushCandidate(body?.image); + + if (Array.isArray(body?.imageUrls)) { + for (const candidate of body.imageUrls) pushCandidate(candidate); + } + + if (Array.isArray(body?.image_urls)) { + for (const candidate of body.image_urls) pushCandidate(candidate); + } + + if (Array.isArray(body?.messages)) { + for (const msg of body.messages) { + if (!Array.isArray(msg?.content)) continue; + for (const part of msg.content) { + if (part?.type === "image_url") { + pushCandidate(part?.image_url?.url); + } + } + } + } + + return { + imageUrl: imageUrls[0] || null, + imageUrls, + maskUrl: + typeof body?.mask_url === "string" + ? body.mask_url + : typeof body?.mask === "string" + ? body.mask + : null, + }; +} + + + +export async function resolveImageSource(source) { + if (typeof source !== "string" || source.trim().length === 0) { + throw new Error("Invalid image source"); + } + + const trimmed = source.trim(); + const dataUriMatch = /^data:([^;]+);base64,(.+)$/i.exec(trimmed); + if (dataUriMatch) { + const [, contentType, base64] = dataUriMatch; + return { + buffer: Buffer.from(base64, "base64"), + base64, + contentType, + }; + } + + if (isHttpUrl(trimmed)) { + const remoteImage = await fetchRemoteImage(trimmed); + return { + buffer: remoteImage.buffer, + base64: remoteImage.buffer.toString("base64"), + contentType: remoteImage.contentType, + }; + } + + return { + buffer: Buffer.from(trimmed, "base64"), + base64: trimmed, + contentType: "application/octet-stream", + }; +} + + + +export function parseSizeToDimensions(size, fallback = 1024) { + if (typeof size !== "string" || !size.includes("x")) { + return { width: fallback, height: fallback }; + } + + const [widthRaw, heightRaw] = size.split("x"); + const width = Number(widthRaw); + const height = Number(heightRaw); + return { + width: Number.isFinite(width) && width > 0 ? width : fallback, + height: Number.isFinite(height) && height > 0 ? height : fallback, + }; +} + + + +export function normalizeRequestedImageFormat( + body, + fallback = "png", + allowedFormats = ["jpeg", "png", "webp"] +) { + const formatCandidate = + typeof body?.output_format === "string" + ? body.output_format.toLowerCase() + : typeof body?.response_format === "string" && + !["url", "b64_json"].includes(body.response_format.toLowerCase()) + ? body.response_format.toLowerCase() + : fallback; + + if (allowedFormats.includes(formatCandidate)) { + return formatCandidate; + } + + return fallback; +} + + + +export async function normalizeProviderImagePayload(payload, body, log) { + const candidates = []; + + const pushCandidate = (value) => { + if (value === undefined || value === null) return; + candidates.push(value); + }; + + if (Array.isArray(payload?.data)) { + for (const item of payload.data) pushCandidate(item); + } + + if (Array.isArray(payload?.images)) { + for (const item of payload.images) pushCandidate(item); + } + + if (payload?.image) pushCandidate({ b64_json: payload.image }); + if (payload?.url) pushCandidate({ url: payload.url }); + if (payload?.sample) pushCandidate({ url: payload.sample }); + if (payload?.result?.sample) pushCandidate({ url: payload.result.sample }); + if (Array.isArray(payload?.result?.images)) { + for (const item of payload.result.images) pushCandidate(item); + } + + const normalized = []; + for (const candidate of candidates) { + const item = await normalizeProviderImageCandidate(candidate, body); + if (item) normalized.push(item); + } + + if (normalized.length === 0 && log) { + log.warn( + "IMAGE", + `Provider returned no recognizable image payload: ${JSON.stringify(payload).slice(0, 240)}` + ); + } + + return normalized; +} + + + +async function normalizeProviderImageCandidate(candidate, body) { + const wantsBase64 = body?.response_format === "b64_json"; + let url = null; + let b64 = null; + + if (typeof candidate === "string") { + const dataUriMatch = /^data:[^;]+;base64,(.+)$/i.exec(candidate); + if (dataUriMatch) { + b64 = dataUriMatch[1]; + } else if (isHttpUrl(candidate)) { + url = candidate; + } else { + b64 = candidate; + } + } else if (candidate && typeof candidate === "object") { + url = + firstString(candidate.url, candidate.image_url, candidate.sample, candidate.file_url) || null; + b64 = + firstString(candidate.b64_json, candidate.image, candidate.base64, candidate.data) || null; + } + + if (wantsBase64 && !b64 && url) { + b64 = (await resolveImageSource(url)).base64; + } + + if (url && !wantsBase64) { + return { url, revised_prompt: body?.prompt }; + } + + if (b64) { + return { b64_json: b64, revised_prompt: body?.prompt }; + } + + if (url) { + return { url, revised_prompt: body?.prompt }; + } + + return null; +} + + + +function firstString(...values) { + for (const value of values) { + if (typeof value === "string" && value.length > 0) return value; + } + return null; +} + + + +export function isHttpUrl(value) { + return typeof value === "string" && /^https?:\/\//i.test(value); +} + + + +/** + * Codex image generation — translate GPT-Image-style /v1/images/generations + * request into a /v1/responses call with the `image_generation` hosted tool, + * parse the SSE stream, and return the base64 PNG in OpenAI image response shape. + * + * Requires ChatGPT OAuth credentials (Codex provider connection). The hosted + * image_generation tool is only served upstream under ChatGPT auth; API-key + * users will receive a 400 from OpenAI. + */ +export function extractImageGenerationCalls( + sseText: string +): Array<{ b64: string; revisedPrompt: string | null }> { + const results: Array<{ b64: string; revisedPrompt: string | null }> = []; + const lines = String(sseText || "").split("\n"); + for (const line of lines) { + const trimmed = line.trim(); + if (!trimmed.startsWith("data:")) continue; + const payload = trimmed.slice(5).trim(); + if (!payload || payload === "[DONE]") continue; + let evt: Record; + try { + evt = JSON.parse(payload) as Record; + } catch { + continue; + } + if (evt?.type !== "response.output_item.done") continue; + const item = evt.item as Record | undefined; + if (!item || item.type !== "image_generation_call") continue; + const result = typeof item.result === "string" ? item.result : ""; + if (!result) continue; + const revisedPrompt = typeof item.revised_prompt === "string" ? item.revised_prompt : null; + results.push({ b64: result, revisedPrompt }); + } + return results; +} + + + +// The image_generation hosted tool accepts { "auto" | "low" | "medium" | "high" } +// for `quality`. Legacy image clients often send "standard" / "hd". Map those values +// so OpenWebUI's quality dropdown doesn't silently get rejected upstream. +export function mapLegacyImageQualityToImageTool(value: string): string { + const normalized = value.toLowerCase(); + if (normalized === "standard") return "medium"; + if (normalized === "hd") return "high"; + return normalized; +} + + + +/** + * Fetch a single image endpoint and normalize response + */ +export async function fetchImageEndpoint(url, headers, body, provider, log) { + try { + let response; + try { + response = await fetchWithTimeout(url, { + method: "POST", + headers, + body, + timeoutMs: getConfiguredTimeout(), + }); + } catch (err: unknown) { + const isAbortError = + typeof err === "object" && + err !== null && + "name" in err && + (err as { name?: unknown }).name === "AbortError"; + if (err instanceof FetchTimeoutError || isAbortError) { + const message = err instanceof Error ? err.message : String(err); + if (log) { + log.error("IMAGE", `${provider} fetch error: ${message}`); + } + return { + success: false, + status: 504, + error: `Image provider error: ${sanitizeErrorMessage(message || err)}`, + }; + } + throw err; + } + + if (!response.ok) { + const errorText = await response.text(); + if (log) { + log.error("IMAGE", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); + } + return { + success: false, + status: response.status, + error: errorText, + }; + } + + const data = await response.json(); + + // Normalize response to OpenAI format + return { + success: true, + data: { + created: data.created || Math.floor(Date.now() / 1000), + data: data.data || [], + }, + }; + } catch (err: unknown) { + const message = err instanceof Error ? err.message : String(err); + if (log) { + log.error("IMAGE", `${provider} fetch error: ${message}`); + } + return { + success: false, + status: 502, + error: `Image provider error: ${sanitizeErrorMessage(message || err)}`, + }; + } +} + + + +export function inferResolutionFromSize(size) { + if (typeof size !== "string") return null; + const [wRaw, hRaw] = size.split("x"); + const width = Number(wRaw); + const height = Number(hRaw); + if (!Number.isFinite(width) || !Number.isFinite(height) || width <= 0 || height <= 0) return null; + + const longestSide = Math.max(width, height); + if (longestSide <= 1024) return "1K"; + if (longestSide <= 2048) return "2K"; + return "4K"; +} + + + +export function normalizePositiveNumber(value, fallback) { + const n = Number(value); + if (!Number.isFinite(n) || n <= 0) return fallback; + return Math.floor(n); +} + diff --git a/open-sse/package.json b/open-sse/package.json index 88578a28bdf..d3ac81a28d1 100644 --- a/open-sse/package.json +++ b/open-sse/package.json @@ -1,6 +1,6 @@ { "name": "@omniroute/open-sse", - "version": "3.8.22", + "version": "3.8.21", "description": "Express SSE sidecar for OmniRoute — handles streaming, protocol translation, and provider orchestration", "type": "module", "main": "index.js", diff --git a/open-sse/services/accountFallback.ts b/open-sse/services/accountFallback.ts index ad0b2a344e0..729ee8b4332 100644 --- a/open-sse/services/accountFallback.ts +++ b/open-sse/services/accountFallback.ts @@ -27,7 +27,9 @@ import { looksLikeQuotaExhausted, type FailureKind, } from "../../src/shared/utils/classify429"; +import { resolveProviderId } from "../../src/shared/constants/providers"; import { resolveUseUpstream429BreakerHints } from "../../src/shared/utils/providerHints"; +import { isRpdExhausted, isRpmExhausted } from "./geminiRateLimitTracker.ts"; export type ProviderProfile = { baseCooldownMs: number; @@ -65,6 +67,9 @@ type ModelFailureState = { failureCount: number; lastFailureAt: number; resetAfterMs: number; + /** Cooldown applied on the last failure — extends the escalation window so a + * model that fails again right after its lockout expires keeps escalating. */ + lastCooldownMs?: number; }; type AccountState = JsonRecord & { id?: string | null; @@ -150,9 +155,6 @@ export const CREDITS_EXHAUSTED_SIGNALS = [ "credits exhausted", "out of credits", "payment required", - "resource has been exhausted", - "resource_exhausted", - "check quota", "free tier of the model has been exhausted", ]; @@ -350,8 +352,20 @@ export async function getRuntimeProviderProfile(provider: string | null | undefi const modelLockouts = new Map(); const modelFailureState = new Map(); +// Aliases (e.g. "cx" → "codex") must share lockout state with their canonical +// provider, otherwise a model locked via one spelling stays routable via the other. +const canonicalProviderCache = new Map(); +function getCanonicalLockProvider(provider: string): string { + let canonical = canonicalProviderCache.get(provider); + if (!canonical) { + canonical = resolveProviderId(provider); + canonicalProviderCache.set(provider, canonical); + } + return canonical; +} + function getModelLockKey(provider: string, connectionId: string, model: string) { - return `${provider}:${connectionId}:${model}`; + return `${getCanonicalLockProvider(provider)}:${connectionId}:${model}`; } function getFailureWindowMs(profile: ProviderProfile | null = null, fallbackMs = 30 * 60 * 1000) { @@ -367,7 +381,9 @@ function cleanupModelLockKey(key: string, now = Date.now()) { const failure = modelFailureState.get(key); if (!failure) return; - if (now - failure.lastFailureAt <= failure.resetAfterMs) return; + // The escalation window extends past the applied cooldown: a model that fails + // again right after its lockout expires must keep escalating, not reset to 1. + if (now - failure.lastFailureAt <= failure.resetAfterMs + (failure.lastCooldownMs ?? 0)) return; if (modelLockouts.has(key)) return; modelFailureState.delete(key); } @@ -468,7 +484,7 @@ export function recordModelLockoutFailure( status: number, fallbackCooldownMs: number, profile: ProviderProfile | null = null, - options: { exactCooldownMs?: number | null } = {} + options: { exactCooldownMs?: number | null; maxCooldownMs?: number } = {} ) { ensureCleanupTimer(); const key = getModelLockKey(provider, connectionId, model); @@ -483,24 +499,39 @@ export function recordModelLockoutFailure( const resetAfterMs = getFailureWindowMs(profile); const previous = modelFailureState.get(key); - const withinWindow = previous && now - previous.lastFailureAt <= previous.resetAfterMs; + // Escalation window extends past the previously applied cooldown so a model + // that fails again right after its lockout expires keeps escalating. + const withinWindow = + previous && + now - previous.lastFailureAt <= previous.resetAfterMs + (previous.lastCooldownMs ?? 0); const failureCount = withinWindow ? previous.failureCount + 1 : 1; - modelFailureState.set(key, { - failureCount, - lastFailureAt: now, - resetAfterMs, - }); const baseCooldownMs = getModelLockBaseCooldown(status, fallbackCooldownMs, profile); + // Cap exponential backoff so repeated failures cannot produce absurdly long + // lockouts; exact cooldowns (e.g. daily-quota until-midnight) are not capped. + const maxCooldownMs = + typeof options.maxCooldownMs === "number" && options.maxCooldownMs > 0 + ? options.maxCooldownMs + : BACKOFF_CONFIG.max; const cooldownMs = typeof options.exactCooldownMs === "number" && options.exactCooldownMs > 0 ? options.exactCooldownMs - : getScaledCooldown( - baseCooldownMs, - failureCount, - profile?.maxBackoffSteps ?? BACKOFF_CONFIG.maxLevel + : Math.min( + getScaledCooldown( + baseCooldownMs, + failureCount, + profile?.maxBackoffSteps ?? BACKOFF_CONFIG.maxLevel + ), + maxCooldownMs ); + modelFailureState.set(key, { + failureCount, + lastFailureAt: now, + resetAfterMs, + lastCooldownMs: cooldownMs, + }); + lockModel(provider, connectionId, model, reason, cooldownMs, { failureCount, lastFailureAt: now, @@ -590,6 +621,36 @@ export function shouldMarkAccountExhaustedFrom429( ); } +export function classifyLockoutReason(status: number): string { + if (status === 429) return "rate_limit"; + if (status === 403) return "quota_exhausted"; + return "unknown"; +} + +export type DecayResult = { cleared: boolean; newFailureCount: number }; + +export function decayModelFailureCount( + provider: string, + connectionId: string, + model: string +): DecayResult { + const key = getModelLockKey(provider, connectionId, model); + const failure = modelFailureState.get(key); + if (!failure) return { cleared: false, newFailureCount: 0 }; + + const newFailureCount = Math.floor(failure.failureCount / 2); + if (newFailureCount === 0) { + modelFailureState.delete(key); + return { cleared: true, newFailureCount: 0 }; + } else { + modelFailureState.set(key, { + ...failure, + failureCount: newFailureCount, + }); + return { cleared: false, newFailureCount }; + } +} + /** * Clear all in-memory model lockouts and failure state (for tests / full reset). */ @@ -933,8 +994,9 @@ export function parseRetryFromErrorText(errorText: unknown): number | null { // 2026-05-17T10:00:00Z" or "Please wait until 2026-05-17T10:00:00.000Z"). // Convert to a future-duration in milliseconds if it parses. const isoMatch = - /\b(?:try again at|wait until|reset(?:s)? at|available at|retry after)\s+(\d{4}-\d{2}-\d{2}[Tt ]\d{2}:\d{2}(?::\d{2})?(?:\.\d+)?(?:Z|[+-]\d{2}:?\d{2})?)/i - .exec(msg); + /\b(?:try again at|wait until|reset(?:s)? at|available at|retry after)\s+(\d{4}-\d{2}-\d{2}[Tt ]\d{2}:\d{2}(?::\d{2})?(?:\.\d+)?(?:Z|[+-]\d{2}:?\d{2})?)/i.exec( + msg + ); if (isoMatch) { const parsedTs = Date.parse(isoMatch[1]); if (Number.isFinite(parsedTs)) { @@ -1021,10 +1083,8 @@ export function classifyErrorText(errorText: unknown): RateLimitReasonValue { const configuredRule = matchErrorRuleByText(errorText); if (configuredRule?.reason) return configuredRule.reason; if (lower.includes("rate_limit")) return RateLimitReason.RATE_LIMIT_EXCEEDED; - if ( - lower.includes("resource exhausted") || - lower.includes("high demand") - ) return RateLimitReason.MODEL_CAPACITY; + if (lower.includes("resource exhausted") || lower.includes("high demand")) + return RateLimitReason.MODEL_CAPACITY; if ( lower.includes("unauthorized") || lower.includes("invalid api key") || @@ -1414,6 +1474,19 @@ export function checkFallbackError( } } + // Gemini-specific: use known published RPM/RPD limits to distinguish 429 types. + // Gemini returns the same error body for both, so we use per-model request + // counters to decide: if daily count >= RPD → quota_exhausted (midnight lockout); + // if minute count >= RPM → rate_limit_exceeded (exponential backoff). + if (provider === "gemini" && status === HTTP_STATUS.RATE_LIMITED && _model) { + if (isRpdExhausted(_model)) { + return buildRetryableFallback(RateLimitReason.QUOTA_EXHAUSTED); + } + if (isRpmExhausted(_model)) { + return buildRetryableFallback(RateLimitReason.RATE_LIMIT_EXCEEDED); + } + } + const configuredRule = isRateLimitStatus && !preserveQuota429 ? matchErrorRuleByStatus(status) diff --git a/open-sse/services/autoCombo/__tests__/autoCombo.test.ts b/open-sse/services/autoCombo/__tests__/autoCombo.test.ts index 7d1f9beec85..e2636f6e52e 100644 --- a/open-sse/services/autoCombo/__tests__/autoCombo.test.ts +++ b/open-sse/services/autoCombo/__tests__/autoCombo.test.ts @@ -2,10 +2,10 @@ * Unit tests for Auto-Combo Engine (Phase 5) */ -import { describe, it, expect, beforeEach } from "vitest"; +import { describe, it, expect, beforeEach, vi } from "vitest"; import { calculateFactors, calculateScore, DEFAULT_WEIGHTS, validateWeights } from "../scoring"; import type { ProviderCandidate, ScoringWeights } from "../scoring"; -import { getTaskFitness, getTaskTypes } from "../taskFitness"; +import { getTaskFitness, getTaskFitnessWithSource, getTaskTypes, getModelsDevTierFitness, invalidateFitnessCache } from "../taskFitness"; import { SelfHealingManager } from "../selfHealing"; import { MODE_PACKS, getModePack, getModePackNames } from "../modePacks"; import { getStrategy } from "../routerStrategy"; @@ -410,3 +410,164 @@ describe("LKGP Strategy", () => { expect(result.provider).toBe("openai"); }); }); + +describe("Task Fitness Resolution Chain", () => { + it("getTaskFitness should return static table score for known models", () => { + const score = getTaskFitness("claude-sonnet", "coding"); + expect(score).toBe(0.95); + }); + + it("getTaskFitness should return 0.5 for unknown models with no wildcard match", () => { + const score = getTaskFitness("unknown-model-xyz", "coding"); + expect(score).toBe(0.5); + }); + + it("getTaskFitness should apply wildcard boosts for model name patterns", () => { + const score = getTaskFitness("deepseek-coder-v2", "coding"); + expect(score).toBeGreaterThan(0.5); + }); + + it("getTaskFitness should apply thinking wildcard for planning tasks", () => { + const score = getTaskFitness("some-thinking-model", "planning"); + expect(score).toBeGreaterThan(0.5); + }); + + it("getTaskFitnessWithSource should return source='fitness_table' for known static models", () => { + const result = getTaskFitnessWithSource("claude-sonnet", "coding"); + expect(result).toEqual({ score: 0.95, source: "fitness_table" }); + }); + + it("getTaskFitnessWithSource should return source='wildcard_boost' for wildcard-matched models", () => { + const result = getTaskFitnessWithSource("fast-model", "coding"); + expect(result).toEqual({ score: expect.any(Number), source: "wildcard_boost" }); + }); + + it("getTaskTypes should return task types without 'default'", () => { + const types = getTaskTypes(); + expect(types).toContain("coding"); + expect(types).toContain("review"); + expect(types).toContain("planning"); + expect(types).not.toContain("default"); + }); + + it("unknown models should return 0.5 (default) when no DB or static entry exists", () => { + const score = getTaskFitness("completely-unknown-model-xyz-999", "coding"); + expect(score).toBe(0.5); + }); + + it("wildcard boosts still work for models containing 'coder'", () => { + const score = getTaskFitness("my-coder-pro", "coding"); + // Base 0.5 + coder boost 0.15 + code boost 0.1 = 0.75 + // "coder" contains "code", so both wildcard patterns match + expect(score).toBe(0.75); + }); + + it("wildcard boosts still work for models containing 'thinking'", () => { + const score = getTaskFitness("my-thinking-model", "planning"); + // Base 0.5 + thinking boost 0.1 = 0.6 + expect(score).toBe(0.6); + }); + + it("wildcard boosts still work for models containing 'thinking' for analysis tasks", () => { + const score = getTaskFitness("my-thinking-model", "analysis"); + // Base 0.5 + thinking boost 0.1 = 0.6 + expect(score).toBe(0.6); + }); + + it("wildcard boosts for 'code' pattern apply to coding tasks", () => { + const score = getTaskFitness("my-code-generator", "coding"); + // Base 0.5 + code boost 0.1 = 0.6 + expect(score).toBe(0.6); + }); + + it("wildcard boosts for 'fast' pattern apply to coding tasks", () => { + const score = getTaskFitness("my-fast-model", "coding"); + // Base 0.5 + fast boost 0.05 = 0.55 + expect(score).toBe(0.55); + }); + + it("getTaskFitnessWithSource returns 'wildcard_boost' for pattern-matched unknown models", () => { + const result = getTaskFitnessWithSource("my-coder-pro", "coding"); + expect(result.source).toBe("wildcard_boost"); + expect(result.score).toBeGreaterThan(0.5); + }); + + it("getTaskFitnessWithSource returns 'fitness_table' for statically known models", () => { + const result = getTaskFitnessWithSource("claude-sonnet", "review"); + expect(result.source).toBe("fitness_table"); + expect(result.score).toBe(0.92); + }); + + it("getTaskFitnessWithSource returns 'wildcard_boost' with 0.5 for unknown models with no pattern", () => { + const result = getTaskFitnessWithSource("totally-random-xyz", "coding"); + expect(result.source).toBe("wildcard_boost"); + expect(result.score).toBe(0.5); + }); +}); + +describe("Task Fitness DB Resolution Chain", () => { + // These tests verify that when DB is available, the resolution chain + // (user_override → arena_elo → models_dev_tier → static → wildcard) + // works correctly. Since the DB module is loaded lazily via require(), + // these tests cover the cases where DB is NOT available (graceful fallback). + + it("falls back to static FITNESS_TABLE when DB is not initialized", () => { + // In the test environment, DB is typically not initialized, + // so getTaskFitness should fall through to the static table + const score = getTaskFitness("claude-sonnet", "coding"); + // Static table has claude-sonnet → 0.95 for coding + expect(score).toBe(0.95); + }); + + it("falls back to static FITNESS_TABLE for review task type", () => { + const score = getTaskFitness("claude-opus", "review"); + // Static table has claude-opus → 0.95 for review + expect(score).toBe(0.95); + }); + + it("falls back to wildcard boosts when no static entry exists and DB unavailable", () => { + // "coder-unknown" has no static entry but matches "coder" wildcard + const score = getTaskFitness("coder-unknown", "coding"); + expect(score).toBeGreaterThan(0.5); + expect(score).toBeLessThanOrEqual(1.0); + }); + + it("getModelsDevTierFitness returns null when DB is not initialized", () => { + // Without a running DB, this should return null gracefully + const score = getModelsDevTierFitness("claude-sonnet", "coding"); + // Either null (no capabilities data) or a number from DB if DB happens to be up + if (score !== null) { + expect(score).toBeGreaterThanOrEqual(0); + expect(score).toBeLessThanOrEqual(1); + } + }); + + it("invalidateFitnessCache does not throw", () => { + expect(() => invalidateFitnessCache()).not.toThrow(); + }); + + it("resolution chain: static table takes priority over wildcard for known models", () => { + // "claude-sonnet" is in the static table with coding=0.95 + // It does NOT match "coder" wildcard because the static table is checked first + const score = getTaskFitness("claude-sonnet", "coding"); + expect(score).toBe(0.95); // From static table, NOT wildcard + }); + + it("getTaskFitnessWithSource identifies fitness_table as source for known models", () => { + const result = getTaskFitnessWithSource("gpt-4o", "coding"); + expect(result.source).toBe("fitness_table"); + expect(result.score).toBe(0.9); + }); + + it("case insensitivity: model names are lowercased before lookup", () => { + const upperScore = getTaskFitness("CLAUDE-SONNET", "coding"); + const lowerScore = getTaskFitness("claude-sonnet", "coding"); + expect(upperScore).toBe(lowerScore); + }); + + it("case insensitivity: task types are lowercased before lookup", () => { + const upperScore = getTaskFitness("claude-sonnet", "CODING"); + const lowerScore = getTaskFitness("claude-sonnet", "coding"); + expect(upperScore).toBe(lowerScore); + }); +}); diff --git a/open-sse/services/autoCombo/taskFitness.ts b/open-sse/services/autoCombo/taskFitness.ts index 123d479b1c6..404a476089d 100644 --- a/open-sse/services/autoCombo/taskFitness.ts +++ b/open-sse/services/autoCombo/taskFitness.ts @@ -3,8 +3,24 @@ * * Maps model patterns × task types → fitness score [0..1]. * Supports wildcards and prefix matching. + * + * Resolution chain (highest → lowest priority): + * 1. User override — DB `model_intelligence` where source='user_override' + * 2. Arena ELO — DB `model_intelligence` where source='arena_elo' + * 3. Models.dev tier — derived from `model_capabilities` table capability data + * 4. Static FITNESS_TABLE — existing hardcoded lookup (current behavior) + * 5. Wildcard boosts — existing pattern matching boosts (current behavior) */ +// ─── Static fitness table (unchanged, fallback layer 4) ───────────────── + +import { getDbInstance } from "../../../src/lib/db/core.ts"; +import { + getModelIntelligenceBySource, + setUserFitnessOverrideEntry, + deleteUserFitnessOverrideEntry, +} from "../../../src/lib/db/modelIntelligence.ts"; + const FITNESS_TABLE: Record> = { coding: { "claude-sonnet": 0.95, @@ -131,34 +147,274 @@ const WILDCARD_BOOSTS: Array<{ pattern: string; taskType: string; boost: number { pattern: "thinking", taskType: "analysis", boost: 0.1 }, ]; +// ─── Models.dev tier → task fitness mapping (resolution layer 3) ──────── + /** - * Get task fitness score for a model × taskType combination. - * Returns 0.5 (neutral) if no mapping found. + * Intelligence tier derived from models.dev capability data. + * Tier assignment rules: + * - `reasoning === true` → "premium" + * - `tool_call === true && context >= 128000` → "standard" + * - `tool_call === true` → "fast" + * - everything else → "budget" */ -export function getTaskFitness(model: string, taskType: string): number { +const TIER_TASK_FITNESS: Record> = { + premium: { + coding: 0.92, + review: 0.93, + planning: 0.94, + analysis: 0.95, + debugging: 0.9, + documentation: 0.88, + default: 0.85, + }, + standard: { + coding: 0.85, + review: 0.84, + planning: 0.85, + analysis: 0.85, + debugging: 0.82, + documentation: 0.85, + default: 0.78, + }, + fast: { + coding: 0.78, + review: 0.72, + planning: 0.7, + analysis: 0.72, + debugging: 0.75, + documentation: 0.8, + default: 0.72, + }, + budget: { + coding: 0.65, + review: 0.6, + planning: 0.55, + analysis: 0.58, + debugging: 0.6, + documentation: 0.7, + default: 0.55, + }, +}; +// ─── DB access helpers ────────────────────────────────────────────────── + +const _intelligenceCache = new Map(); + +function queryModelIntelligence( + model: string, + category: string, + source: string, +): number | null { + const cacheKey = `${model}:${category}:${source}`; + if (_intelligenceCache.has(cacheKey)) { + return _intelligenceCache.get(cacheKey)!; + } + + try { + const entry = getModelIntelligenceBySource(model, source, category); + if (entry) { + _intelligenceCache.set(cacheKey, entry.score); + return entry.score; + } + return null; + } catch { + return null; + } +} + +// ─── Models.dev capability → tier → fitness resolution ────────────────── + +let _capabilitiesCache: Record | null = null; + +interface ModelCapRow { + tool_call: boolean | null; + reasoning: boolean | null; + limit_context: number | null; +} + +function deriveTierFromCapabilities(cap: ModelCapRow): string { + if (cap.reasoning === true) return "premium"; + if (cap.tool_call === true && (cap.limit_context ?? 0) >= 128000) + return "standard"; + if (cap.tool_call === true) return "fast"; + return "budget"; +} + +function loadModelCapabilities(): Record | null { + if (_capabilitiesCache) return _capabilitiesCache; + + try { + const db = getDbInstance(); + const rows = db.prepare("SELECT * FROM model_capabilities").all() as Record< + string, + unknown + >[]; + const cache: Record = {}; + + for (const row of rows) { + const modelId = typeof row.model_id === "string" ? row.model_id : ""; + if (!modelId) continue; + + cache[modelId.toLowerCase()] = { + tool_call: + row.tool_call === true || row.tool_call === 1 + ? true + : row.tool_call === false || row.tool_call === 0 + ? false + : null, + reasoning: + row.reasoning === true || row.reasoning === 1 + ? true + : row.reasoning === false || row.reasoning === 0 + ? false + : null, + limit_context: + typeof row.limit_context === "number" ? row.limit_context : null, + }; + } + + _capabilitiesCache = cache; + return cache; + } catch { + return null; + } +} + +export function getModelsDevTierFitness( + model: string, + taskType: string, +): number | null { const normalizedModel = model.toLowerCase(); const normalizedTask = taskType.toLowerCase(); - const table = FITNESS_TABLE[normalizedTask] || FITNESS_TABLE.default; - // Direct match + const dbScore = queryModelIntelligence( + normalizedModel, + normalizedTask, + "models_dev_tier", + ); + if (dbScore !== null) return dbScore; + + const caps = loadModelCapabilities(); + if (!caps) return null; + + const capRow = caps[normalizedModel]; + if (!capRow) return null; + + const tier = deriveTierFromCapabilities(capRow); + const tierScores = TIER_TASK_FITNESS[tier]; + if (!tierScores) return null; + + return tierScores[normalizedTask] ?? tierScores.default ?? null; +} + +// ─── Resolution chain ─────────────────────────────────────────────────── + +function lookupStaticFitnessTable( + normalizedModel: string, + normalizedTask: string, +): number | null { + const table = FITNESS_TABLE[normalizedTask] || FITNESS_TABLE.default; for (const [pattern, score] of Object.entries(table)) { if (normalizedModel.includes(pattern)) return score; } + return null; +} - // Wildcard boost +function lookupWildcardBoosts( + normalizedModel: string, + normalizedTask: string, +): number { let baseScore = 0.5; for (const wc of WILDCARD_BOOSTS) { if (normalizedModel.includes(wc.pattern) && normalizedTask === wc.taskType) { baseScore += wc.boost; } } - return Math.min(1.0, baseScore); } -/** - * Get all task types available. - */ +export function getTaskFitness(model: string, taskType: string): number { + return getTaskFitnessWithSource(model, taskType).score; +} + +export function getTaskFitnessWithSource( + model: string, + taskType: string, +): { score: number; source: string } { + const normalizedModel = model.toLowerCase(); + const normalizedTask = taskType.toLowerCase(); + + const userOverride = queryModelIntelligence( + normalizedModel, + normalizedTask, + "user_override", + ); + if (userOverride !== null) { + return { score: userOverride, source: "user_override" }; + } + + const arenaElo = queryModelIntelligence( + normalizedModel, + normalizedTask, + "arena_elo", + ); + if (arenaElo !== null) { + return { score: arenaElo, source: "arena_elo" }; + } + + const tierScore = getModelsDevTierFitness(normalizedModel, normalizedTask); + if (tierScore !== null) { + return { score: tierScore, source: "models_dev_tier" }; + } + + const staticScore = lookupStaticFitnessTable( + normalizedModel, + normalizedTask, + ); + if (staticScore !== null) { + return { score: staticScore, source: "fitness_table" }; + } + + return { score: lookupWildcardBoosts(normalizedModel, normalizedTask), source: "wildcard_boost" }; +} + +export function setUserFitnessOverride( + model: string, + category: string, + score: number, +): void { + try { + setUserFitnessOverrideEntry( + model.toLowerCase(), + category.toLowerCase(), + score, + ); + invalidateFitnessCache(); + } catch (err) { + throw new Error( + `Failed to set user fitness override for ${model}/${category}: ${err instanceof Error ? err.message : String(err)}`, + ); + } +} + +export function clearUserFitnessOverride( + model: string, + category: string, +): void { + try { + deleteUserFitnessOverrideEntry(model.toLowerCase(), category.toLowerCase()); + invalidateFitnessCache(); + } catch (err) { + throw new Error( + `Failed to clear user fitness override for ${model}/${category}: ${err instanceof Error ? err.message : String(err)}`, + ); + } +} + export function getTaskTypes(): string[] { return Object.keys(FITNESS_TABLE).filter((k) => k !== "default"); } + +export function invalidateFitnessCache(): void { + _capabilitiesCache = null; + _intelligenceCache.clear(); +} diff --git a/open-sse/services/combo.ts b/open-sse/services/combo.ts index 5095207bcc1..0242df2dffa 100644 --- a/open-sse/services/combo.ts +++ b/open-sse/services/combo.ts @@ -1,4746 +1 @@ -/** - * Shared combo (model combo) handling with fallback support - * Supports: priority, weighted, round-robin, random, least-used, cost-optimized, - * reset-aware, reset-window, strict-random, auto, fill-first, p2c, lkgp, - * context-optimized, and context-relay strategies - */ - -import { - checkFallbackError, - classifyErrorText, - formatRetryAfter, - getRuntimeProviderProfile, - recordProviderFailure, - isProviderFailureCode, - isProviderExhaustedReason, - type ProviderProfile, -} from "./accountFallback.ts"; -import { FETCH_TIMEOUT_MS, RateLimitReason } from "../config/constants.ts"; -import { errorResponse, unavailableResponse } from "../utils/error.ts"; -import { clamp01 } from "../utils/number.ts"; -import { - recordComboIntent, - recordComboRequest, - recordComboShadowRequest, - getComboMetrics, -} from "./comboMetrics.ts"; -import { - resolveComboConfig, - getDefaultComboConfig, - resolveComboTargetTimeoutMs, - PRE_SCREEN_CONCURRENCY, -} from "./comboConfig.ts"; -import { - maybeGenerateHandoff, - resolveContextRelayConfig, - maybeGenerateUniversalHandoff, - injectUniversalHandoffBody, - resolveUniversalHandoffConfig, - SKIP_UNIVERSAL_HANDOFF_FLAG, - type MessageLike, -} from "./contextHandoff.ts"; -import { - recordSessionModelUsage, - getLastSessionModel, - getHandoff, -} from "../../src/lib/db/contextHandoffs.ts"; -import { fetchCodexQuota } from "./codexQuotaFetcher.ts"; -import { getQuotaFetcher } from "./quotaPreflight.ts"; -import * as semaphore from "./rateLimitSemaphore.ts"; -import { getCircuitBreaker } from "../../src/shared/utils/circuitBreaker"; -import { fisherYatesShuffle, getNextFromDeck } from "../../src/shared/utils/shuffleDeck"; -import { parseModel } from "./model.ts"; -import { applyComboAgentMiddleware } from "./comboAgentMiddleware.ts"; -import { checkCredentialGate, logCredentialSkip } from "./credentialGate.ts"; -import { emit } from "../../src/lib/events/eventBus"; -import { notifyWebhookEvent } from "../../src/lib/webhookDispatcher"; -import { - classifyWithConfig, - DEFAULT_INTENT_CONFIG, - type IntentClassifierConfig, -} from "./intentClassifier.ts"; -import { selectProvider as selectAutoProvider } from "./autoCombo/engine.ts"; -import { selectWithStrategy, type SlaRoutingPolicy } from "./autoCombo/routerStrategy.ts"; -import { getTaskFitness } from "./autoCombo/taskFitness.ts"; -import { parseAutoPrefix } from "./autoCombo/autoPrefix.ts"; -import { handlePipelineCombo, buildPipelineResponse } from "./autoCombo/pipelineRouter.ts"; -import { - calculateFactors, - calculateScore, - DEFAULT_WEIGHTS, - type ProviderCandidate, - type ScoringWeights, -} from "./autoCombo/scoring.ts"; -import { - getResolvedModelCapabilities, - supportsReasoning, - supportsToolCalling, -} from "./modelCapabilities.ts"; -import { estimateTokens } from "./contextManager.ts"; -import { getReasoningTokens } from "../../src/lib/usage/tokenAccounting.ts"; -import { getSessionConnection } from "./sessionManager.ts"; -import { orderTargetsByEvalScores } from "./evalRouting.ts"; -import { generateRoutingHints } from "./manifestAdapter"; -import type { RoutingHint } from "./manifestAdapter"; -import type { CompressionMode } from "./compression/types.ts"; -import { getModelContextLimit } from "../../src/lib/modelCapabilities"; -import { getProviderConnections } from "../../src/lib/db/providers"; -import { getProviderModels } from "../config/providerModels.ts"; -import { - getComboModelString, - getComboStepTarget, - getComboStepWeight, - normalizeComboStep, -} from "../../src/lib/combos/steps.ts"; -import { - getConnectionRoutingTags, - matchesRoutingTags, - resolveRequestRoutingTags, - type RoutingTagMatchMode, -} from "../../src/domain/tagRouter.ts"; -import { normalizeRoutingStrategy } from "../../src/shared/constants/routingStrategies.ts"; -import { - isProviderInCooldown, - recordProviderCooldown, - recordProviderSuccess, -} from "./providerCooldownTracker.ts"; -import { - resolveResilienceSettings, - type ResilienceSettings, -} from "../../src/lib/resilience/settings"; - -// Status codes that should mark round-robin target semaphores as cooling down. -const TRANSIENT_FOR_SEMAPHORE = [429, 502, 503, 504]; -// Patterns that signal all accounts for a provider are rate-limited / exhausted. -// Used to detect 503 responses from handleNoCredentials so combo can fallback. -const ALL_ACCOUNTS_RATE_LIMITED_PATTERNS = [/unavailable/i, /service temporarily unavailable/i]; - -function isAllAccountsRateLimitedResponse( - status: number, - contentType: string | null, - errorText: string -): boolean { - if (status !== 503) return false; - if (!contentType?.includes("application/json")) return false; - return ALL_ACCOUNTS_RATE_LIMITED_PATTERNS.some((p) => p.test(errorText)); -} - -// #1731v2 guard: a provider circuit-breaker-open response (503 + `X-OmniRoute-Provider-Breaker` -// header / `provider_circuit_open` error code, see providerCircuitOpenResponse) is an OmniRoute -// resilience signal, NOT a per-connection upstream failure. It must keep being treated as an -// ordinary target failure (try the next target, including same-provider ones) — so it must NOT -// poison exhaustedConnections/exhaustedProviders, otherwise remaining same-provider targets get -// wrongly skipped while the breaker is open. -function isProviderCircuitOpenResult( - result: { headers?: Headers | null; status?: number }, - errorText: string -): boolean { - const breakerHeader = result.headers?.get?.("x-omniroute-provider-breaker"); - if (typeof breakerHeader === "string" && breakerHeader.toLowerCase() === "open") return true; - return /provider_circuit_open/i.test(errorText); -} - -const MAX_COMBO_DEPTH = 3; -const MAX_FALLBACK_WAIT_MS = 5000; -const MAX_GLOBAL_ATTEMPTS = 30; - -function resolveDelayMs(value: unknown, fallback: number): number { - const numericValue = Number(value); - if (!Number.isFinite(numericValue) || numericValue < 0) return fallback; - return numericValue; -} - -function comboModelNotFoundResponse(message: string) { - return errorResponse(404, message); -} - -// Bootstrap defaults from ClawRouter benchmark (used when no local latency history exists yet) -const DEFAULT_MODEL_P95_MS: Record = { - "grok-4-fast-non-reasoning": 1143, - "grok-4-1-fast-non-reasoning": 1244, - "gemini-2.5-flash": 1238, - "kimi-k2.5": 1646, - "gpt-4o-mini": 2764, - "claude-sonnet-4.6": 4000, - "claude-opus-4.6": 6000, - "deepseek-chat": 2000, -}; -const MIN_HISTORY_SAMPLES = 10; -// Assumed fraction of tokens that are output when blending input+output prices -// for auto-combo cost scoring. 0.4 = 40% output, 60% input. -// Matches the example in GitHub issue #1812 (e.g. o3-like model: $3 input/$15 output). -const OUTPUT_TOKEN_RATIO = 0.4; -const RESET_AWARE_SESSION_WINDOW_MS = 5 * 60 * 60 * 1000; -const RESET_AWARE_WEEKLY_WINDOW_MS = 7 * 24 * 60 * 60 * 1000; -const RESET_AWARE_SESSION_REMAINING_WEIGHT = 0.45; -const RESET_AWARE_SESSION_RESET_PRESSURE_WEIGHT = 0.55; -const RESET_AWARE_WEEKLY_REMAINING_WEIGHT = 0.25; -const RESET_AWARE_WEEKLY_RESET_PRESSURE_WEIGHT = 0.75; -const RESET_AWARE_CONNECTION_CACHE_TTL_MS = 30_000; -const RESET_AWARE_QUOTA_FETCH_CONCURRENCY = 5; -const RESET_AWARE_DEFAULTS = { - sessionWeight: 0.35, - weeklyWeight: 0.65, - tieBandPercent: 5, - exhaustionGuardPercent: 10, -}; -const RESET_WINDOW_DEFAULT_TIE_BAND_MS = 60_000; - -// Quota Share soft-policy deprioritization factor (B17). -// When a candidate has quotaSoftPenalty === true, its auto-combo score is -// multiplied by this factor so over-quota-soft keys are de-prioritized -// without being fully blocked (that is done by "hard" policy). -// Override via QUOTA_SOFT_DEPRIORITIZE_FACTOR env var (range 0..1, default 0.7). -export const QUOTA_SOFT_DEPRIORITIZE_FACTOR = Number( - process.env.QUOTA_SOFT_DEPRIORITIZE_FACTOR ?? "0.7" -); - -// G2: Module-level registry of active combo execution candidates. -// Maps executionKey → Map. -// Populated by buildAutoCandidates registrations; cleaned up after each execution. -// This allows chatCore.ts to mark a candidate's quotaSoftPenalty flag so that -// subsequent scoring iterations (auto-combo fallback) deprioritize it. -const _activeExecutionCandidates = new Map>(); - -/** - * Mark a specific candidate (by comboExecutionKey + stepId) with soft quota penalty. - * Called from chatCore.ts when enforceQuotaShare returns a "soft deprioritize" decision. - * The flag is read on subsequent auto-combo scoring iterations (fallback chain) - * within the same combo execution via scoreAutoTargets → QUOTA_SOFT_DEPRIORITIZE_FACTOR. - * - * Guards: - * - null executionKey or stepId → no-op (non-combo or context not available). - * - unknown executionKey → no-op (candidate not yet registered or already cleaned up). - * - Idempotent: calling twice with the same (key, stepId, true) is safe. - */ -export function setCandidateQuotaSoftPenalty( - comboExecutionKey: string | null, - comboStepId: string | null, - penalty: boolean -): void { - if (!comboExecutionKey || !comboStepId) return; - const byStep = _activeExecutionCandidates.get(comboExecutionKey); - if (!byStep) return; - const candidate = byStep.get(comboStepId); - if (candidate) { - candidate.quotaSoftPenalty = penalty; - } -} - -/** - * Register candidates for a combo execution so setCandidateQuotaSoftPenalty can - * locate them by (executionKey, stepId). - * Each candidate object is stored by reference — mutations via setCandidateQuotaSoftPenalty - * propagate back to the original candidate array used by scoreAutoTargets. - * @internal — not exported; only called within combo.ts by buildAutoCandidates callers. - */ -function _registerExecutionCandidates( - candidates: Array<{ executionKey: string; stepId: string; quotaSoftPenalty?: boolean }> -): void { - for (const candidate of candidates) { - if (!candidate.executionKey) continue; - let byStep = _activeExecutionCandidates.get(candidate.executionKey); - if (!byStep) { - byStep = new Map(); - _activeExecutionCandidates.set(candidate.executionKey, byStep); - } - byStep.set(candidate.stepId, candidate); - } -} - -/** - * Unregister all candidates for a given execution key once the execution completes. - * Prevents unbounded memory growth. - * @internal — not exported; called after each handleComboChat iteration. - */ -function _unregisterExecutionCandidates(executionKeys: string[]): void { - for (const key of executionKeys) { - _activeExecutionCandidates.delete(key); - } -} - -const RESET_WINDOW_NAMES = ["weekly", "session", "monthly"] as const; -type ResetWindowName = (typeof RESET_WINDOW_NAMES)[number]; -type QuotaFetchCacheConfig = { - quotaCacheTtlMs: number; - quotaCacheMaxStaleMs: number; -}; -type ResetWindowConfig = ReturnType; -type ComboRetryAfter = string | number | Date; -type ComboErrorBody = { - error?: { code?: string | null; message?: string | null } | string; - message?: string | null; - retryAfter?: ComboRetryAfter | null; -} | null; - -type ComboLike = { - id?: string; - name: string; - strategy?: string | null; - models: unknown[]; - config?: Record | null; - autoConfig?: Record | null; - context_cache_protection?: boolean | number; - system_message?: string | null; - [key: string]: unknown; -}; - -type ComboInput = ComboLike | Record; - -type ComboCollectionLike = ComboInput[] | { combos?: ComboInput[] } | null | undefined; - -type ComboLogger = { - info: (...args: unknown[]) => void; - warn: (...args: unknown[]) => void; - error?: (...args: unknown[]) => void; - debug: (...args: unknown[]) => void; -}; - -export type SingleModelTarget = - | (ResolvedComboTarget & { - allowRateLimitedConnection?: boolean; - modelAbortSignal?: AbortSignal | null; - }) - | { modelAbortSignal: AbortSignal }; - -type HandleSingleModel = ( - body: Record, - modelStr: string, - target?: SingleModelTarget -) => Promise; - -type IsModelAvailable = ( - modelStr: string, - target?: ResolvedComboTarget & { allowRateLimitedConnection?: boolean } -) => Promise | boolean; - -type ComboRelayOptions = { - sessionId?: string | null; - config?: Record | null; - [key: string]: unknown; -}; - -type HandleComboChatOptions = { - body: Record; - combo: ComboLike; - handleSingleModel: HandleSingleModel; - isModelAvailable?: IsModelAvailable; - log: ComboLogger; - settings?: Record | null; - allCombos?: ComboCollectionLike; - relayOptions?: ComboRelayOptions | null; - signal?: AbortSignal | null; - apiKeyAllowedConnections?: string[] | null; -}; - -type HandleRoundRobinOptions = Omit< - HandleComboChatOptions, - "relayOptions" | "apiKeyAllowedConnections" ->; - -type HistoricalLatencyStatsEntry = { - totalRequests?: number; - p95LatencyMs?: number; - latencyStdDev?: number; - successRate?: number; -}; - -type AutoProviderCandidate = ProviderCandidate & { - stepId: string; - executionKey: string; - modelStr: string; - /** - * When true, this candidate's auto-combo score is multiplied by - * QUOTA_SOFT_DEPRIORITIZE_FACTOR (B17 soft-policy penalty). - * Set externally when enforceQuotaShare returns deprioritize=true - * for the key routed through this target's connectionId. - */ - quotaSoftPenalty?: boolean; -}; - -function toRetryAfterDisplayValue(value: ComboRetryAfter): string | Date { - if (typeof value !== "number") return value; - if (value > 0 && value < 1_000_000_000) { - return new Date(Date.now() + value * 1000); - } - return new Date(value); -} - -export type ResolvedComboTarget = { - kind: "model"; - stepId: string; - executionKey: string; - modelStr: string; - provider: string; - providerId: string | null; - connectionId: string | null; - allowedConnectionIds?: string[] | null; - weight: number; - label: string | null; - failoverBeforeRetry?: unknown; - trafficType?: "production" | "shadow"; -}; - -type ShadowRoutingConfig = { - enabled: boolean; - targets: unknown[]; - sampleRate: number; - maxTargets: number; - timeoutMs: number; -}; - -type ComboRuntimeStep = - | ResolvedComboTarget - | { - kind: "combo-ref"; - stepId: string; - executionKey: string; - comboName: string; - weight: number; - label: string | null; - }; - -function isRecord(value: unknown): value is Record { - return !!value && typeof value === "object" && !Array.isArray(value); -} - -function toTrimmedString(value: unknown): string | null { - return typeof value === "string" && value.trim().length > 0 ? value.trim() : null; -} - -function toComboLike(combo: ComboInput): ComboLike { - return { - ...combo, - id: toTrimmedString(combo.id) || undefined, - name: toTrimmedString(combo.name) || "", - models: Array.isArray(combo.models) ? combo.models : [], - config: isRecord(combo.config) ? combo.config : null, - autoConfig: isRecord(combo.autoConfig) ? combo.autoConfig : null, - context_cache_protection: - typeof combo.context_cache_protection === "boolean" || - typeof combo.context_cache_protection === "number" - ? combo.context_cache_protection - : undefined, - system_message: typeof combo.system_message === "string" ? combo.system_message : null, - }; -} - -function getCombosArray(allCombos: ComboCollectionLike): ComboLike[] { - const combos = Array.isArray(allCombos) ? allCombos : allCombos?.combos || []; - return combos.map((combo) => toComboLike(combo)); -} - -/** - * Validate that a successful (HTTP 200) non-streaming response actually contains - * meaningful content. Returns { valid: true } or { valid: false, reason }. - * - * Only inspects non-streaming JSON responses — streaming responses are passed through - * because buffering the full stream would defeat the purpose of streaming. - * - * Checks: - * 1. Body is valid JSON - * 2. Has at least one choice with non-empty content or tool_calls - */ -export async function validateResponseQuality( - response: Response, - isStreaming: boolean, - log: { warn?: (...args: unknown[]) => void } -): Promise<{ valid: boolean; reason?: string; clonedResponse?: Response }> { - if (isStreaming) return { valid: true }; - - const contentType = response.headers.get("content-type") || ""; - if (!contentType.includes("application/json") && !contentType.includes("text/")) { - return { valid: true }; - } - - let cloned: Response; - try { - cloned = response.clone(); - } catch { - return { valid: true }; - } - - let text: string; - try { - text = await cloned.text(); - } catch { - return { valid: true }; - } - - if (!text || text.trim().length === 0) { - return { valid: false, reason: "empty response body" }; - } - - let json: Record; - try { - json = JSON.parse(text); - } catch { - if (text.startsWith("data:") || text.startsWith("event:")) return { valid: true }; - return { valid: false, reason: "response is not valid JSON" }; - } - - const choices = json?.choices; - if (!Array.isArray(choices) || choices.length === 0) { - if (json?.output || json?.result || json?.data || json?.response) return { valid: true }; - if (json?.error) { - const err = json.error as Record; - return { - valid: false, - reason: `upstream error in 200 body: ${err?.message || JSON.stringify(json.error).substring(0, 200)}`, - }; - } - return { valid: true }; - } - - const firstChoice = choices[0]; - const message = firstChoice?.message || firstChoice?.delta; - if (!message) { - return { valid: false, reason: "choice has no message object" }; - } - - const content = message.content; - const toolCalls = message.tool_calls; - // Issue #2341: Reasoning models (Kimi-K2.5-TEE, GLM-5-TEE, etc.) emit their - // output in `reasoning_content` (or `reasoning`) with `content: null`. The - // validator used to flag those as empty and trigger a false-positive 502 - // fallback. Count a non-empty reasoning_content as valid output too. - const reasoningContent = message.reasoning_content ?? message.reasoning; - const hasReasoningContent = - typeof reasoningContent === "string" && reasoningContent.trim().length > 0; - const hasContent = - (content !== null && content !== undefined && content !== "") || hasReasoningContent; - const hasToolCalls = Array.isArray(toolCalls) && toolCalls.length > 0; - - if (!hasContent && !hasToolCalls) { - return { valid: false, reason: "empty content and no tool_calls in response" }; - } - - // Issue #3587: Reasoning models (deepseek-v4-flash, nemotron, etc.) may consume - // ALL max_tokens for reasoning_tokens, leaving content empty. When content is - // empty but reasoning_content exists, and usage shows reasoning consumed nearly - // all completion tokens, treat as invalid so the combo loop retries with more - // tokens or falls back to a non-reasoning model. - const contentIsEmpty = content === null || content === undefined || content === ""; - if (contentIsEmpty && hasReasoningContent && !hasToolCalls) { - const usage = json?.usage as Record | undefined; - if (usage) { - const completionTokens = Number(usage.completion_tokens) || 0; - const reasoningTokens = getReasoningTokens(usage); - // If reasoning consumed 90%+ of completion tokens, the model ran out of - // budget before producing any content output. - if (completionTokens > 0 && reasoningTokens >= completionTokens * 0.9) { - return { - valid: false, - reason: `reasoning consumed ${reasoningTokens}/${completionTokens} tokens — no content output`, - }; - } - } - } - - return { - valid: true, - clonedResponse: new Response(text, { - status: response.status, - statusText: response.statusText, - headers: response.headers, - }), - }; -} - -// In-memory atomic counter per combo for round-robin distribution -// Resets on server restart (by design — no stale state) -// Eviction limits to prevent unbounded memory growth -const MAX_RR_COUNTERS = 500; -const MAX_RESET_AWARE_CACHE = 200; - -const rrCounters = new Map(); - -const resetAwareConnectionCache = new Map< - string, - { fetchedAt: number; connections: Array> } ->(); -const resetAwareQuotaCache = new Map< - string, - { fetchedAt: number; quota: unknown; refreshPromise: Promise | null } ->(); - -/** - * Normalize a model entry to { model, weight } - * Supports both legacy string format and new object format - */ -function normalizeModelEntry(entry: unknown): { model: string; weight: number } { - return { - model: getComboStepTarget(entry) || "", - weight: getComboStepWeight(entry), - }; -} - -function getTargetProvider(modelStr: string, providerId?: string | null): string { - const parsed = parseModel(modelStr); - return providerId || parsed.provider || parsed.providerAlias || "unknown"; -} - -function isStreamReadinessFailureErrorBody(errorBody: unknown): boolean { - if (!errorBody || typeof errorBody !== "object") return false; - const error = (errorBody as Record).error; - if (!error || typeof error !== "object") return false; - const code = (error as Record).code; - return code === "STREAM_READINESS_TIMEOUT" || code === "STREAM_EARLY_EOF"; -} - -/** - * A local per-API-key token-limit breach surfaces as a 429 tagged with - * errorCode "TOKEN_LIMIT_EXCEEDED" (see chatCore.ts Tier 2 early return). This - * is NOT an upstream rate limit, so the combo loop must not cool the shared - * account/provider, must not add it to transientRateLimitedProviders, and must - * not retry it transiently — it propagates to the client as a terminal 429. - */ -function isTokenLimitBreachErrorBody(errorBody: unknown): boolean { - if (!errorBody || typeof errorBody !== "object") return false; - const error = (errorBody as Record).error; - if (!error || typeof error !== "object") return false; - return (error as Record).code === "TOKEN_LIMIT_EXCEEDED"; -} - -function toRecordedTarget(target: ResolvedComboTarget) { - return { - executionKey: target.executionKey, - stepId: target.stepId, - provider: target.provider, - providerId: target.providerId, - connectionId: target.connectionId, - label: target.label, - }; -} - -function normalizeShadowRoutingConfig(config: Record): ShadowRoutingConfig { - const raw = isRecord(config.shadowRouting) ? config.shadowRouting : {}; - const sampleRate = Number(raw.sampleRate ?? 1); - const maxTargets = Number(raw.maxTargets ?? 2); - const timeoutMs = Number(raw.timeoutMs ?? 30000); - return { - enabled: raw.enabled === true, - targets: Array.isArray(raw.targets) ? raw.targets : [], - sampleRate: Number.isFinite(sampleRate) ? Math.max(0, Math.min(1, sampleRate)) : 1, - maxTargets: Number.isFinite(maxTargets) ? Math.max(1, Math.min(10, Math.floor(maxTargets))) : 2, - timeoutMs: Number.isFinite(timeoutMs) - ? Math.max(1000, Math.min(120000, Math.floor(timeoutMs))) - : 30000, - }; -} - -function resolveShadowTargets( - combo: ComboLike, - config: Record, - allCombos: ComboCollectionLike -): ResolvedComboTarget[] { - const shadowConfig = normalizeShadowRoutingConfig(config); - if (!shadowConfig.enabled || shadowConfig.targets.length === 0) return []; - if (shadowConfig.sampleRate <= 0 || Math.random() > shadowConfig.sampleRate) return []; - - const shadowCombo: ComboLike = { - ...combo, - name: `${combo.name}:shadow`, - models: shadowConfig.targets, - }; - return resolveNestedComboTargets(shadowCombo, allCombos, new Set([combo.name]), 0, ["shadow"]) - .slice(0, shadowConfig.maxTargets) - .map((target) => ({ - ...target, - trafficType: "shadow" as const, - })); -} - -async function drainShadowResponse(response: Response): Promise { - try { - if (!response.body) return; - await response.arrayBuffer(); - } catch { - // Shadow draining is best-effort and must never affect the production response. - } -} - -function withTimeout(promise: Promise, timeoutMs: number): Promise { - return new Promise((resolve, reject) => { - const timer = setTimeout(() => reject(new Error("Shadow route timed out")), timeoutMs); - promise.then( - (value) => { - clearTimeout(timer); - resolve(value); - }, - (error) => { - clearTimeout(timer); - reject(error); - } - ); - }); -} - -function cloneRequestBodyForShadowRouting(body: Record): Record { - if (typeof structuredClone === "function") { - return structuredClone(body) as Record; - } - - return JSON.parse(JSON.stringify(body)) as Record; -} - -function scheduleShadowRouting( - combo: ComboLike, - config: Record, - body: Record, - targets: ResolvedComboTarget[], - handleSingleModel: HandleSingleModel, - isModelAvailable: IsModelAvailable | undefined, - strategy: string, - log: ComboLogger -): void { - if (targets.length === 0) return; - const shadowConfig = normalizeShadowRoutingConfig(config); - let shadowBaseBody: Record; - try { - shadowBaseBody = cloneRequestBodyForShadowRouting(body); - } catch (error) { - log.warn("COMBO", "Shadow routing skipped: failed to clone request body", { - error: error instanceof Error ? error.message : String(error), - }); - return; - } - const run = async () => { - await Promise.all( - targets.map(async (target) => { - const startedAt = Date.now(); - try { - const shadowBody = { - ...cloneRequestBodyForShadowRouting(shadowBaseBody), - model: target.modelStr, - stream: false, - }; - if (isModelAvailable) { - const available = await isModelAvailable(target.modelStr, target); - if (!available) { - recordComboShadowRequest(combo.name, target.modelStr, { - success: false, - latencyMs: Date.now() - startedAt, - target: toRecordedTarget(target), - }); - log.info("COMBO", `Shadow target skipped (unavailable): ${target.modelStr}`); - return; - } - } - - const response = await withTimeout( - handleSingleModel(shadowBody, target.modelStr, { - ...target, - failoverBeforeRetry: true, - trafficType: "shadow", - }), - shadowConfig.timeoutMs - ); - await drainShadowResponse(response.clone()); - recordComboShadowRequest(combo.name, target.modelStr, { - success: response.ok, - latencyMs: Date.now() - startedAt, - target: toRecordedTarget(target), - }); - log.info( - "COMBO", - `Shadow target ${target.modelStr} completed with status ${response.status} (${strategy})` - ); - } catch (error) { - recordComboShadowRequest(combo.name, target.modelStr, { - success: false, - latencyMs: Date.now() - startedAt, - target: toRecordedTarget(target), - }); - log.warn("COMBO", `Shadow target ${target.modelStr} failed`, { - error: error instanceof Error ? error.message : String(error), - }); - } - }) - ); - }; - - setTimeout(() => void run(), 0); -} - -function buildExecutionKey(path: string[], stepId: string): string { - return [...path, stepId].join(">"); -} - -function normalizeRuntimeStep( - entry: unknown, - comboName: string, - index: number, - allCombos: ComboCollectionLike, - path: string[] = [] -): ComboRuntimeStep | null { - const step = normalizeComboStep(entry, { - comboName, - index, - allCombos, - }); - if (!step) return null; - - const executionKey = buildExecutionKey(path, step.id); - const label = typeof step.label === "string" ? step.label : null; - const weight = step.weight || 0; - - if (step.kind === "combo-ref") { - return { - kind: "combo-ref", - stepId: step.id, - executionKey, - comboName: step.comboName, - weight, - label, - }; - } - - const modelStr = getComboModelString(step); - if (!modelStr) return null; - - return { - kind: "model", - stepId: step.id, - executionKey, - modelStr, - provider: getTargetProvider(modelStr, step.providerId), - providerId: step.providerId || null, - connectionId: step.connectionId || null, - weight, - label, - } satisfies ResolvedComboTarget; -} - -function getDirectComboTargets(combo: ComboLike): ResolvedComboTarget[] { - return getOrderedTopLevelRuntimeSteps(combo, null).filter( - (entry): entry is ResolvedComboTarget => entry?.kind === "model" - ); -} - -function getTopLevelRuntimeSteps( - combo: ComboLike, - allCombos: ComboCollectionLike, - path: string[] = [] -): ComboRuntimeStep[] { - return (combo.models || []) - .map((entry, index) => normalizeRuntimeStep(entry, combo.name, index, allCombos, path)) - .filter((entry): entry is ComboRuntimeStep => entry !== null); -} - -function getCompositeTierStepOrder(combo: ComboLike): string[] { - const compositeTiers = isRecord(combo?.config) ? combo.config.compositeTiers : null; - if (!isRecord(compositeTiers)) return []; - - const defaultTier = toTrimmedString(compositeTiers.defaultTier); - const tiers = isRecord(compositeTiers.tiers) ? compositeTiers.tiers : null; - if (!defaultTier || !tiers) return []; - - const orderedStepIds: string[] = []; - const visitedTiers = new Set(); - const seenStepIds = new Set(); - type CompositeTierEntry = readonly [ - string, - { readonly stepId: string; readonly fallbackTier: string | null }, - ]; - const tierEntries = new Map( - Object.entries(tiers) - .map(([tierName, rawTier]) => { - if (!isRecord(rawTier)) return null; - const normalizedTierName = toTrimmedString(tierName); - const stepId = toTrimmedString(rawTier.stepId); - const fallbackTier = toTrimmedString(rawTier.fallbackTier); - if (!normalizedTierName || !stepId) return null; - return [normalizedTierName, { stepId, fallbackTier }] as const; - }) - .filter((entry): entry is CompositeTierEntry => entry !== null) - ); - - let currentTier: string | null = defaultTier; - while (currentTier && tierEntries.has(currentTier) && !visitedTiers.has(currentTier)) { - visitedTiers.add(currentTier); - const entry = tierEntries.get(currentTier); - if (!entry) break; - if (!seenStepIds.has(entry.stepId)) { - orderedStepIds.push(entry.stepId); - seenStepIds.add(entry.stepId); - } - currentTier = entry.fallbackTier; - } - - for (const entry of tierEntries.values()) { - if (!seenStepIds.has(entry.stepId)) { - orderedStepIds.push(entry.stepId); - seenStepIds.add(entry.stepId); - } - } - - return orderedStepIds; -} - -function hasCompositeTierRuntimeOrder(combo: ComboLike): boolean { - return getCompositeTierStepOrder(combo).length > 0; -} - -function orderRuntimeStepsByCompositeTiers( - steps: ComboRuntimeStep[], - combo: ComboLike -): ComboRuntimeStep[] { - const orderedStepIds = getCompositeTierStepOrder(combo); - if (orderedStepIds.length === 0) return steps; - - const byStepId = new Map(steps.map((step) => [step.stepId, step])); - const seen = new Set(); - const ordered: ComboRuntimeStep[] = []; - - for (const stepId of orderedStepIds) { - const step = byStepId.get(stepId); - if (!step || seen.has(step.stepId)) continue; - ordered.push(step); - seen.add(step.stepId); - } - - for (const step of steps) { - if (seen.has(step.stepId)) continue; - ordered.push(step); - seen.add(step.stepId); - } - - return ordered; -} - -function getOrderedTopLevelRuntimeSteps( - combo: ComboLike, - allCombos: ComboCollectionLike, - path: string[] = [] -): ComboRuntimeStep[] { - return orderRuntimeStepsByCompositeTiers(getTopLevelRuntimeSteps(combo, allCombos, path), combo); -} - -function expandRuntimeStep( - step: ComboRuntimeStep, - allCombos: ComboCollectionLike, - visited = new Set(), - depth = 0, - path: string[] = [] -): ResolvedComboTarget[] { - if (step.kind === "model") return [step]; - if (depth > MAX_COMBO_DEPTH) return []; - - const combos = getCombosArray(allCombos); - const nestedCombo = combos.find((combo) => combo.name === step.comboName); - if (!nestedCombo || visited.has(step.comboName)) return []; - - return resolveNestedComboTargets(nestedCombo, combos, new Set(visited), depth + 1, [ - ...path, - step.stepId, - ]); -} - -export function resolveNestedComboTargets( - combo: ComboLike, - allCombos: ComboCollectionLike, - visited = new Set(), - depth = 0, - path: string[] = [] -): ResolvedComboTarget[] { - const directTargets = (combo.models || []) - .map((entry, index) => normalizeRuntimeStep(entry, combo.name, index, null, path)) - .filter((entry): entry is ResolvedComboTarget => entry?.kind === "model"); - - if (depth > MAX_COMBO_DEPTH) return directTargets; - if (visited.has(combo.name)) return []; - visited.add(combo.name); - - const runtimeSteps = getOrderedTopLevelRuntimeSteps(combo, allCombos, path); - const resolved: ResolvedComboTarget[] = []; - - for (const step of runtimeSteps) { - if (step.kind === "combo-ref") { - resolved.push(...expandRuntimeStep(step, allCombos, new Set(visited), depth, path)); - continue; - } - resolved.push(step); - } - - return resolved; -} - -/** - * Get combo models from combos data (for open-sse standalone use) - * @param {string} modelStr - Model string to check - * @param {Array|Object} combosData - Array of combos or object with combos - * @returns {Object|null} Full combo object or null if not a combo - */ -export function getComboFromData( - modelStr: string, - combosData: ComboCollectionLike -): ComboLike | null { - const combos = getCombosArray(combosData); - const combo = combos.find((c) => c.name === modelStr); - if (combo?.models && combo.models.length > 0) { - return combo; - } - return null; -} - -/** - * Legacy: Get combo models as string array (backward compat) - */ -export function getComboModelsFromData( - modelStr: string, - combosData: ComboCollectionLike -): string[] | null { - const combo = getComboFromData(modelStr, combosData); - if (!combo) return null; - return combo.models.map((m) => normalizeModelEntry(m).model); -} - -/** - * Validate combo DAG — detect circular references and enforce max depth - * @param {string} comboName - Name of the combo to validate - * @param {Array} allCombos - All combos in the system - * @param {Set} [visited] - Set of already visited combo names (for cycle detection) - * @param {number} [depth] - Current depth level - * @throws {Error} If circular reference or max depth exceeded - */ -export function validateComboDAG( - comboName: string, - allCombos: ComboCollectionLike, - visited = new Set(), - depth = 0 -): void { - if (depth > MAX_COMBO_DEPTH) { - throw new Error(`Max combo nesting depth (${MAX_COMBO_DEPTH}) exceeded at "${comboName}"`); - } - if (visited.has(comboName)) { - throw new Error(`Circular combo reference detected: ${comboName}`); - } - visited.add(comboName); - - const combos = getCombosArray(allCombos); - const combo = combos.find((c) => c.name === comboName); - if (!combo?.models) return; - - for (const entry of combo.models) { - const modelName = normalizeModelEntry(entry).model; - // Check if this model name is itself a combo (not a provider/model pattern) - const nestedCombo = combos.find((c) => c.name === modelName); - if (nestedCombo) { - validateComboDAG(modelName, combos, new Set(visited), depth + 1); - } - } -} - -/** - * Resolve nested combos by expanding inline to a flat model list - * Respects max depth and detects cycles - * @param {Object} combo - The combo object - * @param {Array} allCombos - All combos in the system - * @param {Set} [visited] - For cycle detection - * @param {number} [depth] - Current depth - * @returns {Array} Flat array of model strings - */ -export function resolveNestedComboModels( - combo: ComboLike, - allCombos: ComboCollectionLike, - visited = new Set(), - depth = 0 -): string[] { - if (depth > MAX_COMBO_DEPTH) return combo.models.map((m) => normalizeModelEntry(m).model); - if (visited.has(combo.name)) return []; // cycle safety - visited.add(combo.name); - - const combos = getCombosArray(allCombos); - const resolved: string[] = []; - - for (const entry of combo.models || []) { - const modelName = normalizeModelEntry(entry).model; - const nestedCombo = combos.find((c) => c.name === modelName); - - if (nestedCombo) { - // Recursively expand the nested combo - const nested = resolveNestedComboModels(nestedCombo, combos, new Set(visited), depth + 1); - resolved.push(...nested); - } else { - resolved.push(modelName); - } - } - - return resolved; -} - -function selectWeightedTarget(targets: T[]) { - if (targets.length === 0) return null; - - const totalWeight = targets.reduce((sum, target) => sum + (target.weight || 0), 0); - if (totalWeight <= 0) { - return targets[Math.floor(Math.random() * targets.length)]; - } - - let random = Math.random() * totalWeight; - for (const target of targets) { - random -= target.weight || 0; - if (random <= 0) return target; - } - - return targets.at(-1); -} - -function orderTargetsForWeightedFallback( - targets: T[], - selectedExecutionKey: string, - preserveExistingOrder = false -): T[] { - const selected = targets.find((target) => target.executionKey === selectedExecutionKey); - const rest = targets.filter((target) => target.executionKey !== selectedExecutionKey); - if (!preserveExistingOrder) { - rest.sort((a, b) => b.weight - a.weight); - } - return selected ? [selected, ...rest] : rest; -} - -// shuffleArray and getNextModelFromDeck moved to src/shared/utils/shuffleDeck.ts -// combo.ts now uses the shared, mutex-protected getNextFromDeck with "combo:" namespace. - -/** - * Sort models by pricing (cheapest first) for cost-optimized strategy - * @param {Array} models - Model strings in "provider/model" format - * @returns {Promise>} Sorted model strings - */ -async function sortModelsByCost(models: string[]): Promise { - try { - const { getPricingForModel } = await import("../../src/lib/localDb"); - const withCost = await Promise.all( - models.map(async (modelStr) => { - const parsed = parseModel(modelStr); - const provider = parsed.provider || parsed.providerAlias || "unknown"; - const model = parsed.model || modelStr; - try { - const pricing = await getPricingForModel(provider, model); - const cost = Number(pricing?.input); - return { modelStr, cost: Number.isFinite(cost) ? cost : Infinity }; - } catch { - return { modelStr, cost: Infinity }; - } - }) - ); - withCost.sort((a, b) => a.cost - b.cost); - return withCost.map((e) => e.modelStr); - } catch { - // If pricing lookup fails entirely, return original order - return models; - } -} - -async function sortTargetsByCost(targets: ResolvedComboTarget[]) { - const orderedModels = await sortModelsByCost(targets.map((target) => target.modelStr)); - const byModel = new Map(); - for (const target of targets) { - const queue = byModel.get(target.modelStr) || []; - queue.push(target); - byModel.set(target.modelStr, queue); - } - return orderedModels - .map((modelStr) => { - const queue = byModel.get(modelStr); - return queue?.shift() || null; - }) - .filter((target): target is ResolvedComboTarget => target !== null); -} - -/** - * Sort models by usage count (least-used first) for least-used strategy - * @param {Array} models - Model strings - * @param {string} comboName - Combo name for metrics lookup - * @returns {Array} Sorted model strings - */ -function sortModelsByUsage(models: string[], comboName: string): string[] { - const metrics = getComboMetrics(comboName); - if (!metrics?.byModel) return models; - - const withUsage = models.map((modelStr) => ({ - modelStr, - requests: metrics.byModel[modelStr]?.requests ?? 0, - })); - withUsage.sort((a, b) => a.requests - b.requests); - return withUsage.map((e) => e.modelStr); -} - -function sortTargetsByUsage(targets: ResolvedComboTarget[], comboName: string) { - const orderedModels = sortModelsByUsage( - targets.map((target) => target.modelStr), - comboName - ); - const byModel = new Map(); - for (const target of targets) { - const queue = byModel.get(target.modelStr) || []; - queue.push(target); - byModel.set(target.modelStr, queue); - } - return orderedModels - .map((modelStr) => { - const queue = byModel.get(modelStr); - return queue?.shift() || null; - }) - .filter((target): target is ResolvedComboTarget => target !== null); -} - -/** - * Sort models by context window size (largest first) for context-optimized strategy. - * Uses models.dev synced capabilities to get context limits. - * @param {Array} models - Model strings in "provider/model" format - * @returns {Array} Sorted model strings (largest context first) - */ -function sortModelsByContextSize(models: string[]): string[] { - const withContext = models.map((modelStr) => { - return { modelStr, context: getModelContextLimitForModelString(modelStr) ?? 0 }; - }); - withContext.sort((a, b) => b.context - a.context); - return withContext.map((e) => e.modelStr); -} - -function getModelContextLimitForModelString(modelStr: string) { - const parsed = parseModel(modelStr); - const provider = parsed.provider || parsed.providerAlias || "unknown"; - const model = parsed.model || modelStr; - return getModelContextLimit(provider, model); -} - -type RequestCompatibilityRequirements = { - requiresTools: boolean; - requiresVision: boolean; - requiresStructuredOutput: boolean; - estimatedInputTokens: number; - requestedOutputTokens: number; - requiredContextTokens: number; -}; - -function getPositiveTokenCount(value: unknown): number { - const count = Number(value); - return Number.isFinite(count) && count > 0 ? Math.ceil(count) : 0; -} - -function requestRequiresTools(body: Record): boolean { - if (Array.isArray(body.tools) && body.tools.length > 0) return true; - if (Array.isArray(body.functions) && body.functions.length > 0) return true; - return false; -} - -function requestRequiresStructuredOutput(body: Record): boolean { - const responseFormat = isRecord(body.response_format) ? body.response_format : null; - const type = typeof responseFormat?.type === "string" ? responseFormat.type : null; - return type === "json_object" || type === "json_schema"; -} - -function estimateRequestInputTokens(body: Record): number { - const estimatePayload: Record = {}; - for (const key of ["messages", "input", "tools", "functions", "response_format"]) { - if (body[key] !== undefined) estimatePayload[key] = body[key]; - } - return Object.keys(estimatePayload).length > 0 ? estimateTokens(estimatePayload) : 0; -} - -function valueContainsImagePart(value: unknown, depth = 0): boolean { - if (depth > 8 || value === null || value === undefined) return false; - if (typeof value === "string") return value.startsWith("data:image/"); - if (Array.isArray(value)) return value.some((entry) => valueContainsImagePart(entry, depth + 1)); - if (!isRecord(value)) return false; - - const type = typeof value.type === "string" ? value.type.toLowerCase() : null; - if (type === "image" || type === "image_url" || type === "input_image") return true; - if ("image_url" in value || "input_image" in value) return true; - - const source = isRecord(value.source) ? value.source : null; - const mediaType = typeof source?.media_type === "string" ? source.media_type.toLowerCase() : ""; - if (mediaType.startsWith("image/")) return true; - - return Object.values(value).some((entry) => valueContainsImagePart(entry, depth + 1)); -} - -function deriveRequestCompatibilityRequirements( - body: Record -): RequestCompatibilityRequirements { - const estimatedInputTokens = estimateRequestInputTokens(body); - const requestedOutputTokens = Math.max( - getPositiveTokenCount(body.max_tokens), - getPositiveTokenCount(body.max_completion_tokens) - ); - return { - requiresTools: requestRequiresTools(body), - requiresVision: valueContainsImagePart(body.messages) || valueContainsImagePart(body.input), - requiresStructuredOutput: requestRequiresStructuredOutput(body), - estimatedInputTokens, - requestedOutputTokens, - requiredContextTokens: estimatedInputTokens + requestedOutputTokens, - }; -} - -function getTargetCompatibilityFailures( - target: ResolvedComboTarget, - requirements: RequestCompatibilityRequirements -): string[] { - const capabilities = getResolvedModelCapabilities(target.modelStr); - const failures: string[] = []; - - if ( - requirements.requiresTools && - (capabilities.supportsTools === false || !capabilities.toolCalling) - ) { - failures.push("tools"); - } - - if (requirements.requiresVision && capabilities.supportsVision === false) { - failures.push("vision"); - } - - if (requirements.requiresStructuredOutput && capabilities.structuredOutput === false) { - failures.push("structured_output"); - } - - if ( - requirements.requestedOutputTokens > 0 && - Number.isFinite(capabilities.maxOutputTokens) && - capabilities.maxOutputTokens < requirements.requestedOutputTokens - ) { - failures.push("output_tokens"); - } - - const contextLimit = capabilities.maxInputTokens ?? capabilities.contextWindow ?? null; - if ( - requirements.requiredContextTokens > 0 && - contextLimit !== null && - contextLimit !== undefined && - contextLimit < requirements.requiredContextTokens - ) { - failures.push("context_window"); - } - - return failures; -} - -function filterTargetsByRequestCompatibility( - targets: ResolvedComboTarget[], - body: Record, - log: ComboLogger, - label = "Context-aware fallback" -): ResolvedComboTarget[] { - if (targets.length === 0) return targets; - const requirements = deriveRequestCompatibilityRequirements(body); - const needsFiltering = - requirements.requiresTools || - requirements.requiresVision || - requirements.requiresStructuredOutput || - requirements.requiredContextTokens > 0; - if (!needsFiltering) return targets; - - const rejected: Array<{ target: ResolvedComboTarget; reasons: string[] }> = []; - const compatible = targets.filter((target) => { - const reasons = getTargetCompatibilityFailures(target, requirements); - if (reasons.length === 0) return true; - rejected.push({ target, reasons }); - return false; - }); - - if (compatible.length === targets.length) return targets; - if (compatible.length === 0) { - log.warn( - "COMBO", - `${label}: all ${targets.length} targets were filtered by request requirements; preserving strategy order` - ); - log.debug?.( - "COMBO", - `${label}: rejected targets ${rejected - .map((entry) => `${entry.target.modelStr}(${entry.reasons.join("+")})`) - .join(", ")}` - ); - return targets; - } - - log.info( - "COMBO", - `${label}: kept ${compatible.length}/${targets.length} targets for request requirements` - ); - log.debug?.( - "COMBO", - `${label}: rejected targets ${rejected - .map((entry) => `${entry.target.modelStr}(${entry.reasons.join("+")})`) - .join(", ")}` - ); - return compatible; -} - -function sortTargetsByContextSize(targets: ResolvedComboTarget[]) { - const hasKnownContext = targets.some( - (target) => getModelContextLimitForModelString(target.modelStr) != null - ); - if (!hasKnownContext) return targets; - - const orderedModels = sortModelsByContextSize(targets.map((target) => target.modelStr)); - const byModel = new Map(); - for (const target of targets) { - const queue = byModel.get(target.modelStr) || []; - queue.push(target); - byModel.set(target.modelStr, queue); - } - return orderedModels - .map((modelStr) => { - const queue = byModel.get(modelStr); - return queue?.shift() || null; - }) - .filter((target): target is ResolvedComboTarget => target !== null); -} - -function getP2CTargetScore( - target: ResolvedComboTarget, - metrics: ReturnType -): number { - const breakerState = getCircuitBreaker(target.provider)?.getStatus?.()?.state; - if (breakerState === "OPEN") return -Infinity; - const modelMetric = metrics?.byModel?.[target.modelStr] || null; - const successRate = Number(modelMetric?.successRate); - const avgLatency = Number(modelMetric?.avgLatencyMs); - const successScore = Number.isFinite(successRate) ? successRate / 100 : 0.5; - const latencyScore = - Number.isFinite(avgLatency) && avgLatency > 0 ? 1 / Math.log10(avgLatency + 10) : 0.25; - const breakerPenalty = breakerState === "HALF_OPEN" ? 0.25 : 0; - return successScore + latencyScore - breakerPenalty; -} - -function orderTargetsByPowerOfTwoChoices(targets: ResolvedComboTarget[], comboName: string) { - if (targets.length <= 1) return targets; - const metrics = getComboMetrics(comboName); - const firstIndex = Math.floor(Math.random() * targets.length); - let secondIndex = Math.floor(Math.random() * (targets.length - 1)); - if (secondIndex >= firstIndex) secondIndex++; - - const first = targets[firstIndex]; - const second = targets[secondIndex]; - const selectedIndex = - getP2CTargetScore(second, metrics) > getP2CTargetScore(first, metrics) - ? secondIndex - : firstIndex; - return [targets[selectedIndex], ...targets.filter((_, index) => index !== selectedIndex)]; -} - -function finiteNumberOrNull(value: unknown): number | null { - const numericValue = Number(value); - return Number.isFinite(numericValue) ? numericValue : null; -} - -function getPercentConfig(value: unknown, fallback: number): number { - const numericValue = finiteNumberOrNull(value); - if (numericValue === null) return fallback; - return Math.max(0, Math.min(100, numericValue)); -} - -function getWeightConfig(value: unknown, fallback: number): number { - const numericValue = finiteNumberOrNull(value); - if (numericValue === null || numericValue < 0) return fallback; - return numericValue; -} - -function getDurationConfig(value: unknown, fallback: number, max: number): number { - const numericValue = finiteNumberOrNull(value); - if (numericValue === null || numericValue < 0) return fallback; - return Math.min(max, Math.floor(numericValue)); -} - -function resolveResetAwareConfig(config: Record | null | undefined) { - const sessionWeight = getWeightConfig( - config?.resetAwareSessionWeight, - RESET_AWARE_DEFAULTS.sessionWeight - ); - const weeklyWeight = getWeightConfig( - config?.resetAwareWeeklyWeight, - RESET_AWARE_DEFAULTS.weeklyWeight - ); - const totalWeight = sessionWeight + weeklyWeight; - const normalizedSessionWeight = - totalWeight > 0 ? sessionWeight / totalWeight : RESET_AWARE_DEFAULTS.sessionWeight; - - return { - sessionWeight: normalizedSessionWeight, - weeklyWeight: 1 - normalizedSessionWeight, - tieBand: - getPercentConfig(config?.resetAwareTieBandPercent, RESET_AWARE_DEFAULTS.tieBandPercent) / 100, - exhaustionGuard: - getPercentConfig( - config?.resetAwareExhaustionGuardPercent, - RESET_AWARE_DEFAULTS.exhaustionGuardPercent - ) / 100, - quotaCacheTtlMs: getDurationConfig(config?.resetAwareQuotaCacheTtlMs, 0, 300_000), - quotaCacheMaxStaleMs: getDurationConfig(config?.resetAwareQuotaCacheMaxStaleMs, 0, 3_600_000), - }; -} - -function resolveResetWindowConfig(config: Record | null | undefined) { - const rawWindows = Array.isArray(config?.resetWindowWindows) ? config.resetWindowWindows : null; - const windows = rawWindows - ?.filter((windowName): windowName is ResetWindowName => - (RESET_WINDOW_NAMES as readonly string[]).includes(String(windowName)) - ) - .filter((windowName, index, array) => array.indexOf(windowName) === index); - - const effectiveWindows = - windows && windows.length > 0 - ? windows - : config?.resetWindowIncludeSession === true - ? (["weekly", "session"] as ResetWindowName[]) - : (["weekly"] as ResetWindowName[]); - - return { - windows: effectiveWindows, - tieBandMs: Math.max( - 0, - finiteNumberOrNull(config?.resetWindowTieBandMs) ?? RESET_WINDOW_DEFAULT_TIE_BAND_MS - ), - quotaCacheTtlMs: getDurationConfig(config?.resetWindowQuotaCacheTtlMs, 0, 300_000), - quotaCacheMaxStaleMs: getDurationConfig(config?.resetWindowQuotaCacheMaxStaleMs, 0, 3_600_000), - }; -} - -function resolveSlaRoutingPolicy( - config: Record | null | undefined -): SlaRoutingPolicy | undefined { - if (!config) return undefined; - const nestedSla = isRecord(config.sla) ? config.sla : {}; - const targetP95Ms = finiteNumberOrNull(config.slaTargetP95Ms ?? nestedSla.targetP95Ms); - const maxErrorRate = finiteNumberOrNull(config.slaMaxErrorRate ?? nestedSla.maxErrorRate); - const maxCostPer1MTokens = finiteNumberOrNull( - config.slaMaxCostPer1MTokens ?? nestedSla.maxCostPer1MTokens - ); - const hardConstraints = config.slaHardConstraints ?? nestedSla.hardConstraints; - - const policy: SlaRoutingPolicy = {}; - if (targetP95Ms !== null && targetP95Ms > 0) policy.targetP95Ms = targetP95Ms; - if (maxErrorRate !== null && maxErrorRate >= 0) policy.maxErrorRate = clamp01(maxErrorRate); - if (maxCostPer1MTokens !== null && maxCostPer1MTokens > 0) { - policy.maxCostPer1MTokens = maxCostPer1MTokens; - } - if (typeof hardConstraints === "boolean") policy.hardConstraints = hardConstraints; - - return Object.keys(policy).length > 0 ? policy : undefined; -} - -function getResetAwareProvider(target: ResolvedComboTarget): string | null { - const provider = (target.providerId || target.provider || "").toLowerCase(); - return provider || null; -} - -function normalizeResetAt(value: unknown): string | null { - if (typeof value === "string" && value.trim().length > 0) return value.trim(); - if (typeof value === "number" && Number.isFinite(value)) return String(value); - return null; -} - -function parseResetTimeMs(resetAt: string | null | undefined): number { - if (!resetAt) return NaN; - const resetTime = Date.parse(resetAt); - if (Number.isFinite(resetTime)) return resetTime; - - if (!/^\d+(?:\.\d+)?$/.test(resetAt)) return NaN; - const numericResetAt = Number(resetAt); - if (!Number.isFinite(numericResetAt)) return NaN; - return numericResetAt < 10_000_000_000 ? numericResetAt * 1000 : numericResetAt; -} - -function getQuotaWindow( - quota: unknown, - key: "window5h" | "window7d" | "windowWeekly" | "windowMonthly" -): { percentUsed: number | null; resetAt: string | null } | null { - if (!isRecord(quota)) return null; - const window = quota[key]; - if (!isRecord(window)) return null; - const percentUsed = finiteNumberOrNull(window.percentUsed); - const resetAt = normalizeResetAt(window.resetAt); - return { percentUsed, resetAt }; -} - -function normalizeWindowPercentUsed(value: unknown): number | null { - const numericValue = finiteNumberOrNull(value); - if (numericValue === null) return null; - if (numericValue > 1) return clamp01(numericValue / 100); - return clamp01(numericValue); -} - -function getNamedQuotaWindow( - quota: unknown, - windowName: ResetWindowName -): { percentUsed: number | null; resetAt: string | null } | null { - if (!quota || !isRecord(quota)) return null; - - if (windowName === "session") return getQuotaWindow(quota, "window5h"); - if (windowName === "weekly") { - return getQuotaWindow(quota, "window7d") || getQuotaWindow(quota, "windowWeekly"); - } - if (windowName === "monthly") return getQuotaWindow(quota, "windowMonthly"); - - return null; -} - -function getWindowsMapQuotaWindow( - quota: unknown, - windowName: ResetWindowName -): { percentUsed: number | null; resetAt: string | null } | null { - if (!quota || !isRecord(quota) || !isRecord(quota.windows)) return null; - const candidates = Object.entries(quota.windows) - .map(([key, value]) => ({ key: key.toLowerCase(), value })) - .filter(({ key }) => key === windowName || key.startsWith(`${windowName} `)); - - if (candidates.length === 0) return null; - candidates.sort((a, b) => a.key.localeCompare(b.key)); - const window = candidates[0].value; - if (!isRecord(window)) return null; - - return { - percentUsed: normalizeWindowPercentUsed(window.percentUsed), - resetAt: normalizeResetAt(window.resetAt), - }; -} - -function resolveQuotaWindowByName( - quota: unknown, - windowName: ResetWindowName -): { percentUsed: number | null; resetAt: string | null } | null { - return getNamedQuotaWindow(quota, windowName) || getWindowsMapQuotaWindow(quota, windowName); -} - -function getResetUrgency(resetAt: string | null | undefined, windowMs: number): number { - if (!resetAt) return 0.5; - const resetTime = parseResetTimeMs(resetAt); - if (!Number.isFinite(resetTime)) return 0.5; - const msUntilReset = resetTime - Date.now(); - if (msUntilReset <= 0) return 1; - return clamp01(1 - msUntilReset / windowMs); -} - -function scoreQuotaWindow( - remaining: number, - resetAt: string | null | undefined, - windowMs: number, - remainingWeight: number, - resetPressureWeight: number -): number { - const normalizedRemaining = clamp01(remaining); - const resetUrgency = getResetUrgency(resetAt, windowMs); - const resetPressure = resetUrgency * (1 - normalizedRemaining); - return remainingWeight * normalizedRemaining + resetPressureWeight * resetPressure; -} - -function scoreResetAwareQuota(quota: unknown, config: ReturnType) { - if (!quota || !isRecord(quota)) return { score: 0.5 }; - if (quota.limitReached === true) return { score: -Infinity }; - - const overallPercentUsed = clamp01(finiteNumberOrNull(quota.percentUsed) ?? 0.5); - const sessionWindow = getQuotaWindow(quota, "window5h"); - const weeklyWindow = getQuotaWindow(quota, "window7d") || getQuotaWindow(quota, "windowWeekly"); - const sessionRemaining = clamp01(1 - (sessionWindow?.percentUsed ?? overallPercentUsed)); - const weeklyRemaining = clamp01(1 - (weeklyWindow?.percentUsed ?? overallPercentUsed)); - const sessionScore = scoreQuotaWindow( - sessionRemaining, - sessionWindow?.resetAt, - RESET_AWARE_SESSION_WINDOW_MS, - RESET_AWARE_SESSION_REMAINING_WEIGHT, - RESET_AWARE_SESSION_RESET_PRESSURE_WEIGHT - ); - const weeklyScore = scoreQuotaWindow( - weeklyRemaining, - weeklyWindow?.resetAt ?? normalizeResetAt(quota.resetAt), - RESET_AWARE_WEEKLY_WINDOW_MS, - RESET_AWARE_WEEKLY_REMAINING_WEIGHT, - RESET_AWARE_WEEKLY_RESET_PRESSURE_WEIGHT - ); - let score = config.sessionWeight * sessionScore + config.weeklyWeight * weeklyScore; - - if (config.exhaustionGuard > 0 && sessionRemaining < config.exhaustionGuard) { - score *= Math.max(0.05, sessionRemaining / config.exhaustionGuard); - } - - return { score }; -} - -async function getQuotaAwareConnectionsForTarget( - target: ResolvedComboTarget, - connectionCache: Map>>, - connectionLoadPromises: Map>>>, - comboName: string, - log: { warn?: (...args: unknown[]) => void } -) { - const provider = getResetAwareProvider(target); - if (!provider || !getQuotaFetcher(provider)) return []; - if (!connectionCache.has(provider)) { - const cached = resetAwareConnectionCache.get(provider); - if (cached && Date.now() - cached.fetchedAt < RESET_AWARE_CONNECTION_CACHE_TTL_MS) { - connectionCache.set(provider, cached.connections); - return cached.connections; - } - - if (!connectionLoadPromises.has(provider)) { - connectionLoadPromises.set( - provider, - (async () => { - try { - const connections = await getProviderConnections({ provider, isActive: true }); - const activeConnections = Array.isArray(connections) - ? (connections as Array>) - : []; - if ( - !resetAwareConnectionCache.has(provider) && - resetAwareConnectionCache.size >= MAX_RESET_AWARE_CACHE - ) { - const oldest = resetAwareConnectionCache.keys().next().value; - if (oldest !== undefined) resetAwareConnectionCache.delete(oldest); - } - resetAwareConnectionCache.set(provider, { - connections: activeConnections, - fetchedAt: Date.now(), - }); - return activeConnections; - } catch (error) { - log.warn?.("COMBO", "Reset-aware failed to load quota-aware connections.", { - comboName, - err: error, - operation: "getProviderConnections", - provider, - }); - return []; - } - })() - ); - } - - const connections = await connectionLoadPromises.get(provider)!; - connectionCache.set(provider, connections); - } - return connectionCache.get(provider) || []; -} - -function normalizeConnectionIds(value: unknown): string[] | null { - if (!Array.isArray(value)) return null; - const ids = value.filter( - (connectionId): connectionId is string => - typeof connectionId === "string" && connectionId.trim().length > 0 - ); - return ids.length > 0 ? ids : null; -} - -function filterAllowedConnectionIds( - connectionIds: string[], - apiKeyAllowedConnectionIds: string[] | null | undefined -): string[] { - const allowedIds = normalizeConnectionIds(apiKeyAllowedConnectionIds); - if (!allowedIds) return connectionIds; - const allowedSet = new Set(allowedIds); - return connectionIds.filter((connectionId) => allowedSet.has(connectionId)); -} - -function getTargetConnectionIds( - target: ResolvedComboTarget, - connections: Array> -): string[] { - let connectionIds: string[]; - if (target.connectionId) { - return [target.connectionId]; - } - - if (Array.isArray(target.allowedConnectionIds) && target.allowedConnectionIds.length > 0) { - return target.allowedConnectionIds.filter( - (connectionId): connectionId is string => - typeof connectionId === "string" && connectionId.trim().length > 0 - ); - } - - connectionIds = connections - .map((connection) => (typeof connection.id === "string" ? connection.id : null)) - .filter((connectionId): connectionId is string => !!connectionId); - return connectionIds; -} - -async function mapWithConcurrency( - items: T[], - concurrency: number, - mapper: (item: T, index: number) => Promise -): Promise { - const results = new Array(items.length); - let nextIndex = 0; - const workerCount = Math.max(1, Math.min(concurrency, items.length)); - - await Promise.all( - Array.from({ length: workerCount }, async () => { - while (nextIndex < items.length) { - const currentIndex = nextIndex++; - results[currentIndex] = await mapper(items[currentIndex], currentIndex); - } - }) - ); - - return results; -} - -async function fetchResetAwareQuotaWithCache({ - provider, - connectionId, - connection, - fetcher, - config, - log, - comboName, -}: { - provider: string; - connectionId: string; - connection?: Record; - fetcher: (connectionId: string, connection?: Record) => Promise; - config: QuotaFetchCacheConfig; - log: { debug?: (...args: unknown[]) => void; warn?: (...args: unknown[]) => void }; - comboName: string; -}): Promise { - const cacheKey = `${provider}:${connectionId}`; - const ttlMs = config.quotaCacheTtlMs; - const maxStaleMs = config.quotaCacheMaxStaleMs; - const now = Date.now(); - const cached = resetAwareQuotaCache.get(cacheKey); - - if (ttlMs <= 0 && maxStaleMs <= 0) { - try { - return await fetcher(connectionId, connection); - } catch (error) { - log.warn?.("COMBO", "Reset-aware quota fetch failed.", { - comboName, - connectionId, - err: error, - operation: "quotaFetch", - provider, - }); - return null; - } - } - - const refresh = () => { - const existing = resetAwareQuotaCache.get(cacheKey); - if (existing?.refreshPromise != null) return existing.refreshPromise; - - const refreshPromise = fetcher(connectionId, connection) - .then((quota) => { - if (quota) { - if ( - !resetAwareQuotaCache.has(cacheKey) && - resetAwareQuotaCache.size >= MAX_RESET_AWARE_CACHE - ) { - const oldest = resetAwareQuotaCache.keys().next().value; - if (oldest !== undefined) resetAwareQuotaCache.delete(oldest); - } - resetAwareQuotaCache.set(cacheKey, { - quota, - fetchedAt: Date.now(), - refreshPromise: null, - }); - } else { - resetAwareQuotaCache.delete(cacheKey); - } - return quota; - }) - .catch((error) => { - const previous = resetAwareQuotaCache.get(cacheKey); - if (previous) { - if ( - !resetAwareQuotaCache.has(cacheKey) && - resetAwareQuotaCache.size >= MAX_RESET_AWARE_CACHE - ) { - const oldest = resetAwareQuotaCache.keys().next().value; - if (oldest !== undefined) resetAwareQuotaCache.delete(oldest); - } - resetAwareQuotaCache.set(cacheKey, { ...previous, refreshPromise: null }); - } - log.warn?.("COMBO", "Reset-aware quota fetch failed.", { - comboName, - connectionId, - err: error, - operation: "quotaFetch", - provider, - }); - return null; - }); - - if (!resetAwareQuotaCache.has(cacheKey) && resetAwareQuotaCache.size >= MAX_RESET_AWARE_CACHE) { - const oldest = resetAwareQuotaCache.keys().next().value; - if (oldest !== undefined) resetAwareQuotaCache.delete(oldest); - } - resetAwareQuotaCache.set(cacheKey, { - quota: existing?.quota ?? cached?.quota ?? null, - fetchedAt: existing?.fetchedAt ?? cached?.fetchedAt ?? 0, - refreshPromise, - }); - return refreshPromise; - }; - - if (ttlMs > 0 && cached) { - const age = now - cached.fetchedAt; - if (age <= ttlMs) return cached.quota; - if (maxStaleMs > 0 && age <= ttlMs + maxStaleMs) { - void refresh(); - return cached.quota; - } - } - - return refresh(); -} - -type PreScreenResult = { profile: ProviderProfile | null; available: boolean }; - -export async function preScreenTargets( - targets: ResolvedComboTarget[], - isModelAvailable?: IsModelAvailable | null -): Promise> { - if (targets.length === 0) { - return new Map(); - } - - const results = await mapWithConcurrency( - targets, - PRE_SCREEN_CONCURRENCY, - async (target): Promise<{ key: string; result: PreScreenResult }> => { - const profile = await getRuntimeProviderProfile(target.provider).catch(() => null); - - const breaker = getCircuitBreaker(target.provider); - if (breaker.getStatus().state === "OPEN") { - return { key: target.executionKey, result: { profile, available: false } }; - } - - let available = true; - if (isModelAvailable) { - // IsModelAvailable may return a sync boolean or a Promise; Promise.resolve - // normalizes both so the .catch() never runs against a bare boolean. - available = await Promise.resolve(isModelAvailable(target.modelStr, target)).catch( - () => true - ); - } - return { key: target.executionKey, result: { profile, available } }; - } - ); - - const map = new Map(); - for (const { key, result } of results) { - map.set(key, result); - } - return map; -} - -async function orderTargetsByResetAwareQuota( - targets: ResolvedComboTarget[], - comboName: string, - configSource: Record | null | undefined, - log: { warn?: (...args: unknown[]) => void }, - apiKeyAllowedConnectionIds?: string[] | null -) { - if (targets.length === 0) return targets; - - const config = resolveResetAwareConfig(configSource); - const connectionCache = new Map>>(); - const connectionLoadPromises = new Map>>>(); - const quotaPromises = new Map>(); - const connectionById = new Map>(); - const expandedTargets: ResolvedComboTarget[] = []; - - const targetsWithConnections = await Promise.all( - targets.map(async (target) => ({ - connections: await getQuotaAwareConnectionsForTarget( - target, - connectionCache, - connectionLoadPromises, - comboName, - log - ), - target, - })) - ); - - for (const { target, connections } of targetsWithConnections) { - for (const connection of connections) { - if (typeof connection.id === "string") connectionById.set(connection.id, connection); - } - - const unrestrictedConnectionIds = getTargetConnectionIds(target, connections); - const connectionIds = filterAllowedConnectionIds( - unrestrictedConnectionIds, - apiKeyAllowedConnectionIds - ); - if (connectionIds.length === 0) { - if ( - unrestrictedConnectionIds.length > 0 && - normalizeConnectionIds(apiKeyAllowedConnectionIds) - ) { - continue; - } - expandedTargets.push(target); - continue; - } - - for (const connectionId of connectionIds) { - expandedTargets.push({ - ...target, - connectionId, - executionKey: - target.connectionId === connectionId - ? target.executionKey - : `${target.executionKey}@${connectionId}`, - }); - } - } - - const scoredTargets = await mapWithConcurrency( - expandedTargets, - RESET_AWARE_QUOTA_FETCH_CONCURRENCY, - async (target, index) => { - let quota: unknown = null; - const provider = getResetAwareProvider(target); - const fetcher = provider ? getQuotaFetcher(provider) : null; - if (fetcher && provider && target.connectionId) { - const quotaKey = `${provider}:${target.connectionId}`; - if (!quotaPromises.has(quotaKey)) { - quotaPromises.set( - quotaKey, - fetchResetAwareQuotaWithCache({ - provider, - connectionId: target.connectionId, - connection: connectionById.get(target.connectionId), - fetcher, - config, - log, - comboName, - }) - ); - } - quota = await quotaPromises.get(quotaKey)!; - } - const { score } = scoreResetAwareQuota(quota, config); - return { target, score, index }; - } - ); - - scoredTargets.sort((a, b) => { - if (b.score !== a.score) return b.score - a.score; - return a.index - b.index; - }); - - const bestScore = scoredTargets[0]?.score ?? 0; - const tiedTargets = scoredTargets.filter((entry) => bestScore - entry.score <= config.tieBand); - let orderedTiedTargets = tiedTargets; - if (tiedTargets.length > 1) { - const key = `reset-aware:${comboName}`; - const counter = rrCounters.get(key) || 0; - if (!rrCounters.has(key) && rrCounters.size >= MAX_RR_COUNTERS) { - const oldest = rrCounters.keys().next().value; - if (oldest !== undefined) rrCounters.delete(oldest); - } - rrCounters.set(key, counter + 1); - const startIndex = counter % tiedTargets.length; - orderedTiedTargets = [...tiedTargets.slice(startIndex), ...tiedTargets.slice(0, startIndex)]; - } - - const tiedExecutionKeys = new Set(orderedTiedTargets.map((entry) => entry.target.executionKey)); - return [ - ...orderedTiedTargets, - ...scoredTargets.filter((entry) => !tiedExecutionKeys.has(entry.target.executionKey)), - ].map((entry) => entry.target); -} - -function getResetWindowTimestampMs(quota: unknown, windows: ResetWindowName[]): number { - if (!quota || !isRecord(quota) || quota.limitReached === true) return Infinity; - - let selectedResetMs = Infinity; - for (const windowName of windows) { - const window = resolveQuotaWindowByName(quota, windowName); - const resetMs = parseResetTimeMs(window?.resetAt ?? null); - if (Number.isFinite(resetMs)) { - selectedResetMs = Math.min(selectedResetMs, resetMs); - } - } - - if (!Number.isFinite(selectedResetMs)) { - selectedResetMs = parseResetTimeMs(normalizeResetAt(quota.resetAt)); - } - - return Number.isFinite(selectedResetMs) ? selectedResetMs : Infinity; -} - -function getResetWindowHorizonMs(windows: ResetWindowName[]): number { - if (windows.includes("monthly")) return 30 * 24 * 60 * 60 * 1000; - if (windows.includes("weekly")) return RESET_AWARE_WEEKLY_WINDOW_MS; - return RESET_AWARE_SESSION_WINDOW_MS; -} - -function calculateResetWindowAffinity(quota: unknown, config: ResetWindowConfig): number { - const resetMs = getResetWindowTimestampMs(quota, config.windows); - if (!Number.isFinite(resetMs)) return 0.5; - - const msUntilReset = resetMs - Date.now(); - if (msUntilReset <= 0) return 1; - return clamp01(1 - msUntilReset / getResetWindowHorizonMs(config.windows)); -} - -async function orderTargetsByResetWindow( - targets: ResolvedComboTarget[], - comboName: string, - configSource: Record | null | undefined, - log: { warn?: (...args: unknown[]) => void }, - apiKeyAllowedConnectionIds?: string[] | null -) { - if (targets.length === 0) return targets; - - const config = resolveResetWindowConfig(configSource); - const connectionCache = new Map>>(); - const connectionLoadPromises = new Map>>>(); - const quotaPromises = new Map>(); - const connectionById = new Map>(); - const expandedTargets: ResolvedComboTarget[] = []; - - const targetsWithConnections = await Promise.all( - targets.map(async (target) => ({ - connections: await getQuotaAwareConnectionsForTarget( - target, - connectionCache, - connectionLoadPromises, - comboName, - log - ), - target, - })) - ); - - for (const { target, connections } of targetsWithConnections) { - for (const connection of connections) { - if (typeof connection.id === "string") connectionById.set(connection.id, connection); - } - - const unrestrictedConnectionIds = getTargetConnectionIds(target, connections); - const connectionIds = filterAllowedConnectionIds( - unrestrictedConnectionIds, - apiKeyAllowedConnectionIds - ); - if (connectionIds.length === 0) { - if ( - unrestrictedConnectionIds.length > 0 && - normalizeConnectionIds(apiKeyAllowedConnectionIds) - ) { - continue; - } - expandedTargets.push(target); - continue; - } - - for (const connectionId of connectionIds) { - expandedTargets.push({ - ...target, - connectionId, - executionKey: - target.connectionId === connectionId - ? target.executionKey - : `${target.executionKey}@${connectionId}`, - }); - } - } - - const scoredTargets = await mapWithConcurrency( - expandedTargets, - RESET_AWARE_QUOTA_FETCH_CONCURRENCY, - async (target, index) => { - let quota: unknown = null; - const provider = getResetAwareProvider(target); - const fetcher = provider ? getQuotaFetcher(provider) : null; - if (fetcher && provider && target.connectionId) { - const quotaKey = `${provider}:${target.connectionId}`; - if (!quotaPromises.has(quotaKey)) { - quotaPromises.set( - quotaKey, - fetchResetAwareQuotaWithCache({ - provider, - connectionId: target.connectionId, - connection: connectionById.get(target.connectionId), - fetcher, - config, - log, - comboName, - }) - ); - } - quota = await quotaPromises.get(quotaKey)!; - } - - return { - target, - resetMs: getResetWindowTimestampMs(quota, config.windows), - index, - }; - } - ); - - scoredTargets.sort((a, b) => { - if (a.resetMs !== b.resetMs) return a.resetMs - b.resetMs; - return a.index - b.index; - }); - - const bestResetMs = scoredTargets[0]?.resetMs ?? Infinity; - if (!Number.isFinite(bestResetMs) || config.tieBandMs <= 0) { - return scoredTargets.map((entry) => entry.target); - } - - const tiedTargets = scoredTargets.filter( - (entry) => entry.resetMs - bestResetMs <= config.tieBandMs - ); - if (tiedTargets.length <= 1) return scoredTargets.map((entry) => entry.target); - - const key = `reset-window:${comboName}`; - const counter = rrCounters.get(key) || 0; - if (!rrCounters.has(key) && rrCounters.size >= MAX_RR_COUNTERS) { - const oldest = rrCounters.keys().next().value; - if (oldest !== undefined) rrCounters.delete(oldest); - } - rrCounters.set(key, counter + 1); - const startIndex = counter % tiedTargets.length; - const orderedTiedTargets = [ - ...tiedTargets.slice(startIndex), - ...tiedTargets.slice(0, startIndex), - ]; - const tiedExecutionKeys = new Set(orderedTiedTargets.map((entry) => entry.target.executionKey)); - - return [ - ...orderedTiedTargets, - ...scoredTargets.filter((entry) => !tiedExecutionKeys.has(entry.target.executionKey)), - ].map((entry) => entry.target); -} - -function toTextContent(content: unknown): string { - if (typeof content === "string") return content; - if (!Array.isArray(content)) return ""; - return content - .map((part) => { - if (!isRecord(part)) return ""; - if (typeof part.text === "string") return part.text; - return ""; - }) - .join("\n"); -} - -function extractPromptForIntent(body: Record | null | undefined): string { - if (!body || typeof body !== "object") return ""; - - const fromMessages = Array.isArray(body.messages) - ? [...body.messages].reverse().find((m) => isRecord(m) && m.role === "user") - : null; - if (isRecord(fromMessages)) return toTextContent(fromMessages.content); - - if (typeof body.input === "string") return body.input; - if (Array.isArray(body.input)) { - const text = body.input - .map((item) => { - if (!isRecord(item)) return ""; - if (typeof item.content === "string") return item.content; - if (typeof item.text === "string") return item.text; - return ""; - }) - .filter(Boolean) - .join("\n"); - if (text) return text; - } - - if (typeof body.prompt === "string") return body.prompt; - return ""; -} - -function mapIntentToTaskType(intent: string): "coding" | "analysis" | "default" { - switch (intent) { - case "code": - return "coding"; - case "reasoning": - return "analysis"; - case "simple": - return "default"; - case "medium": - default: - return "default"; - } -} - -function calculateTargetContextAffinity( - target: ResolvedComboTarget, - sessionId: string | null | undefined -): number { - const sessionConnectionId = getSessionConnection(sessionId || null); - if (!sessionConnectionId) return 0.5; - if (target.connectionId === sessionConnectionId) return 1; - if (!target.connectionId) return 0.5; - return 0.1; -} - -function toStringArray(input: unknown): string[] { - if (Array.isArray(input)) { - return input.map((v) => (typeof v === "string" ? v.trim() : "")).filter(Boolean); - } - if (typeof input === "string") { - return input - .split(",") - .map((v) => v.trim()) - .filter(Boolean); - } - return []; -} - -function getIntentConfig( - settings: Record | null | undefined, - combo: ComboLike -): IntentClassifierConfig { - const resolvedSettings = settings || {}; - const comboAutoConfig = combo?.autoConfig || {}; - const comboConfigAuto = isRecord(combo?.config?.auto) ? combo.config.auto : {}; - const comboIntentConfig = - (isRecord(comboAutoConfig.intentConfig) && comboAutoConfig.intentConfig) || - (isRecord(comboConfigAuto.intentConfig) && comboConfigAuto.intentConfig) || - (isRecord(combo?.config?.intentConfig) && combo.config.intentConfig) || - {}; - - return { - ...DEFAULT_INTENT_CONFIG, - ...comboIntentConfig, - ...(typeof resolvedSettings.intentDetectionEnabled === "boolean" - ? { enabled: resolvedSettings.intentDetectionEnabled } - : {}), - ...(Number.isFinite(Number(resolvedSettings.intentSimpleMaxWords)) - ? { simpleMaxWords: Number(resolvedSettings.intentSimpleMaxWords) } - : {}), - ...(toStringArray(resolvedSettings.intentExtraCodeKeywords).length > 0 - ? { extraCodeKeywords: toStringArray(resolvedSettings.intentExtraCodeKeywords) } - : {}), - ...(toStringArray(resolvedSettings.intentExtraReasoningKeywords).length > 0 - ? { extraReasoningKeywords: toStringArray(resolvedSettings.intentExtraReasoningKeywords) } - : {}), - ...(toStringArray(resolvedSettings.intentExtraSimpleKeywords).length > 0 - ? { extraSimpleKeywords: toStringArray(resolvedSettings.intentExtraSimpleKeywords) } - : {}), - }; -} - -function getBootstrapLatencyMs(modelId: string): number { - const normalized = String(modelId || "").toLowerCase(); - return DEFAULT_MODEL_P95_MS[normalized] ?? 1500; -} - -async function buildAutoCandidates( - targets: ResolvedComboTarget[], - comboName: string, - sessionId: string | null | undefined = null, - resetWindowConfig: ResetWindowConfig = resolveResetWindowConfig(null) -): Promise { - const metrics = getComboMetrics(comboName); - const { getPricingForModel } = await import("../../src/lib/localDb"); - const quotaPromises = new Map>(); - let historicalLatencyStats: Record = {}; - try { - const { getModelLatencyStats } = await import("../../src/lib/usageDb"); - historicalLatencyStats = await getModelLatencyStats({ - windowHours: 24, - minSamples: 3, - maxRows: 10000, - }); - } catch { - // keep empty stats — auto-combo will use runtime + bootstrap signals - } - - const uniqueProviders = Array.from( - new Set( - targets.map((target) => target.provider || parseModel(target.modelStr).provider || "unknown") - ) - ); - const connectionPoolCounts = new Map(); - const connectionsByProvider = new Map>>(); - await Promise.all( - uniqueProviders.map(async (provider) => { - try { - const connections = await getProviderConnections({ provider, isActive: true }); - const active = Array.isArray(connections) ? connections : []; - connectionPoolCounts.set(provider, active.length); - connectionsByProvider.set(provider, active); - } catch { - connectionPoolCounts.set(provider, 0); - connectionsByProvider.set(provider, []); - } - }) - ); - - const expandedTargets: ResolvedComboTarget[] = []; - for (const target of targets) { - const provider = target.provider || parseModel(target.modelStr).provider || "unknown"; - const providerConnections = connectionsByProvider.get(provider) || []; - if (target.connectionId) { - expandedTargets.push(target); - continue; - } - const connectionIds = providerConnections - .map((c) => (c && typeof c === "object" && typeof c.id === "string" ? c.id : null)) - .filter((id): id is string => id !== null); - if (connectionIds.length === 0) { - expandedTargets.push(target); - continue; - } - for (const connectionId of connectionIds) { - expandedTargets.push({ - ...target, - connectionId, - executionKey: `${target.executionKey}@${connectionId}`, - }); - } - } - - const candidates = await Promise.all( - expandedTargets.map(async (target) => { - const modelStr = target.modelStr; - const parsed = parseModel(modelStr); - const provider = target.provider || parsed.provider || parsed.providerAlias || "unknown"; - const model = parsed.model || modelStr; - const historicalKey = `${provider}/${model}`; - const historicalModelMetric = historicalLatencyStats[historicalKey] || null; - const historicalTotal = Number(historicalModelMetric?.totalRequests); - const hasHistoricalSignal = - Number.isFinite(historicalTotal) && historicalTotal >= MIN_HISTORY_SAMPLES; - - let costPer1MTokens = 1; - try { - const pricing = await getPricingForModel(provider, model); - const inputPrice = Number(pricing?.input); - const outputPrice = Number(pricing?.output); - if (Number.isFinite(inputPrice) && inputPrice >= 0) { - if (Number.isFinite(outputPrice) && outputPrice >= 0) { - costPer1MTokens = - inputPrice * (1 - OUTPUT_TOKEN_RATIO) + outputPrice * OUTPUT_TOKEN_RATIO; - } else { - costPer1MTokens = inputPrice; - } - } - } catch { - // keep default cost - } - - const modelMetric = metrics?.byModel?.[modelStr] || null; - const avgLatency = Number(modelMetric?.avgLatencyMs); - const successRate = Number(modelMetric?.successRate); - const historicalP95Latency = Number(historicalModelMetric?.p95LatencyMs); - const historicalStdDev = Number(historicalModelMetric?.latencyStdDev); - const historicalSuccessRate = Number(historicalModelMetric?.successRate); // 0..1 - - const p95LatencyMs = hasHistoricalSignal - ? Number.isFinite(historicalP95Latency) && historicalP95Latency > 0 - ? historicalP95Latency - : getBootstrapLatencyMs(model) - : Number.isFinite(avgLatency) && avgLatency > 0 - ? avgLatency - : getBootstrapLatencyMs(model); - - const errorRate = hasHistoricalSignal - ? Number.isFinite(historicalSuccessRate) && - historicalSuccessRate >= 0 && - historicalSuccessRate <= 1 - ? 1 - historicalSuccessRate - : 0.05 - : Number.isFinite(successRate) && successRate >= 0 && successRate <= 100 - ? 1 - successRate / 100 - : 0.05; - const latencyStdDev = - hasHistoricalSignal && Number.isFinite(historicalStdDev) && historicalStdDev > 0 - ? Math.max(10, historicalStdDev) - : Math.max(10, p95LatencyMs * 0.1); - - const breakerStateRaw = getCircuitBreaker(provider)?.getStatus?.()?.state; - const circuitBreakerState: ProviderCandidate["circuitBreakerState"] = - breakerStateRaw === "OPEN" || breakerStateRaw === "HALF_OPEN" ? breakerStateRaw : "CLOSED"; - const contextAffinity = calculateTargetContextAffinity(target, sessionId); - let resetWindowAffinity = 0.5; - const fetcher = getQuotaFetcher(provider); - if (fetcher && target.connectionId) { - const quotaKey = `${provider}:${target.connectionId}`; - if (!quotaPromises.has(quotaKey)) { - quotaPromises.set( - quotaKey, - fetchResetAwareQuotaWithCache({ - provider, - connectionId: target.connectionId, - fetcher, - config: resetWindowConfig, - log: {}, - comboName, - }) - ); - } - const quota = await quotaPromises.get(quotaKey)!; - resetWindowAffinity = calculateResetWindowAffinity(quota, resetWindowConfig); - } - - return { - stepId: target.stepId, - executionKey: target.executionKey, - modelStr, - provider, - model, - quotaRemaining: 100, - quotaTotal: 100, - circuitBreakerState, - costPer1MTokens, - p95LatencyMs, - latencyStdDev, - errorRate, - accountTier: "standard" as const, - quotaResetIntervalSecs: 86400, - contextAffinity, - resetWindowAffinity, - connectionPoolSize: connectionPoolCounts.get(provider) ?? 1, - connectionId: target.connectionId ?? undefined, - }; - }) - ); - - return candidates; -} - -function dedupeTargetsByExecutionKey(targets: ResolvedComboTarget[]) { - const seen = new Set(); - return targets.filter((target) => { - if (seen.has(target.executionKey)) return false; - seen.add(target.executionKey); - return true; - }); -} - -async function applyRequestTagRouting( - targets: ResolvedComboTarget[], - body: Record | null | undefined, - log: { info?: (...args: unknown[]) => void; warn?: (...args: unknown[]) => void } -): Promise { - const { tags, matchMode } = resolveRequestRoutingTags(body); - if (tags.length === 0 || targets.length === 0) { - return targets; - } - - const providerIds = Array.from( - new Set(targets.map((target) => target.providerId || target.provider)) - ).filter( - (providerId): providerId is string => typeof providerId === "string" && providerId.length > 0 - ); - const providerConnections = new Map>>(); - - await Promise.all( - providerIds.map(async (providerId) => { - try { - const connections = await getProviderConnections({ provider: providerId, isActive: true }); - providerConnections.set( - providerId, - Array.isArray(connections) ? (connections as Array>) : [] - ); - } catch (error) { - log.warn?.( - "COMBO", - `Tag routing failed to load connections for provider=${providerId}: ${error instanceof Error ? error.message : String(error)}` - ); - providerConnections.set(providerId, []); - } - }) - ); - - const filteredTargets = targets.reduce((acc, target) => { - const providerKey = target.providerId || target.provider; - const candidateConnections = - providerConnections.get(providerKey)?.filter((connection) => { - const connectionId = - typeof connection.id === "string" && connection.id.trim().length > 0 - ? connection.id - : null; - if (!connectionId) return false; - if (target.connectionId) { - return connectionId === target.connectionId; - } - return true; - }) || []; - - const matchedConnectionIds = candidateConnections - .filter((connection) => - matchesRoutingTags( - getConnectionRoutingTags(connection.providerSpecificData), - tags, - matchMode - ) - ) - .map((connection) => connection.id) - .filter((connectionId): connectionId is string => typeof connectionId === "string"); - - if (matchedConnectionIds.length === 0) { - return acc; - } - - if (target.connectionId) { - acc.push(target); - return acc; - } - - acc.push({ - ...target, - allowedConnectionIds: Array.from(new Set(matchedConnectionIds)), - }); - return acc; - }, []); - - if (filteredTargets.length === 0) { - log.info?.( - "COMBO", - `Tag routing matched 0/${targets.length} targets for [${tags.join(", ")}] (${matchMode}); falling back to the full target set` - ); - return targets; - } - - log.info?.( - "COMBO", - `Tag routing matched ${filteredTargets.length}/${targets.length} targets for [${tags.join(", ")}] (${matchMode})` - ); - return filteredTargets; -} - -export function resolveComboTargets( - combo: ComboLike, - allCombos: ComboCollectionLike -): ResolvedComboTarget[] { - return allCombos ? resolveNestedComboTargets(combo, allCombos) : getDirectComboTargets(combo); -} - -function resolveWeightedTargets( - combo: ComboLike, - allCombos: ComboCollectionLike -): { - orderedTargets: ResolvedComboTarget[]; - selectedStep: ComboRuntimeStep | null; -} { - const topLevelSteps = getOrderedTopLevelRuntimeSteps(combo, allCombos); - if (topLevelSteps.length === 0) { - return { orderedTargets: [], selectedStep: null }; - } - - const selectedStep = selectWeightedTarget(topLevelSteps); - if (!selectedStep) { - return { orderedTargets: [], selectedStep: null }; - } - - const orderedSteps = orderTargetsForWeightedFallback( - topLevelSteps, - selectedStep.executionKey, - hasCompositeTierRuntimeOrder(combo) - ); - const expandedTargets = orderedSteps.flatMap((step) => { - if (!step) return []; - if (!allCombos) { - return step.kind === "model" ? [step] : []; - } - return expandRuntimeStep(step, allCombos, new Set([combo.name])); - }); - - return { - orderedTargets: dedupeTargetsByExecutionKey(expandedTargets), - selectedStep, - }; -} - -function scoreAutoTargets( - targets: ResolvedComboTarget[], - candidates: AutoProviderCandidate[], - taskType: string | null, - weights: ScoringWeights -) { - const candidateByExecutionKey = new Map( - candidates.map((candidate: ProviderCandidate & { executionKey: string }) => [ - candidate.executionKey, - candidate, - ]) - ); - return targets - .map((target) => { - const candidate = candidateByExecutionKey.get(target.executionKey); - if (!candidate) return null; - const factors = calculateFactors( - candidate as ProviderCandidate, - candidates, - taskType ?? "general", - getTaskFitness - ); - let score = calculateScore(factors, weights); - // B17: Quota Share soft-policy deprioritization - if ("quotaSoftPenalty" in candidate && candidate.quotaSoftPenalty === true) { - score *= QUOTA_SOFT_DEPRIORITIZE_FACTOR; - } - return { - target, - score, - }; - }) - .filter((entry): entry is { target: ResolvedComboTarget; score: number } => entry !== null) - .sort((a, b) => b.score - a.score); -} - -/** - * For an auto-combo WITHOUT an explicit `candidatePool`, broaden the eligible - * targets to every model of every active provider connection so the router has - * the full pool to score over. Already-present `modelStr`s are not duplicated. - * - * Best-effort: if loading active connections or provider models throws, the - * explicitly-resolved targets are returned unchanged (the combo still runs). - * Exported for unit testing. Mutates and returns `eligibleTargets`. - */ -export async function expandAutoComboCandidatePool( - eligibleTargets: ResolvedComboTarget[], - combo: { autoConfig?: unknown; config?: unknown } | null | undefined -): Promise { - const localAutoConfig = - (combo?.autoConfig as Record | undefined) || - (isRecord((combo?.config as Record)?.auto) - ? ((combo?.config as Record).auto as Record) - : null) || - (combo?.config as Record | undefined) || - {}; - - if (Array.isArray(localAutoConfig?.candidatePool)) return eligibleTargets; - - try { - const allConnections = await getProviderConnections({ isActive: true }); - const providerIds = [ - ...new Set( - (allConnections as Array<{ provider?: unknown }>) - .map((c) => c.provider) - .filter((p): p is string => typeof p === "string" && p.length > 0) - ), - ]; - for (const providerId of providerIds) { - const providerModels = getProviderModels(providerId); - for (const model of providerModels) { - const modelStr = `${providerId}/${model.id}`; - if (!eligibleTargets.some((t) => t.modelStr === modelStr)) { - eligibleTargets.push({ - kind: "model", - stepId: modelStr, - executionKey: modelStr, - provider: providerId, - providerId: providerId, - modelStr, - weight: 1, - connectionId: null, - label: null, - }); - } - } - } - } catch { - // Best-effort candidate expansion only: if loading active connections or - // provider models fails, fall back to the explicitly-resolved targets - // rather than aborting the combo. The push above is the only mutation, - // so a throw leaves eligibleTargets exactly as explicit resolution built it. - } - - return eligibleTargets; -} - -/** - * Handle combo chat with fallback. - * @param {Object} options - * @param {Object} options.body - Request body - * @param {Object} options.combo - Full combo object { name, models, strategy, config } - * @param {Function} options.handleSingleModel - Function: (body, modelStr) => Promise - * @param {Function} [options.isModelAvailable] - Optional pre-check: (modelStr) => Promise - * @param {Object} options.log - Logger object - * @returns {Promise} - */ -/** @param {object} options */ -export async function handleComboChat({ - body, - combo, - handleSingleModel, - isModelAvailable, - log, - settings, - allCombos, - relayOptions, - signal, - apiKeyAllowedConnections = null, -}: HandleComboChatOptions): Promise { - const strategy = normalizeRoutingStrategy(combo.strategy || "priority"); - const relayConfig = - strategy === "context-relay" ? resolveContextRelayConfig(relayOptions?.config || null) : null; - - const resilienceSettings: ResilienceSettings = settings - ? resolveResilienceSettings(settings) - : resolveResilienceSettings(null); - - const universalHandoffConfig = resolveUniversalHandoffConfig( - (combo.universal_handoff || combo.universalHandoff) as - | Record - | null - | undefined, - relayOptions?.universalHandoffConfig as Record | null | undefined - ); - // ── Server-side context cache pinning (replaces tag roundtrip) ─ - // Uses session_model_history — no client-side tag injection, no visible output pollution. - let pinnedModel: string | null = null; - if ( - combo.context_cache_protection && - relayOptions?.sessionId && - !(body as Record)?.[SKIP_UNIVERSAL_HANDOFF_FLAG] - ) { - const pinned = getLastSessionModel(relayOptions.sessionId, combo.name); - if (pinned) { - body = { ...body, model: pinned }; - pinnedModel = pinned; - log.info("COMBO", `[#401] Context cache: pinned model=${pinned} (server-side)`); - } - } - - // ── Combo Agent Middleware (#399 + #401) ──────────────────────────────── - // Apply system_message override, tool_filter_regex. - // Context cache pinning is handled above via session_model_history. - const { body: agentBody } = applyComboAgentMiddleware( - body, - combo, - "" // provider/model not yet known — resolved per-model in loop - ); - body = agentBody; - const clientRequestedStream = body?.stream === true; - // Context cache pinning is handled above via server-side session_model_history. - // No tag injection on response — use handleSingleModel directly. - // ───────────────────────────────────────────────────────────────────────── - - // Use config cascade before dispatch so all strategies, pinned context routes, - // and round-robin targets share the same timeout policy. - const config = settings - ? resolveComboConfig(combo, settings) - : { ...getDefaultComboConfig(), ...(combo.config || {}) }; - const comboTargetTimeoutMs = resolveComboTargetTimeoutMs(config, FETCH_TIMEOUT_MS); - - // ── Per-model timeout wrapper ──────────────────────────────────────────── - // Combo target timeouts inherit FETCH_TIMEOUT_MS by default. Operators can - // configure targetTimeoutMs to shorten fallback latency, but never to extend - // beyond the current upstream request timeout. - // - // The timeoutController is forwarded to the inner caller via target.modelAbortSignal. - // When the timeout fires we (a) resolve the race with a synthetic 524 and - // (b) abort the inner request so its upstream fetch is cancelled and downstream - // cooldown/breaker/usage mutations stop — preventing "ghost" state mutations - // that diverge from the routing decision the operator sees. - const handleSingleModelWithTimeout = async ( - b: Record, - modelStr: string, - target?: SingleModelTarget - ): Promise => { - if (comboTargetTimeoutMs <= 0) { - return handleSingleModel(b, modelStr, target).catch((err) => - errorResponse(502, err?.message ?? "Upstream model error") - ); - } - - const timeoutController = new AbortController(); - let timeoutId: ReturnType | undefined; - let timedOut = false; - const timeoutPromise = new Promise((resolve) => { - timeoutId = setTimeout(() => { - timedOut = true; - log.warn( - "COMBO", - `Model ${modelStr} exceeded ${comboTargetTimeoutMs}ms timeout — falling back` - ); - // Abort the inner request so its upstream fetch is cancelled and - // downstream cooldown/breaker/usage mutations don't continue mutating - // state behind the routing decision's back. - timeoutController.abort(new Error("combo-per-model-timeout")); - resolve( - new Response(JSON.stringify({ error: { message: `Model ${modelStr} timed out` } }), { - status: 524, - headers: { "Content-Type": "application/json" }, - }) - ); - }, comboTargetTimeoutMs); - }); - const targetWithSignal = { - ...(target ?? {}), - modelAbortSignal: timeoutController.signal, - }; - if (target?.modelAbortSignal) { - if (target.modelAbortSignal.aborted) { - timeoutController.abort(new Error("hedge-cancelled")); - } else { - target.modelAbortSignal.addEventListener("abort", () => { - timeoutController.abort(new Error("hedge-cancelled")); - }); - } - } - try { - return await Promise.race([ - handleSingleModel(b, modelStr, targetWithSignal).catch((err) => { - if (timedOut) { - // Inner call rejected because we aborted it. The synthetic 524 from - // timeoutPromise already wins the race; return an empty response so - // the loser branch resolves cleanly without leaking err.message. - return new Response(null, { status: 599 }); - } - return errorResponse(502, err?.message ?? "Upstream model error"); - }), - timeoutPromise, - ]); - } finally { - clearTimeout(timeoutId); - } - }; - - // Route to pinned model if context caching specifies one (Fix #679) - if (pinnedModel) { - log.info( - "COMBO", - `Bypassing strategy — routing directly to pinned context model: ${pinnedModel}` - ); - return handleSingleModelWithTimeout(body, pinnedModel); - } - - // Route to round-robin handler if strategy matches - if (strategy === "round-robin") { - return handleRoundRobinCombo({ - body, - combo, - handleSingleModel: handleSingleModelWithTimeout, - isModelAvailable, - log, - settings, - allCombos, - signal, - }); - } - - const maxRetries = config.maxRetries ?? 1; - const retryDelayMs = resolveDelayMs(config.retryDelayMs, 2000); - const fallbackDelayMs = resolveDelayMs(config.fallbackDelayMs, 0); - const maxSetRetries = config.maxSetRetries ?? 0; - const setRetryDelayMs = resolveDelayMs(config.setRetryDelayMs, 2000); - - let orderedTargets = - strategy === "weighted" - ? resolveWeightedTargets(combo, allCombos)?.orderedTargets || [] - : resolveComboTargets(combo, allCombos); - - orderedTargets = await applyRequestTagRouting(orderedTargets, body, log); - - if (strategy === "weighted") { - log.info( - "COMBO", - `Weighted selection${allCombos ? " with nested resolution" : ""}: ${orderedTargets.length} total targets` - ); - } else if (allCombos) { - log.info("COMBO", `${strategy} with nested resolution: ${orderedTargets.length} total targets`); - } - - // Pipeline dispatch: route smart/pipeline-enabled combos through the multi-stage pipeline - if (strategy === "auto") { - const autoParsed = parseAutoPrefix(combo.name); - const autoVariant = autoParsed.valid ? autoParsed.variant : undefined; - if (autoVariant === "smart" || config.pipeline_enabled) { - try { - const pipelineRaw = await handlePipelineCombo({ - body, - combo, - handleChatCore: handleSingleModelWithTimeout, - log: { - info: log.info, - warn: log.warn, - error: log.error ?? log.warn, - }, - settings: settings ?? {}, - signal: signal ?? undefined, - }); - // handlePipelineCombo resolves to a PipelineResult (buffered text) or, - // in the streaming-final-stage case, a Response. Callers downstream - // (chat.ts → withSessionHeader) require a Response, so adapt the - // PipelineResult here instead of leaking the raw object. - return pipelineRaw instanceof Response - ? pipelineRaw - : buildPipelineResponse(pipelineRaw, body); - } catch (pipelineErr) { - const pipelineMsg = pipelineErr instanceof Error ? pipelineErr.message : ""; - if (pipelineMsg === "PIPELINE_DISABLED") { - log.info("COMBO", "Pipeline disabled, falling through to standard auto routing"); - } else if (pipelineMsg === "PIPELINE_TOKEN_THRESHOLD") { - log.info( - "COMBO", - "Pipeline skipped (prompt below token threshold), falling through to standard auto routing" - ); - } else { - log.warn("COMBO", "Pipeline dispatch failed, falling through to standard auto routing", { - err: pipelineErr, - }); - } - } - } - } - - if (strategy === "auto") { - const requestHasTools = Array.isArray(body?.tools) && body.tools.length > 0; - let eligibleTargets = [...orderedTargets]; - - if (requestHasTools) { - const filtered = eligibleTargets.filter((target) => supportsToolCalling(target.modelStr)); - if (filtered.length > 0) { - eligibleTargets = filtered; - } else { - log.warn( - "COMBO", - "Auto strategy: all candidates filtered by tool-calling policy, falling back to full pool" - ); - } - } - - // Context-window pre-filter (#1808) - // Estimate input tokens once; exclude candidates whose known context limit is too small. - // Uses the same 4-chars-per-token heuristic as contextManager.ts::compressContext(). - // Null/unknown limits are treated as "include" to avoid incorrectly dropping valid targets. - const requestMessages = body.messages; - const estimatedInputTokens = estimateTokens( - typeof requestMessages === "string" || - (requestMessages !== null && typeof requestMessages === "object") - ? requestMessages - : [] - ); - if (estimatedInputTokens > 0) { - const filteredByContext = eligibleTargets.filter((target) => { - const limit = getModelContextLimitForModelString(target.modelStr); - if (limit === null || limit === undefined) return true; // unknown — include to be safe - return limit >= estimatedInputTokens; - }); - if (filteredByContext.length > 0) { - log.debug?.( - "COMBO", - `Auto strategy: context-window filter kept ${filteredByContext.length}/${eligibleTargets.length} candidates (est. ${estimatedInputTokens} tokens)` - ); - eligibleTargets = filteredByContext; - } else { - log.warn( - "COMBO", - `Auto strategy: all candidates filtered by context-window policy (est. ${estimatedInputTokens} tokens), falling back to full pool` - ); - // eligibleTargets intentionally unchanged — same fallback contract as tool-calling filter - } - - eligibleTargets = await expandAutoComboCandidatePool(eligibleTargets, combo); - } - - const prompt = extractPromptForIntent(body); - const systemPrompt = - typeof combo?.system_message === "string" ? combo.system_message : undefined; - const intentConfig = getIntentConfig(settings, combo); - const intent = classifyWithConfig(prompt, intentConfig, systemPrompt); - recordComboIntent(combo.name, intent); - const taskType = mapIntentToTaskType(intent); - - const rawAutoConfigSource = - combo?.autoConfig || - (isRecord(combo?.config?.auto) ? combo.config.auto : null) || - combo?.config || - {}; - const autoConfigSource: Record = isRecord(rawAutoConfigSource) - ? rawAutoConfigSource - : {}; - const routingStrategy = - typeof autoConfigSource.routerStrategy === "string" - ? autoConfigSource.routerStrategy - : typeof autoConfigSource.routingStrategy === "string" - ? autoConfigSource.routingStrategy - : typeof autoConfigSource.strategyName === "string" - ? autoConfigSource.strategyName - : "rules"; - - const candidatePool = Array.isArray(autoConfigSource.candidatePool) - ? autoConfigSource.candidatePool - : [...new Set(eligibleTargets.map((target) => target.provider))]; - - const weights = - autoConfigSource.weights && typeof autoConfigSource.weights === "object" - ? (autoConfigSource.weights as ScoringWeights) - : DEFAULT_WEIGHTS; - const explorationRate = Number.isFinite(Number(autoConfigSource.explorationRate)) - ? Number(autoConfigSource.explorationRate) - : 0.05; - const budgetCap = Number.isFinite(Number(autoConfigSource.budgetCap)) - ? Number(autoConfigSource.budgetCap) - : undefined; - const modePack = - typeof autoConfigSource.modePack === "string" ? autoConfigSource.modePack : undefined; - const resetWindowConfig = resolveResetWindowConfig(autoConfigSource); - const slaPolicy = resolveSlaRoutingPolicy(autoConfigSource); - - let lastKnownGoodProvider: string | undefined; - try { - const { getLKGP } = await import("../../src/lib/localDb"); - const lkgp = await getLKGP(combo.name, combo.id || combo.name); - if (lkgp) lastKnownGoodProvider = lkgp.provider; - } catch (err) { - log.warn("COMBO", "Failed to retrieve Last Known Good Provider. This is non-fatal.", { err }); - } - - const candidates = await buildAutoCandidates( - eligibleTargets, - combo.name, - relayOptions?.sessionId, - resetWindowConfig - ); - // G2: Register candidates so chatCore can mark quotaSoftPenalty via setCandidateQuotaSoftPenalty. - _registerExecutionCandidates(candidates); - if (candidates.length > 0) { - let selectedProvider: string | null = null; - let selectedModel: string | null = null; - let selectionReason = ""; - - if (routingStrategy !== "rules") { - try { - const decision = selectWithStrategy( - candidates, - { - taskType, - requestHasTools, - lastKnownGoodProvider, - estimatedInputTokens, - sla: slaPolicy, - }, - routingStrategy - ); - selectedProvider = decision.provider; - selectedModel = decision.model; - selectionReason = decision.reason; - } catch (err) { - log.warn( - "COMBO", - `Auto strategy '${routingStrategy}' failed (${err?.message || "unknown"}), falling back to rules` - ); - } - } - - if (!selectedProvider || !selectedModel) { - const selection = selectAutoProvider( - { - id: combo.id || combo.name, - name: combo.name, - type: "auto", - candidatePool, - weights, - modePack, - budgetCap, - explorationRate, - }, - candidates, - taskType - ); - selectedProvider = selection.provider; - selectedModel = selection.model; - selectionReason = `score=${selection.score.toFixed(3)}${selection.isExploration ? " (exploration)" : ""}`; - } - - const scoredTargets = scoreAutoTargets(eligibleTargets, candidates, taskType, weights); - const rankedTargets = scoredTargets.map((entry) => entry.target); - const selectedTarget = - scoredTargets.find((entry) => { - const parsed = parseModel(entry.target.modelStr); - const modelId = parsed.model || entry.target.modelStr; - return entry.target.provider === selectedProvider && modelId === selectedModel; - })?.target || - rankedTargets[0] || - eligibleTargets[0]; - - orderedTargets = dedupeTargetsByExecutionKey( - [selectedTarget, ...rankedTargets, ...eligibleTargets].filter( - (entry): entry is ResolvedComboTarget => entry !== undefined && entry !== null - ) - ); - - log.info( - "COMBO", - `Auto selection: ${selectedTarget?.modelStr || `${selectedProvider}/${selectedModel}`} | intent=${intent} task=${taskType} | strategy=${routingStrategy} | ${selectionReason}` - ); - } else { - log.warn("COMBO", "Auto strategy has no candidates, keeping default ordering"); - } - } else if (strategy === "lkgp") { - try { - const { getLKGP } = await import("../../src/lib/localDb"); - const lkgpProvider = await getLKGP(combo.name, combo.id || combo.name); - - if (lkgpProvider) { - const lkgpRecord = lkgpProvider; - const providerName = lkgpRecord.provider; - const connId = lkgpRecord.connectionId; - - let lkgpIndex = -1; - if (connId) { - lkgpIndex = orderedTargets.findIndex( - (target) => target.provider === providerName && target.connectionId === connId - ); - } - if (lkgpIndex < 0) { - lkgpIndex = orderedTargets.findIndex( - (target) => - target.provider === providerName || - // Issue #2359: Defensive guard. The `target.modelStr` type - // annotation is `string`, but malformed combo entries (e.g., - // local-provider rows whose `modelStr` failed to resolve when - // the executor catalogue was being rebuilt) have leaked - // through and surfaced as `e.startsWith is not a function` - // 500s on combo test/dispatch. The fast path stays - // unchanged for the common case; this only avoids the - // crash when the field is unexpectedly non-string. - (typeof target.modelStr === "string" && - target.modelStr.startsWith(`${providerName}/`)) - ); - } - - if (lkgpIndex > 0) { - const [lkgpTarget] = orderedTargets.splice(lkgpIndex, 1); - orderedTargets.unshift(lkgpTarget); - log.info( - "COMBO", - `[LKGP] Prioritizing last known good provider ${providerName}${connId ? ` (account ${connId})` : ""} for combo "${combo.name}"` - ); - } else if (lkgpIndex === 0) { - log.debug?.( - "COMBO", - `[LKGP] Last known good provider ${providerName}${connId ? ` (account ${connId})` : ""} already first for combo "${combo.name}"` - ); - } - } - } catch (err) { - log.warn("COMBO", "Failed to retrieve Last Known Good Provider. This is non-fatal.", { err }); - } - } else if (strategy === "strict-random") { - const selectedExecutionKey = await getNextFromDeck( - `combo:${combo.name}`, - orderedTargets.map((target) => target.executionKey) - ); - const selectedTarget = - orderedTargets.find((target) => target.executionKey === selectedExecutionKey) || null; - const rest = orderedTargets.filter((target) => target.executionKey !== selectedExecutionKey); - orderedTargets = [selectedTarget, ...rest].filter( - (target): target is ResolvedComboTarget => target !== null - ); - log.info( - "COMBO", - `Strict-random deck: ${selectedExecutionKey} selected (${orderedTargets.length} targets)` - ); - } else if (strategy === "random") { - orderedTargets = fisherYatesShuffle([...orderedTargets]); - log.info("COMBO", `Random shuffle: ${orderedTargets.length} targets`); - } else if (strategy === "fill-first") { - log.info( - "COMBO", - `Fill-first ordering: preserving priority order (${orderedTargets.length} targets)` - ); - } else if (strategy === "p2c") { - orderedTargets = orderTargetsByPowerOfTwoChoices(orderedTargets, combo.name); - log.info("COMBO", `Power-of-two-choices ordering: selected ${orderedTargets[0]?.modelStr}`); - } else if (strategy === "least-used") { - orderedTargets = sortTargetsByUsage(orderedTargets, combo.name); - log.info("COMBO", `Least-used ordering: ${orderedTargets[0]?.modelStr} has fewest requests`); - } else if (strategy === "cost-optimized") { - orderedTargets = await sortTargetsByCost(orderedTargets); - if (config.manifestRouting === true) { - try { - const manifestHint = generateRoutingHints( - orderedTargets.filter((t) => t.kind === "model"), - { - messages: Array.isArray(body?.messages) - ? (body.messages as Array<{ role?: string; content?: string | unknown }>) - : [], - tools: Array.isArray(body?.tools) - ? (body.tools as Array<{ - function?: { name: string; description?: string; parameters?: unknown }; - }>) - : undefined, - model: typeof body?.model === "string" ? body.model : undefined, - } - ); - if (manifestHint.strategyModifier === "require-premium") { - const eligible = orderedTargets.filter( - (t) => - t.kind !== "model" || - manifestHint.eligibleTargets.some( - (e) => e.provider === t.provider && e.modelStr === t.modelStr - ) - ); - if (eligible.length > 0) orderedTargets = eligible; - } - log.debug?.( - { - strategyModifier: manifestHint.strategyModifier, - specificityLevel: manifestHint.specificityLevel, - score: manifestHint.specificity.score, - }, - "manifest routing applied" - ); - } catch (err) { - log.warn({ err }, "manifest routing failed, falling back to standard strategy"); - } - } - log.info("COMBO", `Cost-optimized ordering: cheapest first (${orderedTargets[0]?.modelStr})`); - } else if (strategy === "reset-aware") { - orderedTargets = await orderTargetsByResetAwareQuota( - orderedTargets, - combo.name, - config, - log, - apiKeyAllowedConnections - ); - log.info( - "COMBO", - `Reset-aware ordering: ${orderedTargets[0]?.modelStr}${orderedTargets[0]?.connectionId ? ` (${orderedTargets[0].connectionId})` : ""} first` - ); - } else if (strategy === "reset-window") { - orderedTargets = await orderTargetsByResetWindow( - orderedTargets, - combo.name, - config, - log, - apiKeyAllowedConnections - ); - log.info( - "COMBO", - `Reset-window ordering: ${orderedTargets[0]?.modelStr}${orderedTargets[0]?.connectionId ? ` (${orderedTargets[0].connectionId})` : ""} first` - ); - } else if (strategy === "context-optimized") { - orderedTargets = sortTargetsByContextSize(orderedTargets); - log.info("COMBO", `Context-optimized ordering: largest first (${orderedTargets[0]?.modelStr})`); - } - - orderedTargets = orderTargetsByEvalScores(orderedTargets, config.evalRouting, log); - orderedTargets = filterTargetsByRequestCompatibility(orderedTargets, body, log); - - // Parallel pre-screen: check provider profiles and model availability for all targets - // Only runs for priority strategy where sequential checking causes latency - const preScreenMap = - strategy === "priority" - ? await preScreenTargets(orderedTargets, isModelAvailable).catch( - () => new Map() - ) - : new Map(); - - if (orderedTargets.length === 0) { - return comboModelNotFoundResponse("Combo has no executable targets"); - } - - scheduleShadowRouting( - combo, - config, - body, - resolveShadowTargets(combo, config, allCombos), - handleSingleModel, - isModelAvailable, - strategy, - log - ); - - // G2: Collect execution keys registered by _registerExecutionCandidates above (auto strategy). - // We snapshot them now so cleanup can happen after the attempt loop finishes. - const _registeredExecutionKeys = orderedTargets.map((t) => t.executionKey).filter(Boolean); - - let globalAttempts = 0; - - try { - for (let setTry = 0; setTry <= maxSetRetries; setTry++) { - // #1731: Per-set-iteration set of providers whose quota is fully exhausted. - // Reset each retry so providers excluded in a previous attempt get another chance. - const exhaustedProviders = new Set(); - const exhaustedConnections = new Set(); - const transientRateLimitedProviders = new Set(); - if (setTry > 0) { - log.info("COMBO", `All targets failed — retrying set (${setTry}/${maxSetRetries})`); - await new Promise((resolve) => { - const timer = setTimeout(resolve, setRetryDelayMs); - signal?.addEventListener( - "abort", - () => { - clearTimeout(timer); - resolve(undefined); - }, - { once: true } - ); - }); - if (signal?.aborted) { - log.info("COMBO", "Client disconnected during set retry delay — aborting"); - return errorResponse(499, "Client disconnected"); - } - } - - let lastError: string | null = null; - let earliestRetryAfter: ComboRetryAfter | null = null; - let lastStatus: number | null = null; - const startTime = Date.now(); - let fallbackCount = 0; - let recordedAttempts = 0; - - let globalResolve: ((res: Response) => void) | null = null; - const globalPromise = new Promise((res) => { - globalResolve = res; - }); - const runningTasks = new Set>(); - let anySuccess = false; - const abortControllers = new Map(); - const zeroLatencyOptimizationsEnabled = config.zeroLatencyOptimizationsEnabled === true; - - const executeTarget = async ( - i: number - ): Promise<{ ok: boolean; response?: Response } | null> => { - const target = orderedTargets[i]; - const modelStr = target.modelStr; - const provider = target.provider; - - const cb = getCircuitBreaker(provider); - if (cb.getStatus().state === "OPEN") { - log.info("COMBO", `Skipping ${modelStr} — circuit breaker OPEN for ${provider}`); - if (i > 0) fallbackCount++; - return null; - } - - if ( - resilienceSettings.providerCooldown.enabled && - Boolean(provider && provider !== "unknown") && - isProviderInCooldown(provider, target.connectionId ?? undefined, resilienceSettings) - ) { - log.info("COMBO", `Skipping ${modelStr} — provider ${provider} in global cooldown`); - if (i > 0) fallbackCount++; - return null; - } - - // Use pre-screened profile if available, otherwise fetch on demand - const preScreenEntry = preScreenMap.get(target.executionKey); - const profile = preScreenEntry?.profile ?? (await getRuntimeProviderProfile(provider)); - - const allowRateLimitedConnection = - Boolean(provider && provider !== "unknown") && - transientRateLimitedProviders.has(provider); - const targetForAttempt = allowRateLimitedConnection - ? { - ...target, - allowRateLimitedConnection: true, - modelAbortSignal: abortControllers.get(i)!.signal, - } - : { ...target, modelAbortSignal: abortControllers.get(i)!.signal }; - - // #1731v2: Skip targets whose provider:connection pair had a connection-level error. - if (provider && target.connectionId) { - const connKey = `${provider}:${target.connectionId}`; - if (exhaustedConnections.has(connKey)) { - log.info( - "COMBO", - `Skipping ${modelStr} — connection ${target.connectionId} for provider ${provider} had connection error (#1731v2)` - ); - if (i > 0) fallbackCount++; - return null; - } - } - // #1731: Skip targets from a provider that already signaled full quota exhaustion this request. - if (provider && exhaustedProviders.has(provider)) { - log.info( - "COMBO", - `Skipping ${modelStr} — provider ${provider} marked exhausted this request (#1731)` - ); - if (i > 0) fallbackCount++; - return null; - } - - // Pre-screen may have already determined this target unavailable (e.g. - // circuit-breaker OPEN at resolve time). Skip immediately in that case. - // For targets pre-screened as "available" we still call isModelAvailable - // below because connection cooldowns (rateLimitedUntil) can change - // mid-request after a same-provider failure — the pre-screen snapshot is - // stale by the time we reach the 2nd/3rd same-provider target. - const preCheckedAvailable = preScreenEntry?.available ?? null; - if (preCheckedAvailable === false) { - log.info("COMBO", `Skipping ${modelStr} — pre-screen marked unavailable`); - if (i > 0) fallbackCount++; - return null; - } - if (isModelAvailable) { - const available = await isModelAvailable(modelStr, targetForAttempt); - if (!available) { - log.debug?.( - "COMBO", - `Skipping ${modelStr} — no credentials available or model excluded` - ); - if (i > 0) fallbackCount++; - return null; - } - } - - // Credential gate: skip targets with known-bad credentials (fail-fast) - const connectionId = target.connectionId as string | undefined; - if (connectionId) { - const gateResult = checkCredentialGate(connectionId, provider, modelStr); - if (gateResult.allowed === false) { - logCredentialSkip(log, modelStr, gateResult.reason || "Credential gate blocked"); - if (i > 0) fallbackCount++; - return null; - } - } - - // Retry loop for transient errors - for (let retry = 0; retry <= maxRetries; retry++) { - // Fix #1681: Bail out immediately if the client has disconnected - if (signal?.aborted) { - log.info("COMBO", `Client disconnected — aborting combo loop before model ${modelStr}`); - return { ok: false, response: errorResponse(499, "Client disconnected") }; - } - globalAttempts++; - if (globalAttempts > MAX_GLOBAL_ATTEMPTS) { - log.warn( - "COMBO", - `Maximum combo attempts (${MAX_GLOBAL_ATTEMPTS}) exceeded across all targets and fallbacks. Terminating loop to prevent runaway background requests.` - ); - return { ok: false, response: errorResponse(503, "Maximum combo retry limit reached") }; - } - - // Predictive TTFT Circuit Breaker (skip slow models) - if ( - zeroLatencyOptimizationsEnabled && - config.predictiveTtftMs && - config.predictiveTtftMs > 0 && - retry === 0 - ) { - const cMetrics = getComboMetrics(combo.name); - if (cMetrics) { - const targetKey = orderedTargets[i].executionKey || modelStr; - const m = cMetrics.byTarget[targetKey] || cMetrics.byModel[modelStr]; - if (m && m.requests >= 5 && m.avgLatencyMs > config.predictiveTtftMs) { - log.warn( - "COMBO", - `Predictive TTFT Circuit Breaker: skipping ${modelStr} (avg ${m.avgLatencyMs}ms > max ${config.predictiveTtftMs}ms)` - ); - return null; - } - } - } - - if (retry > 0) { - log.info( - "COMBO", - `Retrying ${modelStr} in ${retryDelayMs}ms (attempt ${retry + 1}/${maxRetries + 1})` - ); - await new Promise((resolve) => { - const timer = setTimeout(resolve, retryDelayMs); - signal?.addEventListener( - "abort", - () => { - clearTimeout(timer); - resolve(undefined); - }, - { once: true } - ); - }); - if (signal?.aborted) { - log.info("COMBO", `Client disconnected during retry delay — aborting`); - return { ok: false, response: errorResponse(499, "Client disconnected") }; - } - } - - log.info( - "COMBO", - `Trying model ${i + 1}/${orderedTargets.length}: ${modelStr}${retry > 0 ? ` (retry ${retry})` : ""}` - ); - emit("combo.target.attempt", { - comboName: combo.name, - targetIndex: i, - provider, - model: modelStr, - timestamp: Date.now(), - strategy, - }); - - // Deep clone the body to ensure context preservation and prevent mutations - // from affecting other targets in the combo - let attemptBody = JSON.parse(JSON.stringify(body)); - - // Proactive Context Compression for fallbacks (Zero-Latency optimization) - if ( - zeroLatencyOptimizationsEnabled && - i > 0 && - config.fallbackCompressionMode && - config.fallbackCompressionMode !== "off" - ) { - const { estimateTokens } = await import("./contextManager.ts"); - const estimatedTokens = estimateTokens(JSON.stringify(attemptBody)); - if (estimatedTokens > (config.fallbackCompressionThreshold ?? 1000)) { - const { applyCompression } = await import("./compression/strategySelector.ts"); - const compressionResult = applyCompression( - attemptBody, - config.fallbackCompressionMode as CompressionMode, - { model: modelStr } - ); - if (compressionResult.compressed) { - log.info( - "COMBO", - `Proactive fallback compression applied (${config.fallbackCompressionMode}): ${estimatedTokens} -> ${compressionResult.stats?.compressedTokens} tokens` - ); - attemptBody = compressionResult.body; - } - } - } - - // Universal handoff: inject existing handoff if model changed - if ( - universalHandoffConfig.enabled && - relayOptions?.sessionId && - !(body as Record)?.[SKIP_UNIVERSAL_HANDOFF_FLAG] - ) { - const lastModel = getLastSessionModel(relayOptions.sessionId, combo.name); - if (lastModel && lastModel !== modelStr) { - const existingHandoff = getHandoff(relayOptions.sessionId, combo.name); - attemptBody = injectUniversalHandoffBody( - attemptBody, // Use the cloned body to maintain isolation - lastModel, - modelStr, - `Model routing: ${lastModel} → ${modelStr}`, - existingHandoff - ); - } - } - - // Issue #3587: Reasoning models (deepseek-v4-flash, nemotron, etc.) consume - // ALL max_tokens for reasoning_tokens, leaving content empty. Add a buffer - // to max_tokens so the model has enough tokens for both reasoning and content. - if (supportsReasoning(modelStr)) { - const bodyRecord = attemptBody as Record; - const currentMaxTokens = Number(bodyRecord.max_tokens) || 0; - if (currentMaxTokens > 0) { - // Add 50% buffer + 1000 floor to ensure reasoning + content both fit - const bufferedMaxTokens = Math.max( - currentMaxTokens + 1000, - Math.ceil(currentMaxTokens * 1.5) - ); - bodyRecord.max_tokens = bufferedMaxTokens; - log.info( - "COMBO", - `Reasoning model ${modelStr}: buffered max_tokens ${currentMaxTokens} -> ${bufferedMaxTokens}` - ); - } - } - const result = await handleSingleModelWithTimeout(attemptBody, modelStr, { - ...targetForAttempt, - failoverBeforeRetry: config.failoverBeforeRetry, - }); - - // Success — validate response quality before returning - if (result.ok) { - const quality = await validateResponseQuality(result, clientRequestedStream, log); - if (!quality.valid) { - log.warn( - "COMBO", - `Model ${modelStr} returned 200 but failed quality check: ${quality.reason}` - ); - recordComboRequest(combo.name, modelStr, { - success: false, - latencyMs: Date.now() - startTime, - fallbackCount, - strategy, - target: toRecordedTarget(target), - }); - recordedAttempts++; - // Fix #1707: Set terminal state so the fallback doesn't emit - // misleading ALL_ACCOUNTS_INACTIVE when the real issue is quality. - lastError = `Upstream response failed quality validation: ${quality.reason}`; - if (!lastStatus) lastStatus = 502; - if (i > 0) fallbackCount++; - emit("combo.target.failed", { - comboName: combo.name, - targetIndex: i, - provider, - model: modelStr, - error: `Quality: ${quality.reason}`, - latencyMs: Date.now() - startTime, - }); - return null; - } - const latencyMs = Date.now() - startTime; - emit("combo.target.succeeded", { - comboName: combo.name, - targetIndex: i, - provider, - model: modelStr, - latencyMs, - }); - log.info( - "COMBO", - `Model ${modelStr} succeeded (${latencyMs}ms, ${fallbackCount} fallbacks)` - ); - recordComboRequest(combo.name, modelStr, { - success: true, - latencyMs, - fallbackCount, - strategy, - target: toRecordedTarget(target), - }); - recordedAttempts++; - - // Reset cooldown on success - if (provider && provider !== "unknown") { - recordProviderSuccess(provider, target.connectionId ?? undefined); - } - // Webhook fan-out: best-effort, never blocks the response stream. - notifyWebhookEvent("request.completed", { - combo: combo.name, - provider, - model: modelStr, - latencyMs, - fallbackCount, - }); - - // Context cache pinning: record model usage for session-based pinning - // (independent of universal handoff — always fires when context_cache_protection is on) - if ( - combo.context_cache_protection && - relayOptions?.sessionId && - !(body as Record)?.[SKIP_UNIVERSAL_HANDOFF_FLAG] - ) { - recordSessionModelUsage( - relayOptions.sessionId, - combo.name, - modelStr, - provider, - target.connectionId ?? undefined - ); - } - - // Universal handoff: record model usage for session - if ( - universalHandoffConfig.enabled && - relayOptions?.sessionId && - !(body as Record)?.[SKIP_UNIVERSAL_HANDOFF_FLAG] - ) { - const prevModel = getLastSessionModel(relayOptions.sessionId, combo.name); - recordSessionModelUsage( - relayOptions.sessionId, - combo.name, - modelStr, - provider, - target.connectionId ?? undefined - ); - if (prevModel && prevModel !== modelStr) { - const handoffSourceMessages = - Array.isArray(body?.messages) && body.messages.length > 0 - ? body.messages - : Array.isArray(body?.input) - ? body.input - : []; - - maybeGenerateUniversalHandoff({ - sessionId: relayOptions.sessionId, - comboName: combo.name, - messages: handoffSourceMessages as MessageLike[], - prevModel, - currModel: modelStr, - universalConfig: universalHandoffConfig, - handleSingleModel: handleSingleModelWithTimeout, - }); - } - - recordSessionModelUsage( - relayOptions.sessionId, - combo.name, - modelStr, - provider, - target.connectionId ?? undefined - ); - } - // Context-relay intentionally splits responsibilities: - // combo.ts decides whether a successful turn should generate a handoff, - // while chat.ts injects the handoff after the real connectionId is resolved. - if ( - strategy === "context-relay" && - relayOptions?.sessionId && - relayConfig && - relayConfig.handoffProviders.includes(provider) && - provider === "codex" - ) { - const connectionId = getSessionConnection(relayOptions.sessionId); - if (connectionId) { - const quotaInfo = await fetchCodexQuota(connectionId).catch(() => null); - if (quotaInfo) { - const resetCandidates = [ - quotaInfo.windows?.session?.resetAt, - quotaInfo.windows?.weekly?.resetAt, - quotaInfo.resetAt, - ] - .filter( - (value): value is string => typeof value === "string" && value.length > 0 - ) - .sort((a, b) => a.localeCompare(b)); - const handoffSourceMessages = - Array.isArray(body?.messages) && body.messages.length > 0 - ? body.messages - : Array.isArray(body?.input) - ? body.input - : []; - - maybeGenerateHandoff({ - sessionId: relayOptions.sessionId, - comboName: combo.name, - connectionId, - percentUsed: quotaInfo.percentUsed, - messages: handoffSourceMessages, - model: modelStr, - expiresAt: resetCandidates[0] || null, - config: relayConfig, - handleSingleModel: handleSingleModelWithTimeout, - }); - } - } - } - - // Record last known good provider (LKGP) for this combo/model (#919) - if (provider) { - const connId = target.connectionId || undefined; - void (async () => { - try { - const { setLKGP } = await import("../../src/lib/localDb"); - await Promise.all([ - setLKGP(combo.name, target.executionKey, provider, connId), - setLKGP(combo.name, combo.id || combo.name, provider, connId), - ]); - } catch (err) { - log.warn( - "COMBO", - "Failed to record Last Known Good Provider. This is non-fatal.", - { - err, - } - ); - } - })(); - } - - return { ok: true, response: quality.clonedResponse ?? result }; - } - - // Extract error info from response - let errorText = result.statusText || ""; - let errorBody: ComboErrorBody = null; - let retryAfter: ComboRetryAfter | null = null; - try { - const cloned = result.clone(); - try { - const text = await cloned.text(); - if (text) { - errorText = text.substring(0, 500); - errorBody = JSON.parse(text); - const parsedError = errorBody?.error; - errorText = - (typeof parsedError === "object" && parsedError?.message) || - (typeof parsedError === "string" ? parsedError : null) || - errorBody?.message || - errorText; - retryAfter = errorBody?.retryAfter || null; - } - } catch { - /* Clone parse failed */ - } - } catch { - /* Clone failed */ - } - - // Track earliest retryAfter - if ( - retryAfter && - (!earliestRetryAfter || new Date(retryAfter) < new Date(earliestRetryAfter)) - ) { - earliestRetryAfter = retryAfter; - } - - // Normalize error text - if (typeof errorText !== "string") { - try { - errorText = JSON.stringify(errorText); - } catch { - errorText = String(errorText); - } - } - - const isStreamReadinessFailure = - (result.status === 502 || result.status === 504) && - isStreamReadinessFailureErrorBody(errorBody); - - // FIX 5: a local per-API-key token-limit 429 must not cool shared accounts. - const isTokenLimitBreach = - result.status === 429 && isTokenLimitBreachErrorBody(errorBody); - - // Fix #1681: Status 499 means client disconnected — stop combo loop immediately. - // There is no point trying fallback models when nobody is listening. - if (result.status === 499) { - log.info("COMBO", `Client disconnected (499) during ${modelStr} — stopping combo loop`); - recordComboRequest(combo.name, modelStr, { - success: false, - latencyMs: Date.now() - startTime, - fallbackCount, - strategy, - target: toRecordedTarget(target), - }); - recordedAttempts++; - // executeTarget must return the {ok,response} contract — a raw Response - // here makes the speculative loop's res.ok/res.response checks both miss, - // so the combo would wrongly fall through to the next model after a 499. - return { ok: false, response: result }; - } - - // Combo fallback is target-level orchestration: a non-ok target response is - // treated as local to that target and the combo continues to the next target. - // Error classification is retained only for retry/cooldown pacing; it must - // not decide whether fallback happens, including for generic 400 responses. - const rawError = errorBody?.error; - const structuredError = - rawError && typeof rawError === "object" - ? { - // Upstream JSON may carry a numeric `code`/`type` (e.g. {"code":40001}). - // Coerce to string if present instead of discarding, so downstream string - // ops (.toLowerCase, .startsWith) can run safely without type crashes. - code: - (rawError as Record).code !== undefined && - (rawError as Record).code !== null - ? String((rawError as Record).code) - : undefined, - type: - (rawError as Record).type !== undefined && - (rawError as Record).type !== null - ? String((rawError as Record).type) - : undefined, - } - : undefined; - const fallbackResult = checkFallbackError( - result.status, - errorText, - 0, - null, - provider, - result.headers, - profile, - structuredError - ); - const { cooldownMs } = fallbackResult; - - // #1731: If the entire provider quota is exhausted, mark it so subsequent - // same-provider targets are skipped immediately. API-key 429s still use - // the short resilience cooldown, but explicit quota text should stop the - // combo from trying another target for the same provider in this request. - const providerExhausted = - Boolean(provider && provider !== "unknown") && - (isProviderExhaustedReason(fallbackResult) || - classifyErrorText(errorText) === RateLimitReason.QUOTA_EXHAUSTED); - if (providerExhausted) { - exhaustedProviders.add(provider); - log.info( - "COMBO", - `Provider ${provider} quota exhausted — marking for skip on remaining targets (#1731)` - ); - } else if ( - result.status === 429 && - !isTokenLimitBreach && - provider && - provider !== "unknown" - ) { - transientRateLimitedProviders.add(provider); - } - // #1731: Connection-level errors (502/503/504) suggest the provider itself is having - // issues (e.g. upstream unreachable, proxy error). Skip remaining same-provider - // targets in this request to avoid hammering a known-bad connection. - if ( - !providerExhausted && - provider && - provider !== "unknown" && - [408, 500, 502, 503, 504, 524].includes(result.status) && - !isProviderCircuitOpenResult(result, errorText) - ) { - const connId = target.connectionId as string | undefined; - if (connId) { - exhaustedConnections.add(`${provider}:${connId}`); - log.info( - "COMBO", - `Provider ${provider} connection ${connId} error (${result.status}) — marking for skip on remaining targets (#1731v2)` - ); - } else { - exhaustedProviders.add(provider); - log.info( - "COMBO", - `Provider ${provider} connection error (${result.status}) — marking for skip on remaining targets (#1731)` - ); - } - } - - // #2101: Prevent infinite fallback loops with 400 Bad Request errors that indicate - // request-body-specific issues (context overflow, malformed request, model access denied). - // These errors are unlikely to be resolved by trying different target models since - // the same problematic request body would be sent to all targets. - if ( - result.status === 400 && - fallbackResult.shouldFallback && - (fallbackResult.reason === RateLimitReason.MODEL_CAPACITY || - errorText.toLowerCase().includes("context") || - errorText.toLowerCase().includes("prompt") || - errorText.toLowerCase().includes("token") || - errorText.toLowerCase().includes("malformed") || - errorText.toLowerCase().includes("invalid") || - errorText.toLowerCase().includes("bad request")) - ) { - log.warn( - "COMBO", - `400 Bad Request with body-specific error detected on ${modelStr} — skipping fallback to other targets to prevent infinite loop` - ); - // Record the failure and break to avoid trying other targets with the same bad request - recordComboRequest(combo.name, modelStr, { - success: false, - latencyMs: Date.now() - startTime, - fallbackCount, - strategy, - target: toRecordedTarget(target), - }); - recordedAttempts++; - lastError = errorText || String(result.status); - if (!lastStatus) lastStatus = result.status; - if (i > 0) fallbackCount++; - log.warn("COMBO", `Model ${modelStr} failed with body-specific error, stopping combo`); - break; // Break out of the target loop to avoid trying other models - } - - // Trigger shared provider circuit breaker for 5xx errors and connection failures. - // If the next target in the combo is on the same provider, don't mark the provider - // as failed — different models on the same provider may still succeed. - // G-02: when fallbackResult.skipProviderBreaker is set (embedded service supervisor - // outage signalled via X-Omni-Fallback-Hint: connection_cooldown) apply connection - // cooldown only — do NOT trip the whole-provider breaker. - const nextTarget = orderedTargets[i + 1]; - const sameProviderNext = - typeof nextTarget?.provider === "string" && nextTarget.provider === provider; - if ( - !isStreamReadinessFailure && - isProviderFailureCode(result.status) && - !sameProviderNext && - !fallbackResult.skipProviderBreaker - ) { - recordProviderFailure(provider, log, target.connectionId, profile); - } - - // Check if this is a transient error worth retrying on same model. - // A token-limit 429 is terminal for the client — never retry it. - const isTransient = - !isStreamReadinessFailure && - !isTokenLimitBreach && - [408, 429, 500, 502, 503, 504].includes(result.status); - if (retry < maxRetries && isTransient && !providerExhausted) { - continue; // Retry same model - } - - // Done retrying this model - recordComboRequest(combo.name, modelStr, { - success: false, - latencyMs: Date.now() - startTime, - fallbackCount, - strategy, - target: toRecordedTarget(target), - }); - recordedAttempts++; - lastError = errorText || String(result.status); - if (!lastStatus) lastStatus = result.status; - if (i > 0) fallbackCount++; - log.warn("COMBO", `Model ${modelStr} failed, trying next`, { status: result.status }); - - if (resilienceSettings.providerCooldown.enabled && provider && provider !== "unknown") { - recordProviderCooldown(provider, target.connectionId ?? undefined, resilienceSettings); - } - - const fallbackWaitMs = - fallbackDelayMs > 0 && cooldownMs > 0 && cooldownMs <= MAX_FALLBACK_WAIT_MS - ? Math.min(cooldownMs, fallbackDelayMs) - : 0; - if ([502, 503, 504].includes(result.status) && fallbackWaitMs > 0) { - log.debug?.("COMBO", `Waiting ${fallbackWaitMs}ms before fallback to next model`); - await new Promise((resolve) => { - const timer = setTimeout(resolve, fallbackWaitMs); - signal?.addEventListener( - "abort", - () => { - clearTimeout(timer); - resolve(undefined); - }, - { once: true } - ); - }); - if (signal?.aborted) { - log.info("COMBO", `Client disconnected during fallback wait — aborting`); - return { ok: false, response: errorResponse(499, "Client disconnected") }; - } - } - - return null; - } - return null; - }; - - for (let i = 0; i < orderedTargets.length; i++) { - if (anySuccess) break; - - const abortController = new AbortController(); - abortControllers.set(i, abortController); - const onClientAbort = () => abortController.abort(); - signal?.addEventListener("abort", onClientAbort); - - const task = (async () => { - try { - const res = await executeTarget(i); - if (res && !anySuccess) { - if (res.ok) { - anySuccess = true; - globalResolve!(res.response!); - for (const [idx, ac] of abortControllers.entries()) { - if (idx !== i) ac.abort(); - } - } else if (res.response) { - // Fatal error, abort combo - anySuccess = true; - globalResolve!(res.response); - } - } - } finally { - signal?.removeEventListener("abort", onClientAbort); - } - })().catch((err) => { - const logError = log.error ?? log.warn; - logError("COMBO", `Speculative task error for target ${i}`, err); - }); - - runningTasks.add(task); - task.finally(() => runningTasks.delete(task)); - - if (zeroLatencyOptimizationsEnabled && config.hedging && i + 1 < orderedTargets.length) { - const hedgeDelay = resolveDelayMs(config.hedgeDelayMs, 500); - let timeoutResolve: () => void; - const timeoutPromise = new Promise((r) => { - timeoutResolve = r; - setTimeout(r, hedgeDelay); - }); - await Promise.race([task, globalPromise, timeoutPromise]); - } else { - await Promise.race([task, globalPromise]); - } - } - - if (!anySuccess && runningTasks.size > 0) { - await Promise.race([globalPromise, Promise.all([...runningTasks])]); - } - - if (anySuccess) { - return await globalPromise; - } - - // All models failed in this set try - const latencyMs = Date.now() - startTime; - if (recordedAttempts === 0) { - recordComboRequest(combo.name, null, { - success: false, - latencyMs, - fallbackCount, - strategy, - }); - } - - // Retry the entire set if more attempts remain - if (setTry < maxSetRetries) continue; - - // All set retries exhausted — return the final error - if (!lastStatus) { - notifyWebhookEvent("request.failed", { - combo: combo.name, - reason: "ALL_ACCOUNTS_INACTIVE", - latencyMs, - fallbackCount, - }); - return new Response( - JSON.stringify({ - error: { - message: "Service temporarily unavailable: all upstream accounts are inactive", - type: "service_unavailable", - code: "ALL_ACCOUNTS_INACTIVE", - }, - }), - { status: 503, headers: { "Content-Type": "application/json" } } - ); - } - - const status = lastStatus; - const msg = lastError || "All combo models unavailable"; - - if (earliestRetryAfter) { - const retryHuman = formatRetryAfter(toRetryAfterDisplayValue(earliestRetryAfter)); - log.warn("COMBO", `All models failed | ${msg} (${retryHuman})`); - return unavailableResponse(status, msg, earliestRetryAfter, retryHuman); - } - - log.warn("COMBO", `All models failed | ${msg}`); - return new Response(JSON.stringify({ error: { message: msg } }), { - status, - headers: { "Content-Type": "application/json" }, - }); - } - - return errorResponse(503, "Combo routing completed without an upstream response"); - } finally { - // G2: Clean up candidate registry to prevent unbounded memory growth. - _unregisterExecutionCandidates(_registeredExecutionKeys); - } -} - -/** - * Handle round-robin combo: each request goes to the next model in circular order. - * Uses semaphore-based concurrency control with queue + rate-limit awareness. - * - * Flow: - * 1. Pick target model via atomic counter (counter % models.length) - * 2. Acquire semaphore slot (may queue if at max concurrency) - * 3. Send request to target model - * 4. On 429 → mark model rate-limited, try next model in rotation - * 5. On semaphore timeout → fallback to next available model - */ -async function handleRoundRobinCombo({ - body, - combo, - handleSingleModel, - isModelAvailable, - log, - settings, - allCombos, - signal, -}: HandleRoundRobinOptions): Promise { - const config = settings - ? resolveComboConfig(combo, settings) - : { ...getDefaultComboConfig(), ...(combo.config || {}) }; - const concurrency = config.concurrencyPerModel ?? 3; - const queueTimeout = config.queueTimeoutMs ?? 30000; - const maxRetries = config.maxRetries ?? 1; - const retryDelayMs = resolveDelayMs(config.retryDelayMs, 2000); - const fallbackDelayMs = resolveDelayMs(config.fallbackDelayMs, 0); - - const resilienceSettings: ResilienceSettings = settings - ? resolveResilienceSettings(settings) - : resolveResilienceSettings(null); - - const orderedTargets = resolveComboTargets(combo, allCombos); - const tagFilteredTargets = await applyRequestTagRouting(orderedTargets, body, log); - const evalRankedTargets = orderTargetsByEvalScores(tagFilteredTargets, config.evalRouting, log); - const filteredTargets = filterTargetsByRequestCompatibility( - evalRankedTargets, - body, - log, - "Context-aware round-robin fallback" - ); - const modelCount = filteredTargets.length; - if (modelCount === 0) { - return comboModelNotFoundResponse("Round-robin combo has no executable targets"); - } - - scheduleShadowRouting( - combo, - config, - body, - resolveShadowTargets(combo, config, allCombos), - handleSingleModel, - isModelAvailable, - "round-robin", - log - ); - - // Get and increment atomic counter - const counter = rrCounters.get(combo.name) || 0; - if (!rrCounters.has(combo.name) && rrCounters.size >= MAX_RR_COUNTERS) { - const oldest = rrCounters.keys().next().value; - if (oldest !== undefined) rrCounters.delete(oldest); - } - rrCounters.set(combo.name, counter + 1); - const startIndex = counter % modelCount; - - const clientRequestedStream = body?.stream === true; - const startTime = Date.now(); - let lastError: string | null = null; - let lastStatus: number | null = null; - let earliestRetryAfter: ComboRetryAfter | null = null; - let globalAttempts = 0; - let fallbackCount = 0; - let recordedAttempts = 0; - - // #1731: Per-request in-memory set of providers whose quota is fully exhausted. - // When a target returns a quota-exhausted 429, remaining targets from the same - // provider are skipped to avoid the cascade through N same-provider targets. - const exhaustedProviders = new Set(); - const exhaustedConnections = new Set(); - const transientRateLimitedProviders = new Set(); - - // Try each model starting from the round-robin target - for (let offset = 0; offset < modelCount; offset++) { - const modelIndex = (startIndex + offset) % modelCount; - const target = filteredTargets[modelIndex]; - const modelStr = target.modelStr; - const provider = target.provider; - const profile = await getRuntimeProviderProfile(provider); - const semaphoreKey = `combo:${combo.name}:${target.executionKey}`; - const allowRateLimitedConnection = - Boolean(provider && provider !== "unknown") && transientRateLimitedProviders.has(provider); - const targetForAttempt = allowRateLimitedConnection - ? { ...target, allowRateLimitedConnection: true } - : target; - - // Pre-check availability - if (isModelAvailable) { - const available = await isModelAvailable(modelStr, targetForAttempt); - if (!available) { - log.debug?.( - "COMBO-RR", - `Skipping ${modelStr} — no credentials available or model excluded` - ); - if (offset > 0) fallbackCount++; - continue; - } - } - - if ( - resilienceSettings.providerCooldown.enabled && - Boolean(provider && provider !== "unknown") && - isProviderInCooldown(provider, target.connectionId as string | undefined, resilienceSettings) - ) { - log.info("COMBO-RR", `Skipping ${modelStr} — provider ${provider} in global cooldown`); - if (offset > 0) fallbackCount++; - continue; - } - - // #1731: Skip targets from a provider that already signaled full quota exhaustion - // this request. - // #1731v2: Skip targets whose provider:connection pair had a connection-level error. - if (provider && target.connectionId) { - const connKey = `${provider}:${target.connectionId}`; - if (exhaustedConnections.has(connKey)) { - log.info( - "COMBO-RR", - `Skipping ${modelStr} — connection ${target.connectionId} for provider ${provider} had connection error (#1731v2)` - ); - if (offset > 0) fallbackCount++; - continue; - } - } - if (provider && exhaustedProviders.has(provider)) { - log.info( - "COMBO-RR", - `Skipping ${modelStr} — provider ${provider} marked exhausted this request (#1731)` - ); - if (offset > 0) fallbackCount++; - continue; - } - - // Acquire semaphore slot (may wait in queue) - let release: () => void; - try { - release = await semaphore.acquire(semaphoreKey, { - maxConcurrency: concurrency, - timeoutMs: queueTimeout, - }); - } catch (err) { - const errCode = isRecord(err) && typeof err.code === "string" ? err.code : null; - if (errCode === "SEMAPHORE_TIMEOUT" || errCode === "SEMAPHORE_QUEUE_FULL") { - log.warn( - "COMBO-RR", - `Semaphore ${errCode === "SEMAPHORE_QUEUE_FULL" ? "queue full" : "timeout"} for ${modelStr}, trying next model` - ); - if (offset > 0) fallbackCount++; - continue; - } - throw err; - } - - // Retry loop within this model - try { - for (let retry = 0; retry <= maxRetries; retry++) { - globalAttempts++; - if (globalAttempts > MAX_GLOBAL_ATTEMPTS) { - log.warn( - "COMBO-RR", - `Maximum combo attempts (${MAX_GLOBAL_ATTEMPTS}) exceeded. Terminating loop to prevent runaway requests.` - ); - return errorResponse(503, "Maximum combo retry limit reached"); - } - if (retry > 0) { - log.info( - "COMBO-RR", - `Retrying ${modelStr} in ${retryDelayMs}ms (attempt ${retry + 1}/${maxRetries + 1})` - ); - await new Promise((r) => setTimeout(r, retryDelayMs)); - } - - log.info( - "COMBO-RR", - `[RR #${counter}] → ${modelStr}${offset > 0 ? ` (fallback +${offset})` : ""}${retry > 0 ? ` (retry ${retry})` : ""}` - ); - - // Issue #3587: Reasoning models consume ALL max_tokens for reasoning_tokens. - // Add buffer to ensure reasoning + content both fit. Apply the buffer to a - // per-attempt COPY — never mutate the shared `body` — so it does not compound - // across round-robin iterations/retries (otherwise 4096 -> 6144 -> 9216 -> ... - // as each reasoning model re-reads an already-buffered value and overshoots the - // model's real limit, triggering 400s). - let attemptBody = body; - if (supportsReasoning(modelStr)) { - const currentMaxTokens = Number((body as Record).max_tokens) || 0; - if (currentMaxTokens > 0) { - const bufferedMaxTokens = Math.max( - currentMaxTokens + 1000, - Math.ceil(currentMaxTokens * 1.5) - ); - attemptBody = { - ...(body as Record), - max_tokens: bufferedMaxTokens, - } as typeof body; - log.info( - "COMBO-RR", - `Reasoning model ${modelStr}: buffered max_tokens ${currentMaxTokens} -> ${bufferedMaxTokens}` - ); - } - } - - const result = await handleSingleModel(attemptBody, modelStr, { - ...targetForAttempt, - failoverBeforeRetry: config.failoverBeforeRetry, - }); - - // Success — validate response quality before returning - if (result.ok) { - const quality = await validateResponseQuality(result, clientRequestedStream, log); - if (!quality.valid) { - log.warn( - "COMBO-RR", - `${modelStr} returned 200 but failed quality check: ${quality.reason}` - ); - recordComboRequest(combo.name, modelStr, { - success: false, - latencyMs: Date.now() - startTime, - fallbackCount, - strategy: "round-robin", - target: toRecordedTarget(target), - }); - recordedAttempts++; - // Fix #1707: Set terminal state so the fallback doesn't emit - // misleading ALL_ACCOUNTS_INACTIVE when the real issue is quality. - lastError = `Upstream response failed quality validation: ${quality.reason}`; - if (!lastStatus) lastStatus = 502; - if (offset > 0) fallbackCount++; - break; // move to next model - } - const latencyMs = Date.now() - startTime; - log.info( - "COMBO-RR", - `${modelStr} succeeded (${latencyMs}ms, ${fallbackCount} fallbacks)` - ); - recordComboRequest(combo.name, modelStr, { - success: true, - latencyMs, - fallbackCount, - strategy: "round-robin", - target: toRecordedTarget(target), - }); - recordedAttempts++; - - if (provider && provider !== "unknown") { - recordProviderSuccess(provider, target.connectionId ?? undefined); - } - - if (provider) { - const connId = target.connectionId || undefined; - void (async () => { - try { - const { setLKGP } = await import("../../src/lib/localDb"); - await Promise.all([ - setLKGP(combo.name, target.executionKey, provider, connId), - setLKGP(combo.name, combo.id || combo.name, provider, connId), - ]); - } catch (err) { - log.warn( - "COMBO-RR", - "Failed to record Last Known Good Provider. This is non-fatal.", - { - err, - } - ); - } - })(); - } - return result; - } - - // Extract error info - let errorText = result.statusText || ""; - let retryAfter: ComboRetryAfter | null = null; - let errorBody: ComboErrorBody = null; - try { - const cloned = result.clone(); - try { - const text = await cloned.text(); - if (text) { - errorText = text.substring(0, 500); - errorBody = JSON.parse(text); - const parsedError = errorBody?.error; - errorText = - (typeof parsedError === "object" && parsedError?.message) || - (typeof parsedError === "string" ? parsedError : null) || - errorBody?.message || - errorText; - retryAfter = errorBody?.retryAfter || null; - } - } catch { - /* Clone parse failed */ - } - } catch { - /* Clone failed */ - } - - if (result.status === 499) { - log.info( - "COMBO-RR", - `Client disconnected (499) during ${modelStr} — stopping combo loop` - ); - recordComboRequest(combo.name, modelStr, { - success: false, - latencyMs: Date.now() - startTime, - fallbackCount, - strategy: "round-robin", - target: toRecordedTarget(target), - }); - recordedAttempts++; - return result; - } - - if ( - retryAfter && - (!earliestRetryAfter || new Date(retryAfter) < new Date(earliestRetryAfter)) - ) { - earliestRetryAfter = retryAfter; - } - - if (typeof errorText !== "string") { - try { - errorText = JSON.stringify(errorText); - } catch { - errorText = String(errorText); - } - } - - const isStreamReadinessFailure = - (result.status === 502 || result.status === 504) && - isStreamReadinessFailureErrorBody(errorBody); - - // FIX 5: a local per-API-key token-limit 429 must not cool shared accounts. - const isTokenLimitBreach = result.status === 429 && isTokenLimitBreachErrorBody(errorBody); - - // Round-robin uses the same target-level fallback rule as other combo - // strategies: non-ok target responses fall through to the next target. - // Classification stays here only to support cooldown/semaphore pacing, - // not to decide whether fallback is allowed. - const rawError = errorBody?.error; - const structuredError = - rawError && typeof rawError === "object" - ? { - // Upstream JSON may carry a numeric `code`/`type` (e.g. {"code":40001}). - // Coerce to string if present instead of discarding, so downstream string - // ops (.toLowerCase, .startsWith) can run safely without type crashes. - code: - (rawError as Record).code !== undefined && - (rawError as Record).code !== null - ? String((rawError as Record).code) - : undefined, - type: - (rawError as Record).type !== undefined && - (rawError as Record).type !== null - ? String((rawError as Record).type) - : undefined, - } - : undefined; - const fallbackResult = checkFallbackError( - result.status, - errorText, - 0, - null, - provider, - result.headers, - profile, - structuredError - ); - const { cooldownMs } = fallbackResult; - - const isAllAccountsRateLimited = isAllAccountsRateLimitedResponse( - result.status, - result.headers?.get("content-type") ?? null, - errorText - ); - - // #1731: If the entire provider quota is exhausted, mark it so subsequent - // same-provider targets are skipped immediately. API-key 429s still use - // the short resilience cooldown, but explicit quota text should stop the - // combo from trying another target for the same provider in this request. - const providerExhausted = - Boolean(provider && provider !== "unknown") && - (isProviderExhaustedReason(fallbackResult) || - classifyErrorText(errorText) === RateLimitReason.QUOTA_EXHAUSTED || - isAllAccountsRateLimited); - if (providerExhausted) { - exhaustedProviders.add(provider); - log.debug?.( - "COMBO-RR", - `Provider ${provider} quota exhausted — marking for skip (#1731)` - ); - } else if ( - result.status === 429 && - !isTokenLimitBreach && - provider && - provider !== "unknown" - ) { - transientRateLimitedProviders.add(provider); - } - - // #1731v2: Connection-level errors (502/503/504) — skip remaining same-connection targets - if ( - !providerExhausted && - provider && - provider !== "unknown" && - [408, 500, 502, 503, 504, 524].includes(result.status) && - !isProviderCircuitOpenResult(result, errorText) - ) { - const connId = target.connectionId as string | undefined; - if (connId) { - exhaustedConnections.add(`${provider}:${connId}`); - log.info( - "COMBO-RR", - `Provider ${provider} connection ${connId} error (${result.status}) — marking for skip (#1731v2)` - ); - } else { - exhaustedProviders.add(provider); - log.info( - "COMBO-RR", - `Provider ${provider} connection error (${result.status}) — marking for skip (#1731)` - ); - } - } - - // Transient errors → mark in semaphore so round-robin stops stampeding this target. - if ( - !isStreamReadinessFailure && - !isTokenLimitBreach && - TRANSIENT_FOR_SEMAPHORE.includes(result.status) && - cooldownMs > 0 - ) { - semaphore.markRateLimited(semaphoreKey, cooldownMs); - log.warn("COMBO-RR", `${modelStr} error ${result.status}, cooldown ${cooldownMs}ms`); - } - - if (isAllAccountsRateLimited) { - log.info( - "COMBO-RR", - `All accounts rate-limited for ${modelStr}, falling back to next model` - ); - } - - // Transient error → retry same model. - // A token-limit 429 is terminal for the client — never retry it. - const isTransient = - !isStreamReadinessFailure && - !isTokenLimitBreach && - [408, 429, 500, 502, 503, 504].includes(result.status); - if (retry < maxRetries && isTransient && !providerExhausted) { - continue; - } - - // Done with this model - recordComboRequest(combo.name, modelStr, { - success: false, - latencyMs: Date.now() - startTime, - fallbackCount, - strategy: "round-robin", - target: toRecordedTarget(target), - }); - recordedAttempts++; - lastError = errorText || String(result.status); - if (!lastStatus) lastStatus = result.status; - if (offset > 0) fallbackCount++; - log.warn("COMBO-RR", `${modelStr} failed, trying next model`, { status: result.status }); - - if (resilienceSettings.providerCooldown.enabled && provider && provider !== "unknown") { - recordProviderCooldown(provider, target.connectionId ?? undefined, resilienceSettings); - } - - const fallbackWaitMs = - fallbackDelayMs > 0 && cooldownMs > 0 && cooldownMs <= MAX_FALLBACK_WAIT_MS - ? Math.min(cooldownMs, fallbackDelayMs) - : 0; - if ([502, 503, 504].includes(result.status) && fallbackWaitMs > 0) { - log.debug?.("COMBO-RR", `Waiting ${fallbackWaitMs}ms before fallback to next model`); - await new Promise((resolve) => { - const timer = setTimeout(resolve, fallbackWaitMs); - signal?.addEventListener( - "abort", - () => { - clearTimeout(timer); - resolve(undefined); - }, - { once: true } - ); - }); - if (signal?.aborted) { - log.info("COMBO-RR", `Client disconnected during fallback wait — aborting`); - return errorResponse(499, "Client disconnected"); - } - } - - break; - } - } finally { - // ALWAYS release semaphore slot - release(); - } - } - - // All models exhausted - const latencyMs = Date.now() - startTime; - if (recordedAttempts === 0) { - recordComboRequest(combo.name, null, { - success: false, - latencyMs, - fallbackCount, - strategy: "round-robin", - }); - } - - if (!lastStatus) { - return new Response( - JSON.stringify({ - error: { - message: "Service temporarily unavailable: all upstream accounts are inactive", - type: "service_unavailable", - code: "ALL_ACCOUNTS_INACTIVE", - }, - }), - { status: 503, headers: { "Content-Type": "application/json" } } - ); - } - - const status = lastStatus; - const msg = lastError || "All round-robin combo models unavailable"; - - if (earliestRetryAfter) { - const retryHuman = formatRetryAfter(toRetryAfterDisplayValue(earliestRetryAfter)); - log.warn("COMBO-RR", `All models failed | ${msg} (${retryHuman})`); - return unavailableResponse(status, msg, earliestRetryAfter, retryHuman); - } - - log.warn("COMBO-RR", `All models failed | ${msg}`); - return new Response(JSON.stringify({ error: { message: msg } }), { - status, - headers: { "Content-Type": "application/json" }, - }); -} +export * from "./combo/index.ts"; diff --git a/open-sse/services/combo/auto.ts b/open-sse/services/combo/auto.ts new file mode 100644 index 00000000000..a116c6e6d31 --- /dev/null +++ b/open-sse/services/combo/auto.ts @@ -0,0 +1,515 @@ +/** + * Shared combo (model combo) handling with fallback support + * Supports: priority, weighted, round-robin, random, least-used, cost-optimized, + * reset-aware, reset-window, strict-random, auto, fill-first, p2c, lkgp, + * context-optimized, and context-relay strategies + */ + +import { + checkFallbackError, + classifyErrorText, + formatRetryAfter, + getRuntimeProviderProfile, + recordProviderFailure, + isProviderFailureCode, + isProviderExhaustedReason, + type ProviderProfile, +} from "../accountFallback.ts"; + +import { FETCH_TIMEOUT_MS, RateLimitReason } from "../../config/constants.ts"; + +import { errorResponse, unavailableResponse } from "../../utils/error.ts"; + +import { clamp01 } from "../../utils/number.ts"; + +import { + recordComboIntent, + recordComboRequest, + recordComboShadowRequest, + getComboMetrics, +} from "../comboMetrics.ts"; + +import { + recordSessionModelUsage, + getLastSessionModel, + getHandoff, +} from "../../../src/lib/db/contextHandoffs.ts"; + +import { getQuotaFetcher } from "../quotaPreflight.ts"; + +import { getCircuitBreaker } from "../../../src/shared/utils/circuitBreaker"; + +import { fisherYatesShuffle, getNextFromDeck } from "../../../src/shared/utils/shuffleDeck"; + +import { parseModel } from "../model.ts"; + +import { emit } from "../../../src/lib/events/eventBus"; + +import { notifyWebhookEvent } from "../../../src/lib/webhookDispatcher"; + +import { getTaskFitness } from "../autoCombo/taskFitness.ts"; + +import { + calculateFactors, + calculateScore, + DEFAULT_WEIGHTS, + type ProviderCandidate, + type ScoringWeights, +} from "../autoCombo/scoring.ts"; + +import { getReasoningTokens } from "../../../src/lib/usage/tokenAccounting.ts"; + +import { getModelContextLimit } from "../../../src/lib/modelCapabilities"; + +import { getProviderConnections } from "../../../src/lib/db/providers"; + +import { getProviderModels } from "../../config/providerModels.ts"; + +import { + getComboModelString, + getComboStepTarget, + getComboStepWeight, + normalizeComboStep, +} from "../../../src/lib/combos/steps.ts"; + +import { + getConnectionRoutingTags, + matchesRoutingTags, + resolveRequestRoutingTags, + type RoutingTagMatchMode, +} from "../../../src/domain/tagRouter.ts"; + +import { normalizeRoutingStrategy } from "../../../src/shared/constants/routingStrategies.ts"; + +import { + resolveResilienceSettings, + type ResilienceSettings, +} from "../../../src/lib/resilience/settings"; + +import { ResolvedComboTarget, ResetWindowConfig, AutoProviderCandidate, HistoricalLatencyStatsEntry, ComboLike, ComboCollectionLike, ComboRuntimeStep } from "./types.ts"; +import { resolveResetWindowConfig, fetchResetAwareQuotaWithCache, calculateResetWindowAffinity } from "./quota.ts"; +import { MIN_HISTORY_SAMPLES, OUTPUT_TOKEN_RATIO, QUOTA_SOFT_DEPRIORITIZE_FACTOR } from "./constants.ts"; +import { getBootstrapLatencyMs, calculateTargetContextAffinity } from "./context.ts"; +import { resolveNestedComboTargets, getDirectComboTargets, getOrderedTopLevelRuntimeSteps, hasCompositeTierRuntimeOrder, expandRuntimeStep, dedupeTargetsByExecutionKey } from "./dag.ts"; +import { selectWeightedTarget, orderTargetsForWeightedFallback } from "./sorting.ts"; +import { isRecord } from "./utils.ts"; + + + +export async function buildAutoCandidates( + targets: ResolvedComboTarget[], + comboName: string, + sessionId: string | null | undefined = null, + resetWindowConfig: ResetWindowConfig = resolveResetWindowConfig(null) +): Promise { + const metrics = getComboMetrics(comboName); + const { getPricingForModel } = await import("../../../src/lib/localDb"); + const quotaPromises = new Map>(); + let historicalLatencyStats: Record = {}; + try { + const { getModelLatencyStats } = await import("../../../src/lib/usageDb"); + historicalLatencyStats = await getModelLatencyStats({ + windowHours: 24, + minSamples: 3, + maxRows: 10000, + }); + } catch { + // keep empty stats — auto-combo will use runtime + bootstrap signals + } + + const uniqueProviders = Array.from( + new Set( + targets.map((target) => target.provider || parseModel(target.modelStr).provider || "unknown") + ) + ); + const connectionPoolCounts = new Map(); + const connectionsByProvider = new Map>>(); + await Promise.all( + uniqueProviders.map(async (provider) => { + try { + const connections = await getProviderConnections({ provider, isActive: true }); + const active = Array.isArray(connections) ? connections : []; + connectionPoolCounts.set(provider, active.length); + connectionsByProvider.set(provider, active); + } catch { + connectionPoolCounts.set(provider, 0); + connectionsByProvider.set(provider, []); + } + }) + ); + + const expandedTargets: ResolvedComboTarget[] = []; + for (const target of targets) { + const provider = target.provider || parseModel(target.modelStr).provider || "unknown"; + const providerConnections = connectionsByProvider.get(provider) || []; + if (target.connectionId) { + expandedTargets.push(target); + continue; + } + const connectionIds = providerConnections + .map((c) => (c && typeof c === "object" && typeof c.id === "string" ? c.id : null)) + .filter((id): id is string => id !== null); + if (connectionIds.length === 0) { + expandedTargets.push(target); + continue; + } + for (const connectionId of connectionIds) { + expandedTargets.push({ + ...target, + connectionId, + executionKey: `${target.executionKey}@${connectionId}`, + }); + } + } + + const candidates = await Promise.all( + expandedTargets.map(async (target) => { + const modelStr = target.modelStr; + const parsed = parseModel(modelStr); + const provider = target.provider || parsed.provider || parsed.providerAlias || "unknown"; + const model = parsed.model || modelStr; + const historicalKey = `${provider}/${model}`; + const historicalModelMetric = historicalLatencyStats[historicalKey] || null; + const historicalTotal = Number(historicalModelMetric?.totalRequests); + const hasHistoricalSignal = + Number.isFinite(historicalTotal) && historicalTotal >= MIN_HISTORY_SAMPLES; + + let costPer1MTokens = 1; + try { + const pricing = await getPricingForModel(provider, model); + const inputPrice = Number(pricing?.input); + const outputPrice = Number(pricing?.output); + if (Number.isFinite(inputPrice) && inputPrice >= 0) { + if (Number.isFinite(outputPrice) && outputPrice >= 0) { + costPer1MTokens = + inputPrice * (1 - OUTPUT_TOKEN_RATIO) + outputPrice * OUTPUT_TOKEN_RATIO; + } else { + costPer1MTokens = inputPrice; + } + } + } catch { + // keep default cost + } + + const modelMetric = metrics?.byModel?.[modelStr] || null; + const avgLatency = Number(modelMetric?.avgLatencyMs); + const successRate = Number(modelMetric?.successRate); + const historicalP95Latency = Number(historicalModelMetric?.p95LatencyMs); + const historicalStdDev = Number(historicalModelMetric?.latencyStdDev); + const historicalSuccessRate = Number(historicalModelMetric?.successRate); // 0..1 + + const p95LatencyMs = hasHistoricalSignal + ? Number.isFinite(historicalP95Latency) && historicalP95Latency > 0 + ? historicalP95Latency + : getBootstrapLatencyMs(model) + : Number.isFinite(avgLatency) && avgLatency > 0 + ? avgLatency + : getBootstrapLatencyMs(model); + + const errorRate = hasHistoricalSignal + ? Number.isFinite(historicalSuccessRate) && + historicalSuccessRate >= 0 && + historicalSuccessRate <= 1 + ? 1 - historicalSuccessRate + : 0.05 + : Number.isFinite(successRate) && successRate >= 0 && successRate <= 100 + ? 1 - successRate / 100 + : 0.05; + const latencyStdDev = + hasHistoricalSignal && Number.isFinite(historicalStdDev) && historicalStdDev > 0 + ? Math.max(10, historicalStdDev) + : Math.max(10, p95LatencyMs * 0.1); + + const breakerStateRaw = getCircuitBreaker(provider)?.getStatus?.()?.state; + const circuitBreakerState: ProviderCandidate["circuitBreakerState"] = + breakerStateRaw === "OPEN" || breakerStateRaw === "HALF_OPEN" ? breakerStateRaw : "CLOSED"; + const contextAffinity = calculateTargetContextAffinity(target, sessionId); + let resetWindowAffinity = 0.5; + const fetcher = getQuotaFetcher(provider); + if (fetcher && target.connectionId) { + const quotaKey = `${provider}:${target.connectionId}`; + if (!quotaPromises.has(quotaKey)) { + quotaPromises.set( + quotaKey, + fetchResetAwareQuotaWithCache({ + provider, + connectionId: target.connectionId, + fetcher, + config: resetWindowConfig, + log: {}, + comboName, + }) + ); + } + const quota = await quotaPromises.get(quotaKey)!; + resetWindowAffinity = calculateResetWindowAffinity(quota, resetWindowConfig); + } + + return { + stepId: target.stepId, + executionKey: target.executionKey, + modelStr, + provider, + model, + quotaRemaining: 100, + quotaTotal: 100, + circuitBreakerState, + costPer1MTokens, + p95LatencyMs, + latencyStdDev, + errorRate, + accountTier: "standard" as const, + quotaResetIntervalSecs: 86400, + contextAffinity, + resetWindowAffinity, + connectionPoolSize: connectionPoolCounts.get(provider) ?? 1, + connectionId: target.connectionId ?? undefined, + }; + }) + ); + + return candidates; +} + + + +export async function applyRequestTagRouting( + targets: ResolvedComboTarget[], + body: Record | null | undefined, + log: { info?: (...args: unknown[]) => void; warn?: (...args: unknown[]) => void } +): Promise { + const { tags, matchMode } = resolveRequestRoutingTags(body); + if (tags.length === 0 || targets.length === 0) { + return targets; + } + + const providerIds = Array.from( + new Set(targets.map((target) => target.providerId || target.provider)) + ).filter( + (providerId): providerId is string => typeof providerId === "string" && providerId.length > 0 + ); + const providerConnections = new Map>>(); + + await Promise.all( + providerIds.map(async (providerId) => { + try { + const connections = await getProviderConnections({ provider: providerId, isActive: true }); + providerConnections.set( + providerId, + Array.isArray(connections) ? (connections as Array>) : [] + ); + } catch (error) { + log.warn?.( + "COMBO", + `Tag routing failed to load connections for provider=${providerId}: ${error instanceof Error ? error.message : String(error)}` + ); + providerConnections.set(providerId, []); + } + }) + ); + + const filteredTargets = targets.reduce((acc, target) => { + const providerKey = target.providerId || target.provider; + const candidateConnections = + providerConnections.get(providerKey)?.filter((connection) => { + const connectionId = + typeof connection.id === "string" && connection.id.trim().length > 0 + ? connection.id + : null; + if (!connectionId) return false; + if (target.connectionId) { + return connectionId === target.connectionId; + } + return true; + }) || []; + + const matchedConnectionIds = candidateConnections + .filter((connection) => + matchesRoutingTags( + getConnectionRoutingTags(connection.providerSpecificData), + tags, + matchMode + ) + ) + .map((connection) => connection.id) + .filter((connectionId): connectionId is string => typeof connectionId === "string"); + + if (matchedConnectionIds.length === 0) { + return acc; + } + + if (target.connectionId) { + acc.push(target); + return acc; + } + + acc.push({ + ...target, + allowedConnectionIds: Array.from(new Set(matchedConnectionIds)), + }); + return acc; + }, []); + + if (filteredTargets.length === 0) { + log.info?.( + "COMBO", + `Tag routing matched 0/${targets.length} targets for [${tags.join(", ")}] (${matchMode}); falling back to the full target set` + ); + return targets; + } + + log.info?.( + "COMBO", + `Tag routing matched ${filteredTargets.length}/${targets.length} targets for [${tags.join(", ")}] (${matchMode})` + ); + return filteredTargets; +} + + + +export function resolveComboTargets( + combo: ComboLike, + allCombos: ComboCollectionLike +): ResolvedComboTarget[] { + return allCombos ? resolveNestedComboTargets(combo, allCombos) : getDirectComboTargets(combo); +} + + + +export function resolveWeightedTargets( + combo: ComboLike, + allCombos: ComboCollectionLike +): { + orderedTargets: ResolvedComboTarget[]; + selectedStep: ComboRuntimeStep | null; +} { + const topLevelSteps = getOrderedTopLevelRuntimeSteps(combo, allCombos); + if (topLevelSteps.length === 0) { + return { orderedTargets: [], selectedStep: null }; + } + + const selectedStep = selectWeightedTarget(topLevelSteps); + if (!selectedStep) { + return { orderedTargets: [], selectedStep: null }; + } + + const orderedSteps = orderTargetsForWeightedFallback( + topLevelSteps, + selectedStep.executionKey, + hasCompositeTierRuntimeOrder(combo) + ); + const expandedTargets = orderedSteps.flatMap((step) => { + if (!step) return []; + if (!allCombos) { + return step.kind === "model" ? [step] : []; + } + return expandRuntimeStep(step, allCombos, new Set([combo.name])); + }); + + return { + orderedTargets: dedupeTargetsByExecutionKey(expandedTargets), + selectedStep, + }; +} + + + +export function scoreAutoTargets( + targets: ResolvedComboTarget[], + candidates: AutoProviderCandidate[], + taskType: string | null, + weights: ScoringWeights +) { + const candidateByExecutionKey = new Map( + candidates.map((candidate: ProviderCandidate & { executionKey: string }) => [ + candidate.executionKey, + candidate, + ]) + ); + return targets + .map((target) => { + const candidate = candidateByExecutionKey.get(target.executionKey); + if (!candidate) return null; + const factors = calculateFactors( + candidate as ProviderCandidate, + candidates, + taskType ?? "general", + getTaskFitness + ); + let score = calculateScore(factors, weights); + // B17: Quota Share soft-policy deprioritization + if ("quotaSoftPenalty" in candidate && candidate.quotaSoftPenalty === true) { + score *= QUOTA_SOFT_DEPRIORITIZE_FACTOR; + } + return { + target, + score, + }; + }) + .filter((entry): entry is { target: ResolvedComboTarget; score: number } => entry !== null) + .sort((a, b) => b.score - a.score); +} + + + +/** + * For an auto-combo WITHOUT an explicit `candidatePool`, broaden the eligible + * targets to every model of every active provider connection so the router has + * the full pool to score over. Already-present `modelStr`s are not duplicated. + * + * Best-effort: if loading active connections or provider models throws, the + * explicitly-resolved targets are returned unchanged (the combo still runs). + * Exported for unit testing. Mutates and returns `eligibleTargets`. + */ +export async function expandAutoComboCandidatePool( + eligibleTargets: ResolvedComboTarget[], + combo: { autoConfig?: unknown; config?: unknown } | null | undefined +): Promise { + const localAutoConfig = + (combo?.autoConfig as Record | undefined) || + (isRecord((combo?.config as Record)?.auto) + ? ((combo?.config as Record).auto as Record) + : null) || + (combo?.config as Record | undefined) || + {}; + + if (Array.isArray(localAutoConfig?.candidatePool)) return eligibleTargets; + + try { + const allConnections = await getProviderConnections({ isActive: true }); + const providerIds = [ + ...new Set( + (allConnections as Array<{ provider?: unknown }>) + .map((c) => c.provider) + .filter((p): p is string => typeof p === "string" && p.length > 0) + ), + ]; + for (const providerId of providerIds) { + const providerModels = getProviderModels(providerId); + for (const model of providerModels) { + const modelStr = `${providerId}/${model.id}`; + if (!eligibleTargets.some((t) => t.modelStr === modelStr)) { + eligibleTargets.push({ + kind: "model", + stepId: modelStr, + executionKey: modelStr, + provider: providerId, + providerId: providerId, + modelStr, + weight: 1, + connectionId: null, + label: null, + }); + } + } + } + } catch { + // Best-effort candidate expansion only: if loading active connections or + // provider models fails, fall back to the explicitly-resolved targets + // rather than aborting the combo. The push above is the only mutation, + // so a throw leaves eligibleTargets exactly as explicit resolution built it. + } + + return eligibleTargets; +} + diff --git a/open-sse/services/combo/chat.ts b/open-sse/services/combo/chat.ts new file mode 100644 index 00000000000..3121d88cd5e --- /dev/null +++ b/open-sse/services/combo/chat.ts @@ -0,0 +1,1578 @@ +/** + * Shared combo (model combo) handling with fallback support + * Supports: priority, weighted, round-robin, random, least-used, cost-optimized, + * reset-aware, reset-window, strict-random, auto, fill-first, p2c, lkgp, + * context-optimized, and context-relay strategies + */ + +import { + checkFallbackError, + classifyErrorText, + formatRetryAfter, + getRuntimeProviderProfile, + recordProviderFailure, + isProviderFailureCode, + isProviderExhaustedReason, + type ProviderProfile, +} from "../accountFallback.ts"; + +import { FETCH_TIMEOUT_MS, RateLimitReason } from "../../config/constants.ts"; + +import { errorResponse, unavailableResponse } from "../../utils/error.ts"; + +import { clamp01 } from "../../utils/number.ts"; + +import { + recordComboIntent, + recordComboRequest, + recordComboShadowRequest, + getComboMetrics, +} from "../comboMetrics.ts"; + +import { + resolveComboConfig, + getDefaultComboConfig, + resolveComboTargetTimeoutMs, + PRE_SCREEN_CONCURRENCY, +} from "../comboConfig.ts"; + +import { + maybeGenerateHandoff, + resolveContextRelayConfig, + maybeGenerateUniversalHandoff, + injectUniversalHandoffBody, + resolveUniversalHandoffConfig, + SKIP_UNIVERSAL_HANDOFF_FLAG, + type MessageLike, +} from "../contextHandoff.ts"; + +import { + recordSessionModelUsage, + getLastSessionModel, + getHandoff, +} from "../../../src/lib/db/contextHandoffs.ts"; + +import { fetchCodexQuota } from "../codexQuotaFetcher.ts"; + +import { getQuotaFetcher } from "../quotaPreflight.ts"; + +import * as semaphore from "../rateLimitSemaphore.ts"; + +import { getCircuitBreaker } from "../../../src/shared/utils/circuitBreaker"; + +import { fisherYatesShuffle, getNextFromDeck } from "../../../src/shared/utils/shuffleDeck"; + +import { parseModel } from "../model.ts"; + +import { applyComboAgentMiddleware } from "../comboAgentMiddleware.ts"; + +import { checkCredentialGate, logCredentialSkip } from "../credentialGate.ts"; + +import { emit } from "../../../src/lib/events/eventBus"; + +import { notifyWebhookEvent } from "../../../src/lib/webhookDispatcher"; + +import { + classifyWithConfig, + DEFAULT_INTENT_CONFIG, + type IntentClassifierConfig, +} from "../intentClassifier.ts"; + +import { selectProvider as selectAutoProvider } from "../autoCombo/engine.ts"; + +import { selectWithStrategy, type SlaRoutingPolicy } from "../autoCombo/routerStrategy.ts"; + +import { getTaskFitness } from "../autoCombo/taskFitness.ts"; + +import { parseAutoPrefix } from "../autoCombo/autoPrefix.ts"; + +import { handlePipelineCombo, buildPipelineResponse } from "../autoCombo/pipelineRouter.ts"; + +import { + calculateFactors, + calculateScore, + DEFAULT_WEIGHTS, + type ProviderCandidate, + type ScoringWeights, +} from "../autoCombo/scoring.ts"; + +import { + getResolvedModelCapabilities, + supportsReasoning, + supportsToolCalling, +} from "../modelCapabilities.ts"; + +import { estimateTokens } from "../contextManager.ts"; + +import { getReasoningTokens } from "../../../src/lib/usage/tokenAccounting.ts"; + +import { getSessionConnection } from "../sessionManager.ts"; + +import { orderTargetsByEvalScores } from "../evalRouting.ts"; + +import { generateRoutingHints } from "../manifestAdapter"; + +import type { CompressionMode } from "../compression/types.ts"; + +import { getModelContextLimit } from "../../../src/lib/modelCapabilities"; + +import { getProviderConnections } from "../../../src/lib/db/providers"; + +import { getProviderModels } from "../../config/providerModels.ts"; + +import { + getComboModelString, + getComboStepTarget, + getComboStepWeight, + normalizeComboStep, +} from "../../../src/lib/combos/steps.ts"; + +import { + getConnectionRoutingTags, + matchesRoutingTags, + resolveRequestRoutingTags, + type RoutingTagMatchMode, +} from "../../../src/domain/tagRouter.ts"; + +import { normalizeRoutingStrategy } from "../../../src/shared/constants/routingStrategies.ts"; + +import { + isProviderInCooldown, + recordProviderCooldown, + recordProviderSuccess, +} from "../providerCooldownTracker.ts"; + +import { + resolveResilienceSettings, + type ResilienceSettings, +} from "../../../src/lib/resilience/settings"; + +import { HandleComboChatOptions, SingleModelTarget, ResolvedComboTarget, PreScreenResult, ComboRetryAfter, ComboErrorBody } from "./types.ts"; +import { handleRoundRobinCombo } from "./roundRobin.ts"; +import { resolveDelayMs, comboModelNotFoundResponse, MAX_GLOBAL_ATTEMPTS, MAX_FALLBACK_WAIT_MS } from "./constants.ts"; +import { resolveWeightedTargets, resolveComboTargets, applyRequestTagRouting, expandAutoComboCandidatePool, buildAutoCandidates, scoreAutoTargets } from "./auto.ts"; +import { getModelContextLimitForModelString, orderTargetsByPowerOfTwoChoices, sortTargetsByUsage, sortTargetsByCost, sortTargetsByContextSize } from "./sorting.ts"; +import { extractPromptForIntent, getIntentConfig, mapIntentToTaskType, filterTargetsByRequestCompatibility } from "./context.ts"; +import { isRecord, validateResponseQuality, toRecordedTarget, isStreamReadinessFailureErrorBody, isTokenLimitBreachErrorBody, toRetryAfterDisplayValue } from "./utils.ts"; +import { resolveResetWindowConfig, resolveSlaRoutingPolicy, orderTargetsByResetAwareQuota, orderTargetsByResetWindow, preScreenTargets } from "./quota.ts"; +import { setCandidateQuotaSoftPenalty, _registerExecutionCandidates, _unregisterExecutionCandidates } from "./state.ts"; +import { dedupeTargetsByExecutionKey } from "./dag.ts"; +import { scheduleShadowRouting, resolveShadowTargets } from "./shadow.ts"; + + + +/** + * Handle combo chat with fallback. + * @param {Object} options + * @param {Object} options.body - Request body + * @param {Object} options.combo - Full combo object { name, models, strategy, config } + * @param {Function} options.handleSingleModel - Function: (body, modelStr) => Promise + * @param {Function} [options.isModelAvailable] - Optional pre-check: (modelStr) => Promise + * @param {Object} options.log - Logger object + * @returns {Promise} + */ +/** @param {object} options */ +export async function handleComboChat({ + body, + combo, + handleSingleModel, + isModelAvailable, + log, + settings, + allCombos, + relayOptions, + signal, + apiKeyAllowedConnections = null, +}: HandleComboChatOptions): Promise { + const strategy = normalizeRoutingStrategy(combo.strategy || "priority"); + const relayConfig = + strategy === "context-relay" ? resolveContextRelayConfig(relayOptions?.config || null) : null; + + const resilienceSettings: ResilienceSettings = settings + ? resolveResilienceSettings(settings) + : resolveResilienceSettings(null); + + const universalHandoffConfig = resolveUniversalHandoffConfig( + (combo.universal_handoff || combo.universalHandoff) as + | Record + | null + | undefined, + relayOptions?.universalHandoffConfig as Record | null | undefined + ); + // ── Server-side context cache pinning (replaces tag roundtrip) ─ + // Uses session_model_history — no client-side tag injection, no visible output pollution. + let pinnedModel: string | null = null; + if ( + combo.context_cache_protection && + relayOptions?.sessionId && + !(body as Record)?.[SKIP_UNIVERSAL_HANDOFF_FLAG] + ) { + const pinned = getLastSessionModel(relayOptions.sessionId, combo.name); + if (pinned) { + body = { ...body, model: pinned }; + pinnedModel = pinned; + log.info("COMBO", `[#401] Context cache: pinned model=${pinned} (server-side)`); + } + } + + // ── Combo Agent Middleware (#399 + #401) ──────────────────────────────── + // Apply system_message override, tool_filter_regex. + // Context cache pinning is handled above via session_model_history. + const { body: agentBody } = applyComboAgentMiddleware( + body, + combo, + "" // provider/model not yet known — resolved per-model in loop + ); + body = agentBody; + const clientRequestedStream = body?.stream === true; + // Context cache pinning is handled above via server-side session_model_history. + // No tag injection on response — use handleSingleModel directly. + // ───────────────────────────────────────────────────────────────────────── + + // Use config cascade before dispatch so all strategies, pinned context routes, + // and round-robin targets share the same timeout policy. + const config = settings + ? resolveComboConfig(combo, settings) + : { ...getDefaultComboConfig(), ...(combo.config || {}) }; + const comboTargetTimeoutMs = resolveComboTargetTimeoutMs(config, FETCH_TIMEOUT_MS); + + // ── Per-model timeout wrapper ──────────────────────────────────────────── + // Combo target timeouts inherit FETCH_TIMEOUT_MS by default. Operators can + // configure targetTimeoutMs to shorten fallback latency, but never to extend + // beyond the current upstream request timeout. + // + // The timeoutController is forwarded to the inner caller via target.modelAbortSignal. + // When the timeout fires we (a) resolve the race with a synthetic 524 and + // (b) abort the inner request so its upstream fetch is cancelled and downstream + // cooldown/breaker/usage mutations stop — preventing "ghost" state mutations + // that diverge from the routing decision the operator sees. + const handleSingleModelWithTimeout = async ( + b: Record, + modelStr: string, + target?: SingleModelTarget + ): Promise => { + if (comboTargetTimeoutMs <= 0) { + return handleSingleModel(b, modelStr, target).catch((err) => + errorResponse(502, err?.message ?? "Upstream model error") + ); + } + + const timeoutController = new AbortController(); + let timeoutId: ReturnType | undefined; + let timedOut = false; + const timeoutPromise = new Promise((resolve) => { + timeoutId = setTimeout(() => { + timedOut = true; + log.warn( + "COMBO", + `Model ${modelStr} exceeded ${comboTargetTimeoutMs}ms timeout — falling back` + ); + // Abort the inner request so its upstream fetch is cancelled and + // downstream cooldown/breaker/usage mutations don't continue mutating + // state behind the routing decision's back. + timeoutController.abort(new Error("combo-per-model-timeout")); + resolve( + new Response(JSON.stringify({ error: { message: `Model ${modelStr} timed out` } }), { + status: 524, + headers: { "Content-Type": "application/json" }, + }) + ); + }, comboTargetTimeoutMs); + }); + const targetWithSignal = { + ...(target ?? {}), + modelAbortSignal: timeoutController.signal, + }; + if (target?.modelAbortSignal) { + if (target.modelAbortSignal.aborted) { + timeoutController.abort(new Error("hedge-cancelled")); + } else { + target.modelAbortSignal.addEventListener("abort", () => { + timeoutController.abort(new Error("hedge-cancelled")); + }); + } + } + try { + return await Promise.race([ + handleSingleModel(b, modelStr, targetWithSignal).catch((err) => { + if (timedOut) { + // Inner call rejected because we aborted it. The synthetic 524 from + // timeoutPromise already wins the race; return an empty response so + // the loser branch resolves cleanly without leaking err.message. + return new Response(null, { status: 599 }); + } + return errorResponse(502, err?.message ?? "Upstream model error"); + }), + timeoutPromise, + ]); + } finally { + clearTimeout(timeoutId); + } + }; + + // Route to pinned model if context caching specifies one (Fix #679) + if (pinnedModel) { + log.info( + "COMBO", + `Bypassing strategy — routing directly to pinned context model: ${pinnedModel}` + ); + return handleSingleModelWithTimeout(body, pinnedModel); + } + + // Route to round-robin handler if strategy matches + if (strategy === "round-robin") { + return handleRoundRobinCombo({ + body, + combo, + handleSingleModel: handleSingleModelWithTimeout, + isModelAvailable, + log, + settings, + allCombos, + signal, + }); + } + + const maxRetries = config.maxRetries ?? 1; + const retryDelayMs = resolveDelayMs(config.retryDelayMs, 2000); + const fallbackDelayMs = resolveDelayMs(config.fallbackDelayMs, 0); + const maxSetRetries = config.maxSetRetries ?? 0; + const setRetryDelayMs = resolveDelayMs(config.setRetryDelayMs, 2000); + + let orderedTargets = + strategy === "weighted" + ? resolveWeightedTargets(combo, allCombos)?.orderedTargets || [] + : resolveComboTargets(combo, allCombos); + + orderedTargets = await applyRequestTagRouting(orderedTargets, body, log); + + if (strategy === "weighted") { + log.info( + "COMBO", + `Weighted selection${allCombos ? " with nested resolution" : ""}: ${orderedTargets.length} total targets` + ); + } else if (allCombos) { + log.info("COMBO", `${strategy} with nested resolution: ${orderedTargets.length} total targets`); + } + + // Pipeline dispatch: route smart/pipeline-enabled combos through the multi-stage pipeline + if (strategy === "auto") { + const autoParsed = parseAutoPrefix(combo.name); + const autoVariant = autoParsed.valid ? autoParsed.variant : undefined; + if (autoVariant === "smart" || config.pipeline_enabled) { + try { + const pipelineRaw = await handlePipelineCombo({ + body, + combo, + handleChatCore: handleSingleModelWithTimeout, + log: { + info: log.info, + warn: log.warn, + error: log.error ?? log.warn, + }, + settings: settings ?? {}, + signal: signal ?? undefined, + }); + // handlePipelineCombo resolves to a PipelineResult (buffered text) or, + // in the streaming-final-stage case, a Response. Callers downstream + // (chat.ts → withSessionHeader) require a Response, so adapt the + // PipelineResult here instead of leaking the raw object. + return pipelineRaw instanceof Response + ? pipelineRaw + : buildPipelineResponse(pipelineRaw, body); + } catch (pipelineErr) { + const pipelineMsg = pipelineErr instanceof Error ? pipelineErr.message : ""; + if (pipelineMsg === "PIPELINE_DISABLED") { + log.info("COMBO", "Pipeline disabled, falling through to standard auto routing"); + } else if (pipelineMsg === "PIPELINE_TOKEN_THRESHOLD") { + log.info( + "COMBO", + "Pipeline skipped (prompt below token threshold), falling through to standard auto routing" + ); + } else { + log.warn("COMBO", "Pipeline dispatch failed, falling through to standard auto routing", { + err: pipelineErr, + }); + } + } + } + } + + if (strategy === "auto") { + const requestHasTools = Array.isArray(body?.tools) && body.tools.length > 0; + let eligibleTargets = [...orderedTargets]; + + if (requestHasTools) { + const filtered = eligibleTargets.filter((target) => supportsToolCalling(target.modelStr)); + if (filtered.length > 0) { + eligibleTargets = filtered; + } else { + log.warn( + "COMBO", + "Auto strategy: all candidates filtered by tool-calling policy, falling back to full pool" + ); + } + } + + // Context-window pre-filter (#1808) + // Estimate input tokens once; exclude candidates whose known context limit is too small. + // Uses the same 4-chars-per-token heuristic as contextManager.ts::compressContext(). + // Null/unknown limits are treated as "include" to avoid incorrectly dropping valid targets. + const requestMessages = body.messages; + const estimatedInputTokens = estimateTokens( + typeof requestMessages === "string" || + (requestMessages !== null && typeof requestMessages === "object") + ? requestMessages + : [] + ); + if (estimatedInputTokens > 0) { + const filteredByContext = eligibleTargets.filter((target) => { + const limit = getModelContextLimitForModelString(target.modelStr); + if (limit === null || limit === undefined) return true; // unknown — include to be safe + return limit >= estimatedInputTokens; + }); + if (filteredByContext.length > 0) { + log.debug?.( + "COMBO", + `Auto strategy: context-window filter kept ${filteredByContext.length}/${eligibleTargets.length} candidates (est. ${estimatedInputTokens} tokens)` + ); + eligibleTargets = filteredByContext; + } else { + log.warn( + "COMBO", + `Auto strategy: all candidates filtered by context-window policy (est. ${estimatedInputTokens} tokens), falling back to full pool` + ); + // eligibleTargets intentionally unchanged — same fallback contract as tool-calling filter + } + + eligibleTargets = await expandAutoComboCandidatePool(eligibleTargets, combo); + } + + const prompt = extractPromptForIntent(body); + const systemPrompt = + typeof combo?.system_message === "string" ? combo.system_message : undefined; + const intentConfig = getIntentConfig(settings, combo); + const intent = classifyWithConfig(prompt, intentConfig, systemPrompt); + recordComboIntent(combo.name, intent); + const taskType = mapIntentToTaskType(intent); + + const rawAutoConfigSource = + combo?.autoConfig || + (isRecord(combo?.config?.auto) ? combo.config.auto : null) || + combo?.config || + {}; + const autoConfigSource: Record = isRecord(rawAutoConfigSource) + ? rawAutoConfigSource + : {}; + const routingStrategy = + typeof autoConfigSource.routerStrategy === "string" + ? autoConfigSource.routerStrategy + : typeof autoConfigSource.routingStrategy === "string" + ? autoConfigSource.routingStrategy + : typeof autoConfigSource.strategyName === "string" + ? autoConfigSource.strategyName + : "rules"; + + const candidatePool = Array.isArray(autoConfigSource.candidatePool) + ? autoConfigSource.candidatePool + : [...new Set(eligibleTargets.map((target) => target.provider))]; + + const weights = + autoConfigSource.weights && typeof autoConfigSource.weights === "object" + ? (autoConfigSource.weights as ScoringWeights) + : DEFAULT_WEIGHTS; + const explorationRate = Number.isFinite(Number(autoConfigSource.explorationRate)) + ? Number(autoConfigSource.explorationRate) + : 0.05; + const budgetCap = Number.isFinite(Number(autoConfigSource.budgetCap)) + ? Number(autoConfigSource.budgetCap) + : undefined; + const modePack = + typeof autoConfigSource.modePack === "string" ? autoConfigSource.modePack : undefined; + const resetWindowConfig = resolveResetWindowConfig(autoConfigSource); + const slaPolicy = resolveSlaRoutingPolicy(autoConfigSource); + + let lastKnownGoodProvider: string | undefined; + try { + const { getLKGP } = await import("../../../src/lib/localDb"); + const lkgp = await getLKGP(combo.name, combo.id || combo.name); + if (lkgp) lastKnownGoodProvider = lkgp.provider; + } catch (err) { + log.warn("COMBO", "Failed to retrieve Last Known Good Provider. This is non-fatal.", { err }); + } + + const candidates = await buildAutoCandidates( + eligibleTargets, + combo.name, + relayOptions?.sessionId, + resetWindowConfig + ); + // G2: Register candidates so chatCore can mark quotaSoftPenalty via setCandidateQuotaSoftPenalty. + _registerExecutionCandidates(candidates); + if (candidates.length > 0) { + let selectedProvider: string | null = null; + let selectedModel: string | null = null; + let selectionReason = ""; + + if (routingStrategy !== "rules") { + try { + const decision = selectWithStrategy( + candidates, + { + taskType, + requestHasTools, + lastKnownGoodProvider, + estimatedInputTokens, + sla: slaPolicy, + }, + routingStrategy + ); + selectedProvider = decision.provider; + selectedModel = decision.model; + selectionReason = decision.reason; + } catch (err) { + log.warn( + "COMBO", + `Auto strategy '${routingStrategy}' failed (${err?.message || "unknown"}), falling back to rules` + ); + } + } + + if (!selectedProvider || !selectedModel) { + const selection = selectAutoProvider( + { + id: combo.id || combo.name, + name: combo.name, + type: "auto", + candidatePool, + weights, + modePack, + budgetCap, + explorationRate, + }, + candidates, + taskType + ); + selectedProvider = selection.provider; + selectedModel = selection.model; + selectionReason = `score=${selection.score.toFixed(3)}${selection.isExploration ? " (exploration)" : ""}`; + } + + const scoredTargets = scoreAutoTargets(eligibleTargets, candidates, taskType, weights); + const rankedTargets = scoredTargets.map((entry) => entry.target); + const selectedTarget = + scoredTargets.find((entry) => { + const parsed = parseModel(entry.target.modelStr); + const modelId = parsed.model || entry.target.modelStr; + return entry.target.provider === selectedProvider && modelId === selectedModel; + })?.target || + rankedTargets[0] || + eligibleTargets[0]; + + orderedTargets = dedupeTargetsByExecutionKey( + [selectedTarget, ...rankedTargets, ...eligibleTargets].filter( + (entry): entry is ResolvedComboTarget => entry !== undefined && entry !== null + ) + ); + + log.info( + "COMBO", + `Auto selection: ${selectedTarget?.modelStr || `${selectedProvider}/${selectedModel}`} | intent=${intent} task=${taskType} | strategy=${routingStrategy} | ${selectionReason}` + ); + } else { + log.warn("COMBO", "Auto strategy has no candidates, keeping default ordering"); + } + } else if (strategy === "lkgp") { + try { + const { getLKGP } = await import("../../../src/lib/localDb"); + const lkgpProvider = await getLKGP(combo.name, combo.id || combo.name); + + if (lkgpProvider) { + const lkgpRecord = lkgpProvider; + const providerName = lkgpRecord.provider; + const connId = lkgpRecord.connectionId; + + let lkgpIndex = -1; + if (connId) { + lkgpIndex = orderedTargets.findIndex( + (target) => target.provider === providerName && target.connectionId === connId + ); + } + if (lkgpIndex < 0) { + lkgpIndex = orderedTargets.findIndex( + (target) => + target.provider === providerName || + // Issue #2359: Defensive guard. The `target.modelStr` type + // annotation is `string`, but malformed combo entries (e.g., + // local-provider rows whose `modelStr` failed to resolve when + // the executor catalogue was being rebuilt) have leaked + // through and surfaced as `e.startsWith is not a function` + // 500s on combo test/dispatch. The fast path stays + // unchanged for the common case; this only avoids the + // crash when the field is unexpectedly non-string. + (typeof target.modelStr === "string" && + target.modelStr.startsWith(`${providerName}/`)) + ); + } + + if (lkgpIndex > 0) { + const [lkgpTarget] = orderedTargets.splice(lkgpIndex, 1); + orderedTargets.unshift(lkgpTarget); + log.info( + "COMBO", + `[LKGP] Prioritizing last known good provider ${providerName}${connId ? ` (account ${connId})` : ""} for combo "${combo.name}"` + ); + } else if (lkgpIndex === 0) { + log.debug?.( + "COMBO", + `[LKGP] Last known good provider ${providerName}${connId ? ` (account ${connId})` : ""} already first for combo "${combo.name}"` + ); + } + } + } catch (err) { + log.warn("COMBO", "Failed to retrieve Last Known Good Provider. This is non-fatal.", { err }); + } + } else if (strategy === "strict-random") { + const selectedExecutionKey = await getNextFromDeck( + `combo:${combo.name}`, + orderedTargets.map((target) => target.executionKey) + ); + const selectedTarget = + orderedTargets.find((target) => target.executionKey === selectedExecutionKey) || null; + const rest = orderedTargets.filter((target) => target.executionKey !== selectedExecutionKey); + orderedTargets = [selectedTarget, ...rest].filter( + (target): target is ResolvedComboTarget => target !== null + ); + log.info( + "COMBO", + `Strict-random deck: ${selectedExecutionKey} selected (${orderedTargets.length} targets)` + ); + } else if (strategy === "random") { + orderedTargets = fisherYatesShuffle([...orderedTargets]); + log.info("COMBO", `Random shuffle: ${orderedTargets.length} targets`); + } else if (strategy === "fill-first") { + log.info( + "COMBO", + `Fill-first ordering: preserving priority order (${orderedTargets.length} targets)` + ); + } else if (strategy === "p2c") { + orderedTargets = orderTargetsByPowerOfTwoChoices(orderedTargets, combo.name); + log.info("COMBO", `Power-of-two-choices ordering: selected ${orderedTargets[0]?.modelStr}`); + } else if (strategy === "least-used") { + orderedTargets = sortTargetsByUsage(orderedTargets, combo.name); + log.info("COMBO", `Least-used ordering: ${orderedTargets[0]?.modelStr} has fewest requests`); + } else if (strategy === "cost-optimized") { + orderedTargets = await sortTargetsByCost(orderedTargets); + if (config.manifestRouting === true) { + try { + const manifestHint = generateRoutingHints( + orderedTargets.filter((t) => t.kind === "model"), + { + messages: Array.isArray(body?.messages) + ? (body.messages as Array<{ role?: string; content?: string | unknown }>) + : [], + tools: Array.isArray(body?.tools) + ? (body.tools as Array<{ + function?: { name: string; description?: string; parameters?: unknown }; + }>) + : undefined, + model: typeof body?.model === "string" ? body.model : undefined, + } + ); + if (manifestHint.strategyModifier === "require-premium") { + const eligible = orderedTargets.filter( + (t) => + t.kind !== "model" || + manifestHint.eligibleTargets.some( + (e) => e.provider === t.provider && e.modelStr === t.modelStr + ) + ); + if (eligible.length > 0) orderedTargets = eligible; + } + log.debug?.( + { + strategyModifier: manifestHint.strategyModifier, + specificityLevel: manifestHint.specificityLevel, + score: manifestHint.specificity.score, + }, + "manifest routing applied" + ); + } catch (err) { + log.warn({ err }, "manifest routing failed, falling back to standard strategy"); + } + } + log.info("COMBO", `Cost-optimized ordering: cheapest first (${orderedTargets[0]?.modelStr})`); + } else if (strategy === "reset-aware") { + orderedTargets = await orderTargetsByResetAwareQuota( + orderedTargets, + combo.name, + config, + log, + apiKeyAllowedConnections + ); + log.info( + "COMBO", + `Reset-aware ordering: ${orderedTargets[0]?.modelStr}${orderedTargets[0]?.connectionId ? ` (${orderedTargets[0].connectionId})` : ""} first` + ); + } else if (strategy === "reset-window") { + orderedTargets = await orderTargetsByResetWindow( + orderedTargets, + combo.name, + config, + log, + apiKeyAllowedConnections + ); + log.info( + "COMBO", + `Reset-window ordering: ${orderedTargets[0]?.modelStr}${orderedTargets[0]?.connectionId ? ` (${orderedTargets[0].connectionId})` : ""} first` + ); + } else if (strategy === "context-optimized") { + orderedTargets = sortTargetsByContextSize(orderedTargets); + log.info("COMBO", `Context-optimized ordering: largest first (${orderedTargets[0]?.modelStr})`); + } + + orderedTargets = orderTargetsByEvalScores(orderedTargets, config.evalRouting, log); + orderedTargets = filterTargetsByRequestCompatibility(orderedTargets, body, log); + + // Parallel pre-screen: check provider profiles and model availability for all targets + // Only runs for priority strategy where sequential checking causes latency + const preScreenMap = + strategy === "priority" + ? await preScreenTargets(orderedTargets, isModelAvailable).catch( + () => new Map() + ) + : new Map(); + + if (orderedTargets.length === 0) { + return comboModelNotFoundResponse("Combo has no executable targets"); + } + + scheduleShadowRouting( + combo, + config, + body, + resolveShadowTargets(combo, config, allCombos), + handleSingleModel, + isModelAvailable, + strategy, + log + ); + + // G2: Collect execution keys registered by _registerExecutionCandidates above (auto strategy). + // We snapshot them now so cleanup can happen after the attempt loop finishes. + const _registeredExecutionKeys = orderedTargets.map((t) => t.executionKey).filter(Boolean); + + let globalAttempts = 0; + + try { + for (let setTry = 0; setTry <= maxSetRetries; setTry++) { + // #1731: Per-set-iteration set of providers whose quota is fully exhausted. + // Reset each retry so providers excluded in a previous attempt get another chance. + const exhaustedProviders = new Set(); + const transientRateLimitedProviders = new Set(); + if (setTry > 0) { + log.info("COMBO", `All targets failed — retrying set (${setTry}/${maxSetRetries})`); + await new Promise((resolve) => { + const timer = setTimeout(resolve, setRetryDelayMs); + signal?.addEventListener( + "abort", + () => { + clearTimeout(timer); + resolve(undefined); + }, + { once: true } + ); + }); + if (signal?.aborted) { + log.info("COMBO", "Client disconnected during set retry delay — aborting"); + return errorResponse(499, "Client disconnected"); + } + } + + let lastError: string | null = null; + let earliestRetryAfter: ComboRetryAfter | null = null; + let lastStatus: number | null = null; + const startTime = Date.now(); + let fallbackCount = 0; + let recordedAttempts = 0; + + let globalResolve: ((res: Response) => void) | null = null; + const globalPromise = new Promise((res) => { + globalResolve = res; + }); + const runningTasks = new Set>(); + let anySuccess = false; + const abortControllers = new Map(); + const zeroLatencyOptimizationsEnabled = config.zeroLatencyOptimizationsEnabled === true; + + const executeTarget = async ( + i: number + ): Promise<{ ok: boolean; response?: Response } | null> => { + const target = orderedTargets[i]; + const modelStr = target.modelStr; + const provider = target.provider; + + const cb = getCircuitBreaker(provider); + if (cb.getStatus().state === "OPEN") { + log.info("COMBO", `Skipping ${modelStr} — circuit breaker OPEN for ${provider}`); + if (i > 0) fallbackCount++; + return null; + } + + if ( + resilienceSettings.providerCooldown.enabled && + Boolean(provider && provider !== "unknown") && + isProviderInCooldown(provider, target.connectionId ?? undefined, resilienceSettings) + ) { + log.info("COMBO", `Skipping ${modelStr} — provider ${provider} in global cooldown`); + if (i > 0) fallbackCount++; + return null; + } + + // Use pre-screened profile if available, otherwise fetch on demand + const preScreenEntry = preScreenMap.get(target.executionKey); + const profile = preScreenEntry?.profile ?? (await getRuntimeProviderProfile(provider)); + + const allowRateLimitedConnection = + Boolean(provider && provider !== "unknown") && + transientRateLimitedProviders.has(provider); + const targetForAttempt = allowRateLimitedConnection + ? { + ...target, + allowRateLimitedConnection: true, + modelAbortSignal: abortControllers.get(i)!.signal, + } + : { ...target, modelAbortSignal: abortControllers.get(i)!.signal }; + + // #1731: Skip targets from a provider that already signaled full quota exhaustion this request. + if (provider && exhaustedProviders.has(provider)) { + log.info( + "COMBO", + `Skipping ${modelStr} — provider ${provider} marked exhausted this request (#1731)` + ); + if (i > 0) fallbackCount++; + return null; + } + + // Pre-screen may have already determined this target unavailable (e.g. + // circuit-breaker OPEN at resolve time). Skip immediately in that case. + // For targets pre-screened as "available" we still call isModelAvailable + // below because connection cooldowns (rateLimitedUntil) can change + // mid-request after a same-provider failure — the pre-screen snapshot is + // stale by the time we reach the 2nd/3rd same-provider target. + const preCheckedAvailable = preScreenEntry?.available ?? null; + if (preCheckedAvailable === false) { + log.info("COMBO", `Skipping ${modelStr} — pre-screen marked unavailable`); + if (i > 0) fallbackCount++; + return null; + } + if (isModelAvailable) { + const available = await isModelAvailable(modelStr, targetForAttempt); + if (!available) { + log.debug?.( + "COMBO", + `Skipping ${modelStr} — no credentials available or model excluded` + ); + if (i > 0) fallbackCount++; + return null; + } + } + + // Credential gate: skip targets with known-bad credentials (fail-fast) + const connectionId = target.connectionId as string | undefined; + if (connectionId) { + const gateResult = checkCredentialGate(connectionId, provider, modelStr); + if (gateResult.allowed === false) { + logCredentialSkip(log, modelStr, gateResult.reason || "Credential gate blocked"); + if (i > 0) fallbackCount++; + return null; + } + } + + // Retry loop for transient errors + for (let retry = 0; retry <= maxRetries; retry++) { + // Fix #1681: Bail out immediately if the client has disconnected + if (signal?.aborted) { + log.info("COMBO", `Client disconnected — aborting combo loop before model ${modelStr}`); + return { ok: false, response: errorResponse(499, "Client disconnected") }; + } + globalAttempts++; + if (globalAttempts > MAX_GLOBAL_ATTEMPTS) { + log.warn( + "COMBO", + `Maximum combo attempts (${MAX_GLOBAL_ATTEMPTS}) exceeded across all targets and fallbacks. Terminating loop to prevent runaway background requests.` + ); + return { ok: false, response: errorResponse(503, "Maximum combo retry limit reached") }; + } + + // Predictive TTFT Circuit Breaker (skip slow models) + if ( + zeroLatencyOptimizationsEnabled && + config.predictiveTtftMs && + config.predictiveTtftMs > 0 && + retry === 0 + ) { + const cMetrics = getComboMetrics(combo.name); + if (cMetrics) { + const targetKey = orderedTargets[i].executionKey || modelStr; + const m = cMetrics.byTarget[targetKey] || cMetrics.byModel[modelStr]; + if (m && m.requests >= 5 && m.avgLatencyMs > config.predictiveTtftMs) { + log.warn( + "COMBO", + `Predictive TTFT Circuit Breaker: skipping ${modelStr} (avg ${m.avgLatencyMs}ms > max ${config.predictiveTtftMs}ms)` + ); + return null; + } + } + } + + if (retry > 0) { + log.info( + "COMBO", + `Retrying ${modelStr} in ${retryDelayMs}ms (attempt ${retry + 1}/${maxRetries + 1})` + ); + await new Promise((resolve) => { + const timer = setTimeout(resolve, retryDelayMs); + signal?.addEventListener( + "abort", + () => { + clearTimeout(timer); + resolve(undefined); + }, + { once: true } + ); + }); + if (signal?.aborted) { + log.info("COMBO", `Client disconnected during retry delay — aborting`); + return { ok: false, response: errorResponse(499, "Client disconnected") }; + } + } + + log.info( + "COMBO", + `Trying model ${i + 1}/${orderedTargets.length}: ${modelStr}${retry > 0 ? ` (retry ${retry})` : ""}` + ); + emit("combo.target.attempt", { + comboName: combo.name, + targetIndex: i, + provider, + model: modelStr, + timestamp: Date.now(), + strategy, + }); + + // Deep clone the body to ensure context preservation and prevent mutations + // from affecting other targets in the combo + let attemptBody = JSON.parse(JSON.stringify(body)); + + // Proactive Context Compression for fallbacks (Zero-Latency optimization) + if ( + zeroLatencyOptimizationsEnabled && + i > 0 && + config.fallbackCompressionMode && + config.fallbackCompressionMode !== "off" + ) { + const { estimateTokens } = await import("../contextManager.ts"); + const estimatedTokens = estimateTokens(JSON.stringify(attemptBody)); + if (estimatedTokens > (config.fallbackCompressionThreshold ?? 1000)) { + const { applyCompression } = await import("../compression/strategySelector.ts"); + const compressionResult = applyCompression( + attemptBody, + config.fallbackCompressionMode as CompressionMode, + { model: modelStr } + ); + if (compressionResult.compressed) { + log.info( + "COMBO", + `Proactive fallback compression applied (${config.fallbackCompressionMode}): ${estimatedTokens} -> ${compressionResult.stats?.compressedTokens} tokens` + ); + attemptBody = compressionResult.body; + } + } + } + + // Universal handoff: inject existing handoff if model changed + if ( + universalHandoffConfig.enabled && + relayOptions?.sessionId && + !(body as Record)?.[SKIP_UNIVERSAL_HANDOFF_FLAG] + ) { + const lastModel = getLastSessionModel(relayOptions.sessionId, combo.name); + if (lastModel && lastModel !== modelStr) { + const existingHandoff = getHandoff(relayOptions.sessionId, combo.name); + attemptBody = injectUniversalHandoffBody( + attemptBody, // Use the cloned body to maintain isolation + lastModel, + modelStr, + `Model routing: ${lastModel} → ${modelStr}`, + existingHandoff + ); + } + } + + // Issue #3587: Reasoning models (deepseek-v4-flash, nemotron, etc.) consume + // ALL max_tokens for reasoning_tokens, leaving content empty. Add a buffer + // to max_tokens so the model has enough tokens for both reasoning and content. + if (supportsReasoning(modelStr)) { + const currentMaxTokens = Number((attemptBody as Record).max_tokens) || 0; + if (currentMaxTokens > 0) { + const bufferedMaxTokens = Math.max( + currentMaxTokens + 1000, + Math.ceil(currentMaxTokens * 1.5) + ); + attemptBody = { + ...(attemptBody as Record), + max_tokens: bufferedMaxTokens, + } as typeof attemptBody; + log.info( + "COMBO", + `Reasoning model ${modelStr}: buffered max_tokens ${currentMaxTokens} -> ${bufferedMaxTokens}` + ); + } + } + const result = await handleSingleModelWithTimeout(attemptBody, modelStr, { + ...targetForAttempt, + failoverBeforeRetry: config.failoverBeforeRetry, + }); + + // Success — validate response quality before returning + if (result.ok) { + const quality = await validateResponseQuality(result, clientRequestedStream, log); + if (!quality.valid) { + log.warn( + "COMBO", + `Model ${modelStr} returned 200 but failed quality check: ${quality.reason}` + ); + recordComboRequest(combo.name, modelStr, { + success: false, + latencyMs: Date.now() - startTime, + fallbackCount, + strategy, + target: toRecordedTarget(target), + }); + recordedAttempts++; + // Fix #1707: Set terminal state so the fallback doesn't emit + // misleading ALL_ACCOUNTS_INACTIVE when the real issue is quality. + lastError = `Upstream response failed quality validation: ${quality.reason}`; + if (!lastStatus) lastStatus = 502; + if (i > 0) fallbackCount++; + emit("combo.target.failed", { + comboName: combo.name, + targetIndex: i, + provider, + model: modelStr, + error: `Quality: ${quality.reason}`, + latencyMs: Date.now() - startTime, + }); + return null; + } + const latencyMs = Date.now() - startTime; + emit("combo.target.succeeded", { + comboName: combo.name, + targetIndex: i, + provider, + model: modelStr, + latencyMs, + }); + log.info( + "COMBO", + `Model ${modelStr} succeeded (${latencyMs}ms, ${fallbackCount} fallbacks)` + ); + recordComboRequest(combo.name, modelStr, { + success: true, + latencyMs, + fallbackCount, + strategy, + target: toRecordedTarget(target), + }); + recordedAttempts++; + + // Reset cooldown on success + if (provider && provider !== "unknown") { + recordProviderSuccess(provider, target.connectionId ?? undefined); + } + // Webhook fan-out: best-effort, never blocks the response stream. + notifyWebhookEvent("request.completed", { + combo: combo.name, + provider, + model: modelStr, + latencyMs, + fallbackCount, + }); + + // Context cache pinning: record model usage for session-based pinning + // (independent of universal handoff — always fires when context_cache_protection is on) + if ( + combo.context_cache_protection && + relayOptions?.sessionId && + !(body as Record)?.[SKIP_UNIVERSAL_HANDOFF_FLAG] + ) { + recordSessionModelUsage( + relayOptions.sessionId, + combo.name, + modelStr, + provider, + target.connectionId ?? undefined + ); + } + + // Universal handoff: record model usage for session + if ( + universalHandoffConfig.enabled && + relayOptions?.sessionId && + !(body as Record)?.[SKIP_UNIVERSAL_HANDOFF_FLAG] + ) { + const prevModel = getLastSessionModel(relayOptions.sessionId, combo.name); + recordSessionModelUsage( + relayOptions.sessionId, + combo.name, + modelStr, + provider, + target.connectionId ?? undefined + ); + if (prevModel && prevModel !== modelStr) { + const handoffSourceMessages = + Array.isArray(body?.messages) && body.messages.length > 0 + ? body.messages + : Array.isArray(body?.input) + ? body.input + : []; + + maybeGenerateUniversalHandoff({ + sessionId: relayOptions.sessionId, + comboName: combo.name, + messages: handoffSourceMessages as MessageLike[], + prevModel, + currModel: modelStr, + universalConfig: universalHandoffConfig, + handleSingleModel: handleSingleModelWithTimeout, + }); + } + + recordSessionModelUsage( + relayOptions.sessionId, + combo.name, + modelStr, + provider, + target.connectionId ?? undefined + ); + } + // Context-relay intentionally splits responsibilities: + // combo.ts decides whether a successful turn should generate a handoff, + // while chat.ts injects the handoff after the real connectionId is resolved. + if ( + strategy === "context-relay" && + relayOptions?.sessionId && + relayConfig && + relayConfig.handoffProviders.includes(provider) && + provider === "codex" + ) { + const connectionId = getSessionConnection(relayOptions.sessionId); + if (connectionId) { + const quotaInfo = await fetchCodexQuota(connectionId).catch(() => null); + if (quotaInfo) { + const resetCandidates = [ + quotaInfo.windows?.session?.resetAt, + quotaInfo.windows?.weekly?.resetAt, + quotaInfo.resetAt, + ] + .filter( + (value): value is string => typeof value === "string" && value.length > 0 + ) + .sort((a, b) => a.localeCompare(b)); + const handoffSourceMessages = + Array.isArray(body?.messages) && body.messages.length > 0 + ? body.messages + : Array.isArray(body?.input) + ? body.input + : []; + + maybeGenerateHandoff({ + sessionId: relayOptions.sessionId, + comboName: combo.name, + connectionId, + percentUsed: quotaInfo.percentUsed, + messages: handoffSourceMessages, + model: modelStr, + expiresAt: resetCandidates[0] || null, + config: relayConfig, + handleSingleModel: handleSingleModelWithTimeout, + }); + } + } + } + + // Record last known good provider (LKGP) for this combo/model (#919) + if (provider) { + const connId = target.connectionId || undefined; + void (async () => { + try { + const { setLKGP } = await import("../../../src/lib/localDb"); + await Promise.all([ + setLKGP(combo.name, target.executionKey, provider, connId), + setLKGP(combo.name, combo.id || combo.name, provider, connId), + ]); + } catch (err) { + log.warn( + "COMBO", + "Failed to record Last Known Good Provider. This is non-fatal.", + { + err, + } + ); + } + })(); + } + + return { ok: true, response: quality.clonedResponse ?? result }; + } + + // Extract error info from response + let errorText = result.statusText || ""; + let errorBody: ComboErrorBody = null; + let retryAfter: ComboRetryAfter | null = null; + try { + const cloned = result.clone(); + try { + const text = await cloned.text(); + if (text) { + errorText = text.substring(0, 500); + errorBody = JSON.parse(text); + const parsedError = errorBody?.error; + errorText = + (typeof parsedError === "object" && parsedError?.message) || + (typeof parsedError === "string" ? parsedError : null) || + errorBody?.message || + errorText; + retryAfter = errorBody?.retryAfter || null; + } + } catch { + /* Clone parse failed */ + } + } catch { + /* Clone failed */ + } + + // Track earliest retryAfter + if ( + retryAfter && + (!earliestRetryAfter || new Date(retryAfter) < new Date(earliestRetryAfter)) + ) { + earliestRetryAfter = retryAfter; + } + + // Normalize error text + if (typeof errorText !== "string") { + try { + errorText = JSON.stringify(errorText); + } catch { + errorText = String(errorText); + } + } + + const isStreamReadinessFailure = + (result.status === 502 || result.status === 504) && + isStreamReadinessFailureErrorBody(errorBody); + + // FIX 5: a local per-API-key token-limit 429 must not cool shared accounts. + const isTokenLimitBreach = + result.status === 429 && isTokenLimitBreachErrorBody(errorBody); + + // Fix #1681: Status 499 means client disconnected — stop combo loop immediately. + // There is no point trying fallback models when nobody is listening. + if (result.status === 499) { + log.info("COMBO", `Client disconnected (499) during ${modelStr} — stopping combo loop`); + recordComboRequest(combo.name, modelStr, { + success: false, + latencyMs: Date.now() - startTime, + fallbackCount, + strategy, + target: toRecordedTarget(target), + }); + recordedAttempts++; + // executeTarget must return the {ok,response} contract — a raw Response + // here makes the speculative loop's res.ok/res.response checks both miss, + // so the combo would wrongly fall through to the next model after a 499. + return { ok: false, response: result }; + } + + // Combo fallback is target-level orchestration: a non-ok target response is + // treated as local to that target and the combo continues to the next target. + // Error classification is retained only for retry/cooldown pacing; it must + // not decide whether fallback happens, including for generic 400 responses. + const rawError = errorBody?.error; + const structuredError = + rawError && typeof rawError === "object" + ? { + // Upstream JSON may carry a numeric `code`/`type` (e.g. {"code":40001}). + // Coerce to string if present instead of discarding, so downstream string + // ops (.toLowerCase, .startsWith) can run safely without type crashes. + code: + (rawError as Record).code !== undefined && + (rawError as Record).code !== null + ? String((rawError as Record).code) + : undefined, + type: + (rawError as Record).type !== undefined && + (rawError as Record).type !== null + ? String((rawError as Record).type) + : undefined, + } + : undefined; + const fallbackResult = checkFallbackError( + result.status, + errorText, + 0, + null, + provider, + result.headers, + profile, + structuredError + ); + const { cooldownMs } = fallbackResult; + + // #1731: If the entire provider quota is exhausted, mark it so subsequent + // same-provider targets are skipped immediately. API-key 429s still use + // the short resilience cooldown, but explicit quota text should stop the + // combo from trying another target for the same provider in this request. + const providerExhausted = + Boolean(provider && provider !== "unknown") && + (isProviderExhaustedReason(fallbackResult) || + classifyErrorText(errorText) === RateLimitReason.QUOTA_EXHAUSTED); + if (providerExhausted) { + exhaustedProviders.add(provider); + log.info( + "COMBO", + `Provider ${provider} quota exhausted — marking for skip on remaining targets (#1731)` + ); + } else if ( + result.status === 429 && + !isTokenLimitBreach && + provider && + provider !== "unknown" + ) { + transientRateLimitedProviders.add(provider); + } + + // #2101: Prevent infinite fallback loops with 400 Bad Request errors that indicate + // request-body-specific issues (context overflow, malformed request, model access denied). + // These errors are unlikely to be resolved by trying different target models since + // the same problematic request body would be sent to all targets. + if ( + result.status === 400 && + fallbackResult.shouldFallback && + (fallbackResult.reason === RateLimitReason.MODEL_CAPACITY || + errorText.toLowerCase().includes("context") || + errorText.toLowerCase().includes("prompt") || + errorText.toLowerCase().includes("token") || + errorText.toLowerCase().includes("malformed") || + errorText.toLowerCase().includes("invalid") || + errorText.toLowerCase().includes("bad request")) + ) { + log.warn( + "COMBO", + `400 Bad Request with body-specific error detected on ${modelStr} — skipping fallback to other targets to prevent infinite loop` + ); + // Record the failure and break to avoid trying other targets with the same bad request + recordComboRequest(combo.name, modelStr, { + success: false, + latencyMs: Date.now() - startTime, + fallbackCount, + strategy, + target: toRecordedTarget(target), + }); + recordedAttempts++; + lastError = errorText || String(result.status); + if (!lastStatus) lastStatus = result.status; + if (i > 0) fallbackCount++; + log.warn("COMBO", `Model ${modelStr} failed with body-specific error, stopping combo`); + break; // Break out of the target loop to avoid trying other models + } + + // Trigger shared provider circuit breaker for 5xx errors and connection failures. + // If the next target in the combo is on the same provider, don't mark the provider + // as failed — different models on the same provider may still succeed. + // G-02: when fallbackResult.skipProviderBreaker is set (embedded service supervisor + // outage signalled via X-Omni-Fallback-Hint: connection_cooldown) apply connection + // cooldown only — do NOT trip the whole-provider breaker. + const nextTarget = orderedTargets[i + 1]; + const sameProviderNext = + typeof nextTarget?.provider === "string" && nextTarget.provider === provider; + if ( + !isStreamReadinessFailure && + isProviderFailureCode(result.status) && + !sameProviderNext && + !fallbackResult.skipProviderBreaker + ) { + recordProviderFailure(provider, log, target.connectionId, profile); + } + + // Check if this is a transient error worth retrying on same model. + // A token-limit 429 is terminal for the client — never retry it. + const isTransient = + !isStreamReadinessFailure && + !isTokenLimitBreach && + [408, 429, 500, 502, 503, 504].includes(result.status); + if (retry < maxRetries && isTransient && !providerExhausted) { + continue; // Retry same model + } + + // Done retrying this model + recordComboRequest(combo.name, modelStr, { + success: false, + latencyMs: Date.now() - startTime, + fallbackCount, + strategy, + target: toRecordedTarget(target), + }); + recordedAttempts++; + lastError = errorText || String(result.status); + if (!lastStatus) lastStatus = result.status; + if (i > 0) fallbackCount++; + log.warn("COMBO", `Model ${modelStr} failed, trying next`, { status: result.status }); + + if (resilienceSettings.providerCooldown.enabled && provider && provider !== "unknown") { + recordProviderCooldown(provider, target.connectionId ?? undefined, resilienceSettings); + } + + const fallbackWaitMs = + fallbackDelayMs > 0 && cooldownMs > 0 && cooldownMs <= MAX_FALLBACK_WAIT_MS + ? Math.min(cooldownMs, fallbackDelayMs) + : 0; + if ([502, 503, 504].includes(result.status) && fallbackWaitMs > 0) { + log.debug?.("COMBO", `Waiting ${fallbackWaitMs}ms before fallback to next model`); + await new Promise((resolve) => { + const timer = setTimeout(resolve, fallbackWaitMs); + signal?.addEventListener( + "abort", + () => { + clearTimeout(timer); + resolve(undefined); + }, + { once: true } + ); + }); + if (signal?.aborted) { + log.info("COMBO", `Client disconnected during fallback wait — aborting`); + return { ok: false, response: errorResponse(499, "Client disconnected") }; + } + } + + return null; + } + return null; + }; + + for (let i = 0; i < orderedTargets.length; i++) { + if (anySuccess) break; + + const abortController = new AbortController(); + abortControllers.set(i, abortController); + const onClientAbort = () => abortController.abort(); + signal?.addEventListener("abort", onClientAbort); + + const task = (async () => { + try { + const res = await executeTarget(i); + if (res && !anySuccess) { + if (res.ok) { + anySuccess = true; + globalResolve!(res.response!); + for (const [idx, ac] of abortControllers.entries()) { + if (idx !== i) ac.abort(); + } + } else if (res.response) { + // Fatal error, abort combo + anySuccess = true; + globalResolve!(res.response); + } + } + } finally { + signal?.removeEventListener("abort", onClientAbort); + } + })().catch((err) => { + const logError = log.error ?? log.warn; + logError("COMBO", `Speculative task error for target ${i}`, err); + }); + + runningTasks.add(task); + task.finally(() => runningTasks.delete(task)); + + if (zeroLatencyOptimizationsEnabled && config.hedging && i + 1 < orderedTargets.length) { + const hedgeDelay = resolveDelayMs(config.hedgeDelayMs, 500); + let timeoutResolve: () => void; + const timeoutPromise = new Promise((r) => { + timeoutResolve = r; + setTimeout(r, hedgeDelay); + }); + await Promise.race([task, globalPromise, timeoutPromise]); + } else { + await Promise.race([task, globalPromise]); + } + } + + if (!anySuccess && runningTasks.size > 0) { + await Promise.race([globalPromise, Promise.all([...runningTasks])]); + } + + if (anySuccess) { + return await globalPromise; + } + + // All models failed in this set try + const latencyMs = Date.now() - startTime; + if (recordedAttempts === 0) { + recordComboRequest(combo.name, null, { + success: false, + latencyMs, + fallbackCount, + strategy, + }); + } + + // Retry the entire set if more attempts remain + if (setTry < maxSetRetries) continue; + + // All set retries exhausted — return the final error + if (!lastStatus) { + notifyWebhookEvent("request.failed", { + combo: combo.name, + reason: "ALL_ACCOUNTS_INACTIVE", + latencyMs, + fallbackCount, + }); + return new Response( + JSON.stringify({ + error: { + message: "Service temporarily unavailable: all upstream accounts are inactive", + type: "service_unavailable", + code: "ALL_ACCOUNTS_INACTIVE", + }, + }), + { status: 503, headers: { "Content-Type": "application/json" } } + ); + } + + const status = lastStatus; + const msg = lastError || "All combo models unavailable"; + + if (earliestRetryAfter) { + const retryHuman = formatRetryAfter(toRetryAfterDisplayValue(earliestRetryAfter)); + log.warn("COMBO", `All models failed | ${msg} (${retryHuman})`); + return unavailableResponse(status, msg, earliestRetryAfter, retryHuman); + } + + log.warn("COMBO", `All models failed | ${msg}`); + return new Response(JSON.stringify({ error: { message: msg } }), { + status, + headers: { "Content-Type": "application/json" }, + }); + } + + return errorResponse(503, "Combo routing completed without an upstream response"); + } finally { + // G2: Clean up candidate registry to prevent unbounded memory growth. + _unregisterExecutionCandidates(_registeredExecutionKeys); + } +} + diff --git a/open-sse/services/combo/constants.ts b/open-sse/services/combo/constants.ts new file mode 100644 index 00000000000..f63f8cb5b13 --- /dev/null +++ b/open-sse/services/combo/constants.ts @@ -0,0 +1,148 @@ +/** + * Shared combo (model combo) handling with fallback support + * Supports: priority, weighted, round-robin, random, least-used, cost-optimized, + * reset-aware, reset-window, strict-random, auto, fill-first, p2c, lkgp, + * context-optimized, and context-relay strategies + */ + +import { + checkFallbackError, + classifyErrorText, + formatRetryAfter, + getRuntimeProviderProfile, + recordProviderFailure, + isProviderFailureCode, + isProviderExhaustedReason, + type ProviderProfile, +} from "../accountFallback.ts"; + +import { errorResponse, unavailableResponse } from "../../utils/error.ts"; + +import { clamp01 } from "../../utils/number.ts"; + +import { parseModel } from "../model.ts"; + +import { + calculateFactors, + calculateScore, + DEFAULT_WEIGHTS, + type ProviderCandidate, + type ScoringWeights, +} from "../autoCombo/scoring.ts"; + + + + +// Status codes that should mark round-robin target semaphores as cooling down. +export const TRANSIENT_FOR_SEMAPHORE = [429, 502, 503, 504]; + + +// Patterns that signal all accounts for a provider are rate-limited / exhausted. +// Used to detect 503 responses from handleNoCredentials so combo can fallback. +const ALL_ACCOUNTS_RATE_LIMITED_PATTERNS = [/unavailable/i, /service temporarily unavailable/i]; + + + +export function isAllAccountsRateLimitedResponse( + status: number, + contentType: string | null, + errorText: string +): boolean { + if (status !== 503) return false; + if (!contentType?.includes("application/json")) return false; + return ALL_ACCOUNTS_RATE_LIMITED_PATTERNS.some((p) => p.test(errorText)); +} + + + +export const MAX_COMBO_DEPTH = 3; + + +export const MAX_FALLBACK_WAIT_MS = 5000; + + +export const MAX_GLOBAL_ATTEMPTS = 30; + + + +export function resolveDelayMs(value: unknown, fallback: number): number { + const numericValue = Number(value); + if (!Number.isFinite(numericValue) || numericValue < 0) return fallback; + return numericValue; +} + + + +export function comboModelNotFoundResponse(message: string) { + return errorResponse(404, message); +} + + + +// Bootstrap defaults from ClawRouter benchmark (used when no local latency history exists yet) +export const DEFAULT_MODEL_P95_MS: Record = { + "grok-4-fast-non-reasoning": 1143, + "grok-4-1-fast-non-reasoning": 1244, + "gemini-2.5-flash": 1238, + "kimi-k2.5": 1646, + "gpt-4o-mini": 2764, + "claude-sonnet-4.6": 4000, + "claude-opus-4.6": 6000, + "deepseek-chat": 2000, +}; + + +export const MIN_HISTORY_SAMPLES = 10; + + +// Assumed fraction of tokens that are output when blending input+output prices +// for auto-combo cost scoring. 0.4 = 40% output, 60% input. +// Matches the example in GitHub issue #1812 (e.g. o3-like model: $3 input/$15 output). +export const OUTPUT_TOKEN_RATIO = 0.4; + + +export const RESET_AWARE_SESSION_WINDOW_MS = 5 * 60 * 60 * 1000; + + +export const RESET_AWARE_WEEKLY_WINDOW_MS = 7 * 24 * 60 * 60 * 1000; + + +export const RESET_AWARE_SESSION_REMAINING_WEIGHT = 0.45; + + +export const RESET_AWARE_SESSION_RESET_PRESSURE_WEIGHT = 0.55; + + +export const RESET_AWARE_WEEKLY_REMAINING_WEIGHT = 0.25; + + +export const RESET_AWARE_WEEKLY_RESET_PRESSURE_WEIGHT = 0.75; + + +export const RESET_AWARE_CONNECTION_CACHE_TTL_MS = 30_000; + + +export const RESET_AWARE_QUOTA_FETCH_CONCURRENCY = 5; + + +export const RESET_AWARE_DEFAULTS = { + sessionWeight: 0.35, + weeklyWeight: 0.65, + tieBandPercent: 5, + exhaustionGuardPercent: 10, +}; + + +export const RESET_WINDOW_DEFAULT_TIE_BAND_MS = 60_000; + + + +// Quota Share soft-policy deprioritization factor (B17). +// When a candidate has quotaSoftPenalty === true, its auto-combo score is +// multiplied by this factor so over-quota-soft keys are de-prioritized +// without being fully blocked (that is done by "hard" policy). +// Override via QUOTA_SOFT_DEPRIORITIZE_FACTOR env var (range 0..1, default 0.7). +export const QUOTA_SOFT_DEPRIORITIZE_FACTOR = Number( + process.env.QUOTA_SOFT_DEPRIORITIZE_FACTOR ?? "0.7" +); + diff --git a/open-sse/services/combo/context.ts b/open-sse/services/combo/context.ts new file mode 100644 index 00000000000..cb26bef62c8 --- /dev/null +++ b/open-sse/services/combo/context.ts @@ -0,0 +1,319 @@ +/** + * Shared combo (model combo) handling with fallback support + * Supports: priority, weighted, round-robin, random, least-used, cost-optimized, + * reset-aware, reset-window, strict-random, auto, fill-first, p2c, lkgp, + * context-optimized, and context-relay strategies + */ + +import { + checkFallbackError, + classifyErrorText, + formatRetryAfter, + getRuntimeProviderProfile, + recordProviderFailure, + isProviderFailureCode, + isProviderExhaustedReason, + type ProviderProfile, +} from "../accountFallback.ts"; + +import { FETCH_TIMEOUT_MS, RateLimitReason } from "../../config/constants.ts"; + +import { clamp01 } from "../../utils/number.ts"; + +import { + classifyWithConfig, + DEFAULT_INTENT_CONFIG, + type IntentClassifierConfig, +} from "../intentClassifier.ts"; + +import { + getResolvedModelCapabilities, + supportsReasoning, + supportsToolCalling, +} from "../modelCapabilities.ts"; + +import { estimateTokens } from "../contextManager.ts"; + +import { getSessionConnection } from "../sessionManager.ts"; + +import { getProviderModels } from "../../config/providerModels.ts"; + +import { + resolveResilienceSettings, + type ResilienceSettings, +} from "../../../src/lib/resilience/settings"; + +import { isRecord, toTextContent, toStringArray } from "./utils.ts"; +import { RequestCompatibilityRequirements, ResolvedComboTarget, ComboLogger, ComboLike } from "./types.ts"; +import { DEFAULT_MODEL_P95_MS } from "./constants.ts"; + + + +function getPositiveTokenCount(value: unknown): number { + const count = Number(value); + return Number.isFinite(count) && count > 0 ? Math.ceil(count) : 0; +} + + + +function requestRequiresTools(body: Record): boolean { + if (Array.isArray(body.tools) && body.tools.length > 0) return true; + if (Array.isArray(body.functions) && body.functions.length > 0) return true; + return false; +} + + + +function requestRequiresStructuredOutput(body: Record): boolean { + const responseFormat = isRecord(body.response_format) ? body.response_format : null; + const type = typeof responseFormat?.type === "string" ? responseFormat.type : null; + return type === "json_object" || type === "json_schema"; +} + + + +function estimateRequestInputTokens(body: Record): number { + const estimatePayload: Record = {}; + for (const key of ["messages", "input", "tools", "functions", "response_format"]) { + if (body[key] !== undefined) estimatePayload[key] = body[key]; + } + return Object.keys(estimatePayload).length > 0 ? estimateTokens(estimatePayload) : 0; +} + + + +function valueContainsImagePart(value: unknown, depth = 0): boolean { + if (depth > 8 || value === null || value === undefined) return false; + if (typeof value === "string") return value.startsWith("data:image/"); + if (Array.isArray(value)) return value.some((entry) => valueContainsImagePart(entry, depth + 1)); + if (!isRecord(value)) return false; + + const type = typeof value.type === "string" ? value.type.toLowerCase() : null; + if (type === "image" || type === "image_url" || type === "input_image") return true; + if ("image_url" in value || "input_image" in value) return true; + + const source = isRecord(value.source) ? value.source : null; + const mediaType = typeof source?.media_type === "string" ? source.media_type.toLowerCase() : ""; + if (mediaType.startsWith("image/")) return true; + + return Object.values(value).some((entry) => valueContainsImagePart(entry, depth + 1)); +} + + + +function deriveRequestCompatibilityRequirements( + body: Record +): RequestCompatibilityRequirements { + const estimatedInputTokens = estimateRequestInputTokens(body); + const requestedOutputTokens = Math.max( + getPositiveTokenCount(body.max_tokens), + getPositiveTokenCount(body.max_completion_tokens) + ); + return { + requiresTools: requestRequiresTools(body), + requiresVision: valueContainsImagePart(body.messages) || valueContainsImagePart(body.input), + requiresStructuredOutput: requestRequiresStructuredOutput(body), + estimatedInputTokens, + requestedOutputTokens, + requiredContextTokens: estimatedInputTokens + requestedOutputTokens, + }; +} + + + +function getTargetCompatibilityFailures( + target: ResolvedComboTarget, + requirements: RequestCompatibilityRequirements +): string[] { + const capabilities = getResolvedModelCapabilities(target.modelStr); + const failures: string[] = []; + + if ( + requirements.requiresTools && + (capabilities.supportsTools === false || !capabilities.toolCalling) + ) { + failures.push("tools"); + } + + if (requirements.requiresVision && capabilities.supportsVision === false) { + failures.push("vision"); + } + + if (requirements.requiresStructuredOutput && capabilities.structuredOutput === false) { + failures.push("structured_output"); + } + + if ( + requirements.requestedOutputTokens > 0 && + Number.isFinite(capabilities.maxOutputTokens) && + capabilities.maxOutputTokens < requirements.requestedOutputTokens + ) { + failures.push("output_tokens"); + } + + const contextLimit = capabilities.maxInputTokens ?? capabilities.contextWindow ?? null; + if ( + requirements.requiredContextTokens > 0 && + contextLimit !== null && + contextLimit !== undefined && + contextLimit < requirements.requiredContextTokens + ) { + failures.push("context_window"); + } + + return failures; +} + + + +export function filterTargetsByRequestCompatibility( + targets: ResolvedComboTarget[], + body: Record, + log: ComboLogger, + label = "Context-aware fallback" +): ResolvedComboTarget[] { + if (targets.length === 0) return targets; + const requirements = deriveRequestCompatibilityRequirements(body); + const needsFiltering = + requirements.requiresTools || + requirements.requiresVision || + requirements.requiresStructuredOutput || + requirements.requiredContextTokens > 0; + if (!needsFiltering) return targets; + + const rejected: Array<{ target: ResolvedComboTarget; reasons: string[] }> = []; + const compatible = targets.filter((target) => { + const reasons = getTargetCompatibilityFailures(target, requirements); + if (reasons.length === 0) return true; + rejected.push({ target, reasons }); + return false; + }); + + if (compatible.length === targets.length) return targets; + if (compatible.length === 0) { + log.warn( + "COMBO", + `${label}: all ${targets.length} targets were filtered by request requirements; preserving strategy order` + ); + log.debug?.( + "COMBO", + `${label}: rejected targets ${rejected + .map((entry) => `${entry.target.modelStr}(${entry.reasons.join("+")})`) + .join(", ")}` + ); + return targets; + } + + log.info( + "COMBO", + `${label}: kept ${compatible.length}/${targets.length} targets for request requirements` + ); + log.debug?.( + "COMBO", + `${label}: rejected targets ${rejected + .map((entry) => `${entry.target.modelStr}(${entry.reasons.join("+")})`) + .join(", ")}` + ); + return compatible; +} + + + +export function extractPromptForIntent(body: Record | null | undefined): string { + if (!body || typeof body !== "object") return ""; + + const fromMessages = Array.isArray(body.messages) + ? [...body.messages].reverse().find((m) => isRecord(m) && m.role === "user") + : null; + if (isRecord(fromMessages)) return toTextContent(fromMessages.content); + + if (typeof body.input === "string") return body.input; + if (Array.isArray(body.input)) { + const text = body.input + .map((item) => { + if (!isRecord(item)) return ""; + if (typeof item.content === "string") return item.content; + if (typeof item.text === "string") return item.text; + return ""; + }) + .filter(Boolean) + .join("\n"); + if (text) return text; + } + + if (typeof body.prompt === "string") return body.prompt; + return ""; +} + + + +export function mapIntentToTaskType(intent: string): "coding" | "analysis" | "default" { + switch (intent) { + case "code": + return "coding"; + case "reasoning": + return "analysis"; + case "simple": + return "default"; + case "medium": + default: + return "default"; + } +} + + + +export function calculateTargetContextAffinity( + target: ResolvedComboTarget, + sessionId: string | null | undefined +): number { + const sessionConnectionId = getSessionConnection(sessionId || null); + if (!sessionConnectionId) return 0.5; + if (target.connectionId === sessionConnectionId) return 1; + if (!target.connectionId) return 0.5; + return 0.1; +} + + + +export function getIntentConfig( + settings: Record | null | undefined, + combo: ComboLike +): IntentClassifierConfig { + const resolvedSettings = settings || {}; + const comboAutoConfig = combo?.autoConfig || {}; + const comboConfigAuto = isRecord(combo?.config?.auto) ? combo.config.auto : {}; + const comboIntentConfig = + (isRecord(comboAutoConfig.intentConfig) && comboAutoConfig.intentConfig) || + (isRecord(comboConfigAuto.intentConfig) && comboConfigAuto.intentConfig) || + (isRecord(combo?.config?.intentConfig) && combo.config.intentConfig) || + {}; + + return { + ...DEFAULT_INTENT_CONFIG, + ...comboIntentConfig, + ...(typeof resolvedSettings.intentDetectionEnabled === "boolean" + ? { enabled: resolvedSettings.intentDetectionEnabled } + : {}), + ...(Number.isFinite(Number(resolvedSettings.intentSimpleMaxWords)) + ? { simpleMaxWords: Number(resolvedSettings.intentSimpleMaxWords) } + : {}), + ...(toStringArray(resolvedSettings.intentExtraCodeKeywords).length > 0 + ? { extraCodeKeywords: toStringArray(resolvedSettings.intentExtraCodeKeywords) } + : {}), + ...(toStringArray(resolvedSettings.intentExtraReasoningKeywords).length > 0 + ? { extraReasoningKeywords: toStringArray(resolvedSettings.intentExtraReasoningKeywords) } + : {}), + ...(toStringArray(resolvedSettings.intentExtraSimpleKeywords).length > 0 + ? { extraSimpleKeywords: toStringArray(resolvedSettings.intentExtraSimpleKeywords) } + : {}), + }; +} + + + +export function getBootstrapLatencyMs(modelId: string): number { + const normalized = String(modelId || "").toLowerCase(); + return DEFAULT_MODEL_P95_MS[normalized] ?? 1500; +} + diff --git a/open-sse/services/combo/dag.ts b/open-sse/services/combo/dag.ts new file mode 100644 index 00000000000..bcac74e15a3 --- /dev/null +++ b/open-sse/services/combo/dag.ts @@ -0,0 +1,383 @@ +/** + * Shared combo (model combo) handling with fallback support + * Supports: priority, weighted, round-robin, random, least-used, cost-optimized, + * reset-aware, reset-window, strict-random, auto, fill-first, p2c, lkgp, + * context-optimized, and context-relay strategies + */ + +import { + checkFallbackError, + classifyErrorText, + formatRetryAfter, + getRuntimeProviderProfile, + recordProviderFailure, + isProviderFailureCode, + isProviderExhaustedReason, + type ProviderProfile, +} from "../accountFallback.ts"; + +import { FETCH_TIMEOUT_MS, RateLimitReason } from "../../config/constants.ts"; + +import { clamp01 } from "../../utils/number.ts"; + +import { parseModel } from "../model.ts"; + +import { getProviderModels } from "../../config/providerModels.ts"; + +import { + getComboModelString, + getComboStepTarget, + getComboStepWeight, + normalizeComboStep, +} from "../../../src/lib/combos/steps.ts"; + +import { ComboCollectionLike, ComboRuntimeStep, ResolvedComboTarget, ComboLike } from "./types.ts"; +import { getTargetProvider, isRecord, toTrimmedString, getCombosArray, normalizeModelEntry } from "./utils.ts"; +import { MAX_COMBO_DEPTH } from "./constants.ts"; + + + +function buildExecutionKey(path: string[], stepId: string): string { + return [...path, stepId].join(">"); +} + + + +function normalizeRuntimeStep( + entry: unknown, + comboName: string, + index: number, + allCombos: ComboCollectionLike, + path: string[] = [] +): ComboRuntimeStep | null { + const step = normalizeComboStep(entry, { + comboName, + index, + allCombos, + }); + if (!step) return null; + + const executionKey = buildExecutionKey(path, step.id); + const label = typeof step.label === "string" ? step.label : null; + const weight = step.weight || 0; + + if (step.kind === "combo-ref") { + return { + kind: "combo-ref", + stepId: step.id, + executionKey, + comboName: step.comboName, + weight, + label, + }; + } + + const modelStr = getComboModelString(step); + if (!modelStr) return null; + + return { + kind: "model", + stepId: step.id, + executionKey, + modelStr, + provider: getTargetProvider(modelStr, step.providerId), + providerId: step.providerId || null, + connectionId: step.connectionId || null, + weight, + label, + } satisfies ResolvedComboTarget; +} + + + +export function getDirectComboTargets(combo: ComboLike): ResolvedComboTarget[] { + return getOrderedTopLevelRuntimeSteps(combo, null).filter( + (entry): entry is ResolvedComboTarget => entry?.kind === "model" + ); +} + + + +function getTopLevelRuntimeSteps( + combo: ComboLike, + allCombos: ComboCollectionLike, + path: string[] = [] +): ComboRuntimeStep[] { + return (combo.models || []) + .map((entry, index) => normalizeRuntimeStep(entry, combo.name, index, allCombos, path)) + .filter((entry): entry is ComboRuntimeStep => entry !== null); +} + + + +function getCompositeTierStepOrder(combo: ComboLike): string[] { + const compositeTiers = isRecord(combo?.config) ? combo.config.compositeTiers : null; + if (!isRecord(compositeTiers)) return []; + + const defaultTier = toTrimmedString(compositeTiers.defaultTier); + const tiers = isRecord(compositeTiers.tiers) ? compositeTiers.tiers : null; + if (!defaultTier || !tiers) return []; + + const orderedStepIds: string[] = []; + const visitedTiers = new Set(); + const seenStepIds = new Set(); + type CompositeTierEntry = readonly [ + string, + { readonly stepId: string; readonly fallbackTier: string | null }, + ]; + const tierEntries = new Map( + Object.entries(tiers) + .map(([tierName, rawTier]) => { + if (!isRecord(rawTier)) return null; + const normalizedTierName = toTrimmedString(tierName); + const stepId = toTrimmedString(rawTier.stepId); + const fallbackTier = toTrimmedString(rawTier.fallbackTier); + if (!normalizedTierName || !stepId) return null; + return [normalizedTierName, { stepId, fallbackTier }] as const; + }) + .filter((entry): entry is CompositeTierEntry => entry !== null) + ); + + let currentTier: string | null = defaultTier; + while (currentTier && tierEntries.has(currentTier) && !visitedTiers.has(currentTier)) { + visitedTiers.add(currentTier); + const entry = tierEntries.get(currentTier); + if (!entry) break; + if (!seenStepIds.has(entry.stepId)) { + orderedStepIds.push(entry.stepId); + seenStepIds.add(entry.stepId); + } + currentTier = entry.fallbackTier; + } + + for (const entry of tierEntries.values()) { + if (!seenStepIds.has(entry.stepId)) { + orderedStepIds.push(entry.stepId); + seenStepIds.add(entry.stepId); + } + } + + return orderedStepIds; +} + + + +export function hasCompositeTierRuntimeOrder(combo: ComboLike): boolean { + return getCompositeTierStepOrder(combo).length > 0; +} + + + +function orderRuntimeStepsByCompositeTiers( + steps: ComboRuntimeStep[], + combo: ComboLike +): ComboRuntimeStep[] { + const orderedStepIds = getCompositeTierStepOrder(combo); + if (orderedStepIds.length === 0) return steps; + + const byStepId = new Map(steps.map((step) => [step.stepId, step])); + const seen = new Set(); + const ordered: ComboRuntimeStep[] = []; + + for (const stepId of orderedStepIds) { + const step = byStepId.get(stepId); + if (!step || seen.has(step.stepId)) continue; + ordered.push(step); + seen.add(step.stepId); + } + + for (const step of steps) { + if (seen.has(step.stepId)) continue; + ordered.push(step); + seen.add(step.stepId); + } + + return ordered; +} + + + +export function getOrderedTopLevelRuntimeSteps( + combo: ComboLike, + allCombos: ComboCollectionLike, + path: string[] = [] +): ComboRuntimeStep[] { + return orderRuntimeStepsByCompositeTiers(getTopLevelRuntimeSteps(combo, allCombos, path), combo); +} + + + +export function expandRuntimeStep( + step: ComboRuntimeStep, + allCombos: ComboCollectionLike, + visited = new Set(), + depth = 0, + path: string[] = [] +): ResolvedComboTarget[] { + if (step.kind === "model") return [step]; + if (depth > MAX_COMBO_DEPTH) return []; + + const combos = getCombosArray(allCombos); + const nestedCombo = combos.find((combo) => combo.name === step.comboName); + if (!nestedCombo || visited.has(step.comboName)) return []; + + return resolveNestedComboTargets(nestedCombo, combos, new Set(visited), depth + 1, [ + ...path, + step.stepId, + ]); +} + + + +export function resolveNestedComboTargets( + combo: ComboLike, + allCombos: ComboCollectionLike, + visited = new Set(), + depth = 0, + path: string[] = [] +): ResolvedComboTarget[] { + const directTargets = (combo.models || []) + .map((entry, index) => normalizeRuntimeStep(entry, combo.name, index, null, path)) + .filter((entry): entry is ResolvedComboTarget => entry?.kind === "model"); + + if (depth > MAX_COMBO_DEPTH) return directTargets; + if (visited.has(combo.name)) return []; + visited.add(combo.name); + + const runtimeSteps = getOrderedTopLevelRuntimeSteps(combo, allCombos, path); + const resolved: ResolvedComboTarget[] = []; + + for (const step of runtimeSteps) { + if (step.kind === "combo-ref") { + resolved.push(...expandRuntimeStep(step, allCombos, new Set(visited), depth, path)); + continue; + } + resolved.push(step); + } + + return resolved; +} + + + +/** + * Get combo models from combos data (for open-sse standalone use) + * @param {string} modelStr - Model string to check + * @param {Array|Object} combosData - Array of combos or object with combos + * @returns {Object|null} Full combo object or null if not a combo + */ +export function getComboFromData( + modelStr: string, + combosData: ComboCollectionLike +): ComboLike | null { + const combos = getCombosArray(combosData); + const combo = combos.find((c) => c.name === modelStr); + if (combo?.models && combo.models.length > 0) { + return combo; + } + return null; +} + + + +/** + * Legacy: Get combo models as string array (backward compat) + */ +export function getComboModelsFromData( + modelStr: string, + combosData: ComboCollectionLike +): string[] | null { + const combo = getComboFromData(modelStr, combosData); + if (!combo) return null; + return combo.models.map((m) => normalizeModelEntry(m).model); +} + + + +/** + * Validate combo DAG — detect circular references and enforce max depth + * @param {string} comboName - Name of the combo to validate + * @param {Array} allCombos - All combos in the system + * @param {Set} [visited] - Set of already visited combo names (for cycle detection) + * @param {number} [depth] - Current depth level + * @throws {Error} If circular reference or max depth exceeded + */ +export function validateComboDAG( + comboName: string, + allCombos: ComboCollectionLike, + visited = new Set(), + depth = 0 +): void { + if (depth > MAX_COMBO_DEPTH) { + throw new Error(`Max combo nesting depth (${MAX_COMBO_DEPTH}) exceeded at "${comboName}"`); + } + if (visited.has(comboName)) { + throw new Error(`Circular combo reference detected: ${comboName}`); + } + visited.add(comboName); + + const combos = getCombosArray(allCombos); + const combo = combos.find((c) => c.name === comboName); + if (!combo?.models) return; + + for (const entry of combo.models) { + const modelName = normalizeModelEntry(entry).model; + // Check if this model name is itself a combo (not a provider/model pattern) + const nestedCombo = combos.find((c) => c.name === modelName); + if (nestedCombo) { + validateComboDAG(modelName, combos, new Set(visited), depth + 1); + } + } +} + + + +/** + * Resolve nested combos by expanding inline to a flat model list + * Respects max depth and detects cycles + * @param {Object} combo - The combo object + * @param {Array} allCombos - All combos in the system + * @param {Set} [visited] - For cycle detection + * @param {number} [depth] - Current depth + * @returns {Array} Flat array of model strings + */ +export function resolveNestedComboModels( + combo: ComboLike, + allCombos: ComboCollectionLike, + visited = new Set(), + depth = 0 +): string[] { + if (depth > MAX_COMBO_DEPTH) return combo.models.map((m) => normalizeModelEntry(m).model); + if (visited.has(combo.name)) return []; // cycle safety + visited.add(combo.name); + + const combos = getCombosArray(allCombos); + const resolved: string[] = []; + + for (const entry of combo.models || []) { + const modelName = normalizeModelEntry(entry).model; + const nestedCombo = combos.find((c) => c.name === modelName); + + if (nestedCombo) { + // Recursively expand the nested combo + const nested = resolveNestedComboModels(nestedCombo, combos, new Set(visited), depth + 1); + resolved.push(...nested); + } else { + resolved.push(modelName); + } + } + + return resolved; +} + + + +export function dedupeTargetsByExecutionKey(targets: ResolvedComboTarget[]) { + const seen = new Set(); + return targets.filter((target) => { + if (seen.has(target.executionKey)) return false; + seen.add(target.executionKey); + return true; + }); +} + diff --git a/open-sse/services/combo/index.ts b/open-sse/services/combo/index.ts new file mode 100644 index 00000000000..30c05abd0c3 --- /dev/null +++ b/open-sse/services/combo/index.ts @@ -0,0 +1,13 @@ +// Re-export everything to maintain backward compatibility +export * from "./constants.ts"; +export * from "./state.ts"; +export * from "./types.ts"; +export * from "./utils.ts"; +export * from "./shadow.ts"; +export * from "./dag.ts"; +export * from "./sorting.ts"; +export * from "./context.ts"; +export * from "./quota.ts"; +export * from "./auto.ts"; +export * from "./chat.ts"; +export * from "./roundRobin.ts"; diff --git a/open-sse/services/combo/quota.ts b/open-sse/services/combo/quota.ts new file mode 100644 index 00000000000..c8a25689348 --- /dev/null +++ b/open-sse/services/combo/quota.ts @@ -0,0 +1,884 @@ +/** + * Shared combo (model combo) handling with fallback support + * Supports: priority, weighted, round-robin, random, least-used, cost-optimized, + * reset-aware, reset-window, strict-random, auto, fill-first, p2c, lkgp, + * context-optimized, and context-relay strategies + */ + +import { + checkFallbackError, + classifyErrorText, + formatRetryAfter, + getRuntimeProviderProfile, + recordProviderFailure, + isProviderFailureCode, + isProviderExhaustedReason, + type ProviderProfile, +} from "../accountFallback.ts"; + +import { FETCH_TIMEOUT_MS, RateLimitReason } from "../../config/constants.ts"; + +import { errorResponse, unavailableResponse } from "../../utils/error.ts"; + +import { clamp01 } from "../../utils/number.ts"; + +import { + resolveComboConfig, + getDefaultComboConfig, + resolveComboTargetTimeoutMs, + PRE_SCREEN_CONCURRENCY, +} from "../comboConfig.ts"; + +import { getQuotaFetcher } from "../quotaPreflight.ts"; + +import { getCircuitBreaker } from "../../../src/shared/utils/circuitBreaker"; + +import { selectWithStrategy, type SlaRoutingPolicy } from "../autoCombo/routerStrategy.ts"; + +import { getProviderConnections } from "../../../src/lib/db/providers"; + +import { getProviderModels } from "../../config/providerModels.ts"; + +import { finiteNumberOrNull, isRecord } from "./utils.ts"; +import { RESET_AWARE_DEFAULTS, RESET_WINDOW_DEFAULT_TIE_BAND_MS, RESET_AWARE_SESSION_WINDOW_MS, RESET_AWARE_SESSION_REMAINING_WEIGHT, RESET_AWARE_SESSION_RESET_PRESSURE_WEIGHT, RESET_AWARE_WEEKLY_WINDOW_MS, RESET_AWARE_WEEKLY_REMAINING_WEIGHT, RESET_AWARE_WEEKLY_RESET_PRESSURE_WEIGHT, RESET_AWARE_CONNECTION_CACHE_TTL_MS, RESET_AWARE_QUOTA_FETCH_CONCURRENCY } from "./constants.ts"; +import { ResetWindowName, RESET_WINDOW_NAMES, ResolvedComboTarget, QuotaFetchCacheConfig, IsModelAvailable, PreScreenResult, ResetWindowConfig } from "./types.ts"; +import { resetAwareConnectionCache, MAX_RESET_AWARE_CACHE, resetAwareQuotaCache, rrCounters, MAX_RR_COUNTERS } from "./state.ts"; + + + +function getPercentConfig(value: unknown, fallback: number): number { + const numericValue = finiteNumberOrNull(value); + if (numericValue === null) return fallback; + return Math.max(0, Math.min(100, numericValue)); +} + + + +function getWeightConfig(value: unknown, fallback: number): number { + const numericValue = finiteNumberOrNull(value); + if (numericValue === null || numericValue < 0) return fallback; + return numericValue; +} + + + +function getDurationConfig(value: unknown, fallback: number, max: number): number { + const numericValue = finiteNumberOrNull(value); + if (numericValue === null || numericValue < 0) return fallback; + return Math.min(max, Math.floor(numericValue)); +} + + + +function resolveResetAwareConfig(config: Record | null | undefined) { + const sessionWeight = getWeightConfig( + config?.resetAwareSessionWeight, + RESET_AWARE_DEFAULTS.sessionWeight + ); + const weeklyWeight = getWeightConfig( + config?.resetAwareWeeklyWeight, + RESET_AWARE_DEFAULTS.weeklyWeight + ); + const totalWeight = sessionWeight + weeklyWeight; + const normalizedSessionWeight = + totalWeight > 0 ? sessionWeight / totalWeight : RESET_AWARE_DEFAULTS.sessionWeight; + + return { + sessionWeight: normalizedSessionWeight, + weeklyWeight: 1 - normalizedSessionWeight, + tieBand: + getPercentConfig(config?.resetAwareTieBandPercent, RESET_AWARE_DEFAULTS.tieBandPercent) / 100, + exhaustionGuard: + getPercentConfig( + config?.resetAwareExhaustionGuardPercent, + RESET_AWARE_DEFAULTS.exhaustionGuardPercent + ) / 100, + quotaCacheTtlMs: getDurationConfig(config?.resetAwareQuotaCacheTtlMs, 0, 300_000), + quotaCacheMaxStaleMs: getDurationConfig(config?.resetAwareQuotaCacheMaxStaleMs, 0, 3_600_000), + }; +} + + + +export function resolveResetWindowConfig(config: Record | null | undefined) { + const rawWindows = Array.isArray(config?.resetWindowWindows) ? config.resetWindowWindows : null; + const windows = rawWindows + ?.filter((windowName): windowName is ResetWindowName => + (RESET_WINDOW_NAMES as readonly string[]).includes(String(windowName)) + ) + .filter((windowName, index, array) => array.indexOf(windowName) === index); + + const effectiveWindows = + windows && windows.length > 0 + ? windows + : config?.resetWindowIncludeSession === true + ? (["weekly", "session"] as ResetWindowName[]) + : (["weekly"] as ResetWindowName[]); + + return { + windows: effectiveWindows, + tieBandMs: Math.max( + 0, + finiteNumberOrNull(config?.resetWindowTieBandMs) ?? RESET_WINDOW_DEFAULT_TIE_BAND_MS + ), + quotaCacheTtlMs: getDurationConfig(config?.resetWindowQuotaCacheTtlMs, 0, 300_000), + quotaCacheMaxStaleMs: getDurationConfig(config?.resetWindowQuotaCacheMaxStaleMs, 0, 3_600_000), + }; +} + + + +export function resolveSlaRoutingPolicy( + config: Record | null | undefined +): SlaRoutingPolicy | undefined { + if (!config) return undefined; + const nestedSla = isRecord(config.sla) ? config.sla : {}; + const targetP95Ms = finiteNumberOrNull(config.slaTargetP95Ms ?? nestedSla.targetP95Ms); + const maxErrorRate = finiteNumberOrNull(config.slaMaxErrorRate ?? nestedSla.maxErrorRate); + const maxCostPer1MTokens = finiteNumberOrNull( + config.slaMaxCostPer1MTokens ?? nestedSla.maxCostPer1MTokens + ); + const hardConstraints = config.slaHardConstraints ?? nestedSla.hardConstraints; + + const policy: SlaRoutingPolicy = {}; + if (targetP95Ms !== null && targetP95Ms > 0) policy.targetP95Ms = targetP95Ms; + if (maxErrorRate !== null && maxErrorRate >= 0) policy.maxErrorRate = clamp01(maxErrorRate); + if (maxCostPer1MTokens !== null && maxCostPer1MTokens > 0) { + policy.maxCostPer1MTokens = maxCostPer1MTokens; + } + if (typeof hardConstraints === "boolean") policy.hardConstraints = hardConstraints; + + return Object.keys(policy).length > 0 ? policy : undefined; +} + + + +function getResetAwareProvider(target: ResolvedComboTarget): string | null { + const provider = (target.providerId || target.provider || "").toLowerCase(); + return provider || null; +} + + + +function normalizeResetAt(value: unknown): string | null { + if (typeof value === "string" && value.trim().length > 0) return value.trim(); + if (typeof value === "number" && Number.isFinite(value)) return String(value); + return null; +} + + + +function parseResetTimeMs(resetAt: string | null | undefined): number { + if (!resetAt) return NaN; + const resetTime = Date.parse(resetAt); + if (Number.isFinite(resetTime)) return resetTime; + + if (!/^\d+(?:\.\d+)?$/.test(resetAt)) return NaN; + const numericResetAt = Number(resetAt); + if (!Number.isFinite(numericResetAt)) return NaN; + return numericResetAt < 10_000_000_000 ? numericResetAt * 1000 : numericResetAt; +} + + + +function getQuotaWindow( + quota: unknown, + key: "window5h" | "window7d" | "windowWeekly" | "windowMonthly" +): { percentUsed: number | null; resetAt: string | null } | null { + if (!isRecord(quota)) return null; + const window = quota[key]; + if (!isRecord(window)) return null; + const percentUsed = finiteNumberOrNull(window.percentUsed); + const resetAt = normalizeResetAt(window.resetAt); + return { percentUsed, resetAt }; +} + + + +function normalizeWindowPercentUsed(value: unknown): number | null { + const numericValue = finiteNumberOrNull(value); + if (numericValue === null) return null; + if (numericValue > 1) return clamp01(numericValue / 100); + return clamp01(numericValue); +} + + + +function getNamedQuotaWindow( + quota: unknown, + windowName: ResetWindowName +): { percentUsed: number | null; resetAt: string | null } | null { + if (!quota || !isRecord(quota)) return null; + + if (windowName === "session") return getQuotaWindow(quota, "window5h"); + if (windowName === "weekly") { + return getQuotaWindow(quota, "window7d") || getQuotaWindow(quota, "windowWeekly"); + } + if (windowName === "monthly") return getQuotaWindow(quota, "windowMonthly"); + + return null; +} + + + +function getWindowsMapQuotaWindow( + quota: unknown, + windowName: ResetWindowName +): { percentUsed: number | null; resetAt: string | null } | null { + if (!quota || !isRecord(quota) || !isRecord(quota.windows)) return null; + const candidates = Object.entries(quota.windows) + .map(([key, value]) => ({ key: key.toLowerCase(), value })) + .filter(({ key }) => key === windowName || key.startsWith(`${windowName} `)); + + if (candidates.length === 0) return null; + candidates.sort((a, b) => a.key.localeCompare(b.key)); + const window = candidates[0].value; + if (!isRecord(window)) return null; + + return { + percentUsed: normalizeWindowPercentUsed(window.percentUsed), + resetAt: normalizeResetAt(window.resetAt), + }; +} + + + +function resolveQuotaWindowByName( + quota: unknown, + windowName: ResetWindowName +): { percentUsed: number | null; resetAt: string | null } | null { + return getNamedQuotaWindow(quota, windowName) || getWindowsMapQuotaWindow(quota, windowName); +} + + + +function getResetUrgency(resetAt: string | null | undefined, windowMs: number): number { + if (!resetAt) return 0.5; + const resetTime = parseResetTimeMs(resetAt); + if (!Number.isFinite(resetTime)) return 0.5; + const msUntilReset = resetTime - Date.now(); + if (msUntilReset <= 0) return 1; + return clamp01(1 - msUntilReset / windowMs); +} + + + +function scoreQuotaWindow( + remaining: number, + resetAt: string | null | undefined, + windowMs: number, + remainingWeight: number, + resetPressureWeight: number +): number { + const normalizedRemaining = clamp01(remaining); + const resetUrgency = getResetUrgency(resetAt, windowMs); + const resetPressure = resetUrgency * (1 - normalizedRemaining); + return remainingWeight * normalizedRemaining + resetPressureWeight * resetPressure; +} + + + +function scoreResetAwareQuota(quota: unknown, config: ReturnType) { + if (!quota || !isRecord(quota)) return { score: 0.5 }; + if (quota.limitReached === true) return { score: -Infinity }; + + const overallPercentUsed = clamp01(finiteNumberOrNull(quota.percentUsed) ?? 0.5); + const sessionWindow = getQuotaWindow(quota, "window5h"); + const weeklyWindow = getQuotaWindow(quota, "window7d") || getQuotaWindow(quota, "windowWeekly"); + const sessionRemaining = clamp01(1 - (sessionWindow?.percentUsed ?? overallPercentUsed)); + const weeklyRemaining = clamp01(1 - (weeklyWindow?.percentUsed ?? overallPercentUsed)); + const sessionScore = scoreQuotaWindow( + sessionRemaining, + sessionWindow?.resetAt, + RESET_AWARE_SESSION_WINDOW_MS, + RESET_AWARE_SESSION_REMAINING_WEIGHT, + RESET_AWARE_SESSION_RESET_PRESSURE_WEIGHT + ); + const weeklyScore = scoreQuotaWindow( + weeklyRemaining, + weeklyWindow?.resetAt ?? normalizeResetAt(quota.resetAt), + RESET_AWARE_WEEKLY_WINDOW_MS, + RESET_AWARE_WEEKLY_REMAINING_WEIGHT, + RESET_AWARE_WEEKLY_RESET_PRESSURE_WEIGHT + ); + let score = config.sessionWeight * sessionScore + config.weeklyWeight * weeklyScore; + + if (config.exhaustionGuard > 0 && sessionRemaining < config.exhaustionGuard) { + score *= Math.max(0.05, sessionRemaining / config.exhaustionGuard); + } + + return { score }; +} + + + +async function getQuotaAwareConnectionsForTarget( + target: ResolvedComboTarget, + connectionCache: Map>>, + connectionLoadPromises: Map>>>, + comboName: string, + log: { warn?: (...args: unknown[]) => void } +) { + const provider = getResetAwareProvider(target); + if (!provider || !getQuotaFetcher(provider)) return []; + if (!connectionCache.has(provider)) { + const cached = resetAwareConnectionCache.get(provider); + if (cached && Date.now() - cached.fetchedAt < RESET_AWARE_CONNECTION_CACHE_TTL_MS) { + connectionCache.set(provider, cached.connections); + return cached.connections; + } + + if (!connectionLoadPromises.has(provider)) { + connectionLoadPromises.set( + provider, + (async () => { + try { + const connections = await getProviderConnections({ provider, isActive: true }); + const activeConnections = Array.isArray(connections) + ? (connections as Array>) + : []; + if ( + !resetAwareConnectionCache.has(provider) && + resetAwareConnectionCache.size >= MAX_RESET_AWARE_CACHE + ) { + const oldest = resetAwareConnectionCache.keys().next().value; + if (oldest !== undefined) resetAwareConnectionCache.delete(oldest); + } + resetAwareConnectionCache.set(provider, { + connections: activeConnections, + fetchedAt: Date.now(), + }); + return activeConnections; + } catch (error) { + log.warn?.("COMBO", "Reset-aware failed to load quota-aware connections.", { + comboName, + err: error, + operation: "getProviderConnections", + provider, + }); + return []; + } + })() + ); + } + + const connections = await connectionLoadPromises.get(provider)!; + connectionCache.set(provider, connections); + } + return connectionCache.get(provider) || []; +} + + + +function normalizeConnectionIds(value: unknown): string[] | null { + if (!Array.isArray(value)) return null; + const ids = value.filter( + (connectionId): connectionId is string => + typeof connectionId === "string" && connectionId.trim().length > 0 + ); + return ids.length > 0 ? ids : null; +} + + + +function filterAllowedConnectionIds( + connectionIds: string[], + apiKeyAllowedConnectionIds: string[] | null | undefined +): string[] { + const allowedIds = normalizeConnectionIds(apiKeyAllowedConnectionIds); + if (!allowedIds) return connectionIds; + const allowedSet = new Set(allowedIds); + return connectionIds.filter((connectionId) => allowedSet.has(connectionId)); +} + + + +function getTargetConnectionIds( + target: ResolvedComboTarget, + connections: Array> +): string[] { + let connectionIds: string[]; + if (target.connectionId) { + return [target.connectionId]; + } + + if (Array.isArray(target.allowedConnectionIds) && target.allowedConnectionIds.length > 0) { + return target.allowedConnectionIds.filter( + (connectionId): connectionId is string => + typeof connectionId === "string" && connectionId.trim().length > 0 + ); + } + + connectionIds = connections + .map((connection) => (typeof connection.id === "string" ? connection.id : null)) + .filter((connectionId): connectionId is string => !!connectionId); + return connectionIds; +} + + + +async function mapWithConcurrency( + items: T[], + concurrency: number, + mapper: (item: T, index: number) => Promise +): Promise { + const results = new Array(items.length); + let nextIndex = 0; + const workerCount = Math.max(1, Math.min(concurrency, items.length)); + + await Promise.all( + Array.from({ length: workerCount }, async () => { + while (nextIndex < items.length) { + const currentIndex = nextIndex++; + results[currentIndex] = await mapper(items[currentIndex], currentIndex); + } + }) + ); + + return results; +} + + + +export async function fetchResetAwareQuotaWithCache({ + provider, + connectionId, + connection, + fetcher, + config, + log, + comboName, +}: { + provider: string; + connectionId: string; + connection?: Record; + fetcher: (connectionId: string, connection?: Record) => Promise; + config: QuotaFetchCacheConfig; + log: { debug?: (...args: unknown[]) => void; warn?: (...args: unknown[]) => void }; + comboName: string; +}): Promise { + const cacheKey = `${provider}:${connectionId}`; + const ttlMs = config.quotaCacheTtlMs; + const maxStaleMs = config.quotaCacheMaxStaleMs; + const now = Date.now(); + const cached = resetAwareQuotaCache.get(cacheKey); + + if (ttlMs <= 0 && maxStaleMs <= 0) { + try { + return await fetcher(connectionId, connection); + } catch (error) { + log.warn?.("COMBO", "Reset-aware quota fetch failed.", { + comboName, + connectionId, + err: error, + operation: "quotaFetch", + provider, + }); + return null; + } + } + + const refresh = () => { + const existing = resetAwareQuotaCache.get(cacheKey); + if (existing?.refreshPromise != null) return existing.refreshPromise; + + const refreshPromise = fetcher(connectionId, connection) + .then((quota) => { + if (quota) { + if ( + !resetAwareQuotaCache.has(cacheKey) && + resetAwareQuotaCache.size >= MAX_RESET_AWARE_CACHE + ) { + const oldest = resetAwareQuotaCache.keys().next().value; + if (oldest !== undefined) resetAwareQuotaCache.delete(oldest); + } + resetAwareQuotaCache.set(cacheKey, { + quota, + fetchedAt: Date.now(), + refreshPromise: null, + }); + } else { + resetAwareQuotaCache.delete(cacheKey); + } + return quota; + }) + .catch((error) => { + const previous = resetAwareQuotaCache.get(cacheKey); + if (previous) { + if ( + !resetAwareQuotaCache.has(cacheKey) && + resetAwareQuotaCache.size >= MAX_RESET_AWARE_CACHE + ) { + const oldest = resetAwareQuotaCache.keys().next().value; + if (oldest !== undefined) resetAwareQuotaCache.delete(oldest); + } + resetAwareQuotaCache.set(cacheKey, { ...previous, refreshPromise: null }); + } + log.warn?.("COMBO", "Reset-aware quota fetch failed.", { + comboName, + connectionId, + err: error, + operation: "quotaFetch", + provider, + }); + return null; + }); + + if (!resetAwareQuotaCache.has(cacheKey) && resetAwareQuotaCache.size >= MAX_RESET_AWARE_CACHE) { + const oldest = resetAwareQuotaCache.keys().next().value; + if (oldest !== undefined) resetAwareQuotaCache.delete(oldest); + } + resetAwareQuotaCache.set(cacheKey, { + quota: existing?.quota ?? cached?.quota ?? null, + fetchedAt: existing?.fetchedAt ?? cached?.fetchedAt ?? 0, + refreshPromise, + }); + return refreshPromise; + }; + + if (ttlMs > 0 && cached) { + const age = now - cached.fetchedAt; + if (age <= ttlMs) return cached.quota; + if (maxStaleMs > 0 && age <= ttlMs + maxStaleMs) { + void refresh(); + return cached.quota; + } + } + + return refresh(); +} + + + +export async function preScreenTargets( + targets: ResolvedComboTarget[], + isModelAvailable?: IsModelAvailable | null +): Promise> { + if (targets.length === 0) { + return new Map(); + } + + const results = await mapWithConcurrency( + targets, + PRE_SCREEN_CONCURRENCY, + async (target): Promise<{ key: string; result: PreScreenResult }> => { + const profile = await getRuntimeProviderProfile(target.provider).catch(() => null); + + const breaker = getCircuitBreaker(target.provider); + if (breaker.getStatus().state === "OPEN") { + return { key: target.executionKey, result: { profile, available: false } }; + } + + let available = true; + if (isModelAvailable) { + // IsModelAvailable may return a sync boolean or a Promise; Promise.resolve + // normalizes both so the .catch() never runs against a bare boolean. + available = await Promise.resolve(isModelAvailable(target.modelStr, target)).catch( + () => true + ); + } + return { key: target.executionKey, result: { profile, available } }; + } + ); + + const map = new Map(); + for (const { key, result } of results) { + map.set(key, result); + } + return map; +} + + + +export async function orderTargetsByResetAwareQuota( + targets: ResolvedComboTarget[], + comboName: string, + configSource: Record | null | undefined, + log: { warn?: (...args: unknown[]) => void }, + apiKeyAllowedConnectionIds?: string[] | null +) { + if (targets.length === 0) return targets; + + const config = resolveResetAwareConfig(configSource); + const connectionCache = new Map>>(); + const connectionLoadPromises = new Map>>>(); + const quotaPromises = new Map>(); + const connectionById = new Map>(); + const expandedTargets: ResolvedComboTarget[] = []; + + const targetsWithConnections = await Promise.all( + targets.map(async (target) => ({ + connections: await getQuotaAwareConnectionsForTarget( + target, + connectionCache, + connectionLoadPromises, + comboName, + log + ), + target, + })) + ); + + for (const { target, connections } of targetsWithConnections) { + for (const connection of connections) { + if (typeof connection.id === "string") connectionById.set(connection.id, connection); + } + + const unrestrictedConnectionIds = getTargetConnectionIds(target, connections); + const connectionIds = filterAllowedConnectionIds( + unrestrictedConnectionIds, + apiKeyAllowedConnectionIds + ); + if (connectionIds.length === 0) { + if ( + unrestrictedConnectionIds.length > 0 && + normalizeConnectionIds(apiKeyAllowedConnectionIds) + ) { + continue; + } + expandedTargets.push(target); + continue; + } + + for (const connectionId of connectionIds) { + expandedTargets.push({ + ...target, + connectionId, + executionKey: + target.connectionId === connectionId + ? target.executionKey + : `${target.executionKey}@${connectionId}`, + }); + } + } + + const scoredTargets = await mapWithConcurrency( + expandedTargets, + RESET_AWARE_QUOTA_FETCH_CONCURRENCY, + async (target, index) => { + let quota: unknown = null; + const provider = getResetAwareProvider(target); + const fetcher = provider ? getQuotaFetcher(provider) : null; + if (fetcher && provider && target.connectionId) { + const quotaKey = `${provider}:${target.connectionId}`; + if (!quotaPromises.has(quotaKey)) { + quotaPromises.set( + quotaKey, + fetchResetAwareQuotaWithCache({ + provider, + connectionId: target.connectionId, + connection: connectionById.get(target.connectionId), + fetcher, + config, + log, + comboName, + }) + ); + } + quota = await quotaPromises.get(quotaKey)!; + } + const { score } = scoreResetAwareQuota(quota, config); + return { target, score, index }; + } + ); + + scoredTargets.sort((a, b) => { + if (b.score !== a.score) return b.score - a.score; + return a.index - b.index; + }); + + const bestScore = scoredTargets[0]?.score ?? 0; + const tiedTargets = scoredTargets.filter((entry) => bestScore - entry.score <= config.tieBand); + let orderedTiedTargets = tiedTargets; + if (tiedTargets.length > 1) { + const key = `reset-aware:${comboName}`; + const counter = rrCounters.get(key) || 0; + if (!rrCounters.has(key) && rrCounters.size >= MAX_RR_COUNTERS) { + const oldest = rrCounters.keys().next().value; + if (oldest !== undefined) rrCounters.delete(oldest); + } + rrCounters.set(key, counter + 1); + const startIndex = counter % tiedTargets.length; + orderedTiedTargets = [...tiedTargets.slice(startIndex), ...tiedTargets.slice(0, startIndex)]; + } + + const tiedExecutionKeys = new Set(orderedTiedTargets.map((entry) => entry.target.executionKey)); + return [ + ...orderedTiedTargets, + ...scoredTargets.filter((entry) => !tiedExecutionKeys.has(entry.target.executionKey)), + ].map((entry) => entry.target); +} + + + +function getResetWindowTimestampMs(quota: unknown, windows: ResetWindowName[]): number { + if (!quota || !isRecord(quota) || quota.limitReached === true) return Infinity; + + let selectedResetMs = Infinity; + for (const windowName of windows) { + const window = resolveQuotaWindowByName(quota, windowName); + const resetMs = parseResetTimeMs(window?.resetAt ?? null); + if (Number.isFinite(resetMs)) { + selectedResetMs = Math.min(selectedResetMs, resetMs); + } + } + + if (!Number.isFinite(selectedResetMs)) { + selectedResetMs = parseResetTimeMs(normalizeResetAt(quota.resetAt)); + } + + return Number.isFinite(selectedResetMs) ? selectedResetMs : Infinity; +} + + + +function getResetWindowHorizonMs(windows: ResetWindowName[]): number { + if (windows.includes("monthly")) return 30 * 24 * 60 * 60 * 1000; + if (windows.includes("weekly")) return RESET_AWARE_WEEKLY_WINDOW_MS; + return RESET_AWARE_SESSION_WINDOW_MS; +} + + + +export function calculateResetWindowAffinity(quota: unknown, config: ResetWindowConfig): number { + const resetMs = getResetWindowTimestampMs(quota, config.windows); + if (!Number.isFinite(resetMs)) return 0.5; + + const msUntilReset = resetMs - Date.now(); + if (msUntilReset <= 0) return 1; + return clamp01(1 - msUntilReset / getResetWindowHorizonMs(config.windows)); +} + + + +export async function orderTargetsByResetWindow( + targets: ResolvedComboTarget[], + comboName: string, + configSource: Record | null | undefined, + log: { warn?: (...args: unknown[]) => void }, + apiKeyAllowedConnectionIds?: string[] | null +) { + if (targets.length === 0) return targets; + + const config = resolveResetWindowConfig(configSource); + const connectionCache = new Map>>(); + const connectionLoadPromises = new Map>>>(); + const quotaPromises = new Map>(); + const connectionById = new Map>(); + const expandedTargets: ResolvedComboTarget[] = []; + + const targetsWithConnections = await Promise.all( + targets.map(async (target) => ({ + connections: await getQuotaAwareConnectionsForTarget( + target, + connectionCache, + connectionLoadPromises, + comboName, + log + ), + target, + })) + ); + + for (const { target, connections } of targetsWithConnections) { + for (const connection of connections) { + if (typeof connection.id === "string") connectionById.set(connection.id, connection); + } + + const unrestrictedConnectionIds = getTargetConnectionIds(target, connections); + const connectionIds = filterAllowedConnectionIds( + unrestrictedConnectionIds, + apiKeyAllowedConnectionIds + ); + if (connectionIds.length === 0) { + if ( + unrestrictedConnectionIds.length > 0 && + normalizeConnectionIds(apiKeyAllowedConnectionIds) + ) { + continue; + } + expandedTargets.push(target); + continue; + } + + for (const connectionId of connectionIds) { + expandedTargets.push({ + ...target, + connectionId, + executionKey: + target.connectionId === connectionId + ? target.executionKey + : `${target.executionKey}@${connectionId}`, + }); + } + } + + const scoredTargets = await mapWithConcurrency( + expandedTargets, + RESET_AWARE_QUOTA_FETCH_CONCURRENCY, + async (target, index) => { + let quota: unknown = null; + const provider = getResetAwareProvider(target); + const fetcher = provider ? getQuotaFetcher(provider) : null; + if (fetcher && provider && target.connectionId) { + const quotaKey = `${provider}:${target.connectionId}`; + if (!quotaPromises.has(quotaKey)) { + quotaPromises.set( + quotaKey, + fetchResetAwareQuotaWithCache({ + provider, + connectionId: target.connectionId, + connection: connectionById.get(target.connectionId), + fetcher, + config, + log, + comboName, + }) + ); + } + quota = await quotaPromises.get(quotaKey)!; + } + + return { + target, + resetMs: getResetWindowTimestampMs(quota, config.windows), + index, + }; + } + ); + + scoredTargets.sort((a, b) => { + if (a.resetMs !== b.resetMs) return a.resetMs - b.resetMs; + return a.index - b.index; + }); + + const bestResetMs = scoredTargets[0]?.resetMs ?? Infinity; + if (!Number.isFinite(bestResetMs) || config.tieBandMs <= 0) { + return scoredTargets.map((entry) => entry.target); + } + + const tiedTargets = scoredTargets.filter( + (entry) => entry.resetMs - bestResetMs <= config.tieBandMs + ); + if (tiedTargets.length <= 1) return scoredTargets.map((entry) => entry.target); + + const key = `reset-window:${comboName}`; + const counter = rrCounters.get(key) || 0; + if (!rrCounters.has(key) && rrCounters.size >= MAX_RR_COUNTERS) { + const oldest = rrCounters.keys().next().value; + if (oldest !== undefined) rrCounters.delete(oldest); + } + rrCounters.set(key, counter + 1); + const startIndex = counter % tiedTargets.length; + const orderedTiedTargets = [ + ...tiedTargets.slice(startIndex), + ...tiedTargets.slice(0, startIndex), + ]; + const tiedExecutionKeys = new Set(orderedTiedTargets.map((entry) => entry.target.executionKey)); + + return [ + ...orderedTiedTargets, + ...scoredTargets.filter((entry) => !tiedExecutionKeys.has(entry.target.executionKey)), + ].map((entry) => entry.target); +} + diff --git a/open-sse/services/combo/roundRobin.ts b/open-sse/services/combo/roundRobin.ts new file mode 100644 index 00000000000..640367137c4 --- /dev/null +++ b/open-sse/services/combo/roundRobin.ts @@ -0,0 +1,624 @@ +/** + * Shared combo (model combo) handling with fallback support + * Supports: priority, weighted, round-robin, random, least-used, cost-optimized, + * reset-aware, reset-window, strict-random, auto, fill-first, p2c, lkgp, + * context-optimized, and context-relay strategies + */ + +import { + checkFallbackError, + classifyErrorText, + formatRetryAfter, + getRuntimeProviderProfile, + recordProviderFailure, + isProviderFailureCode, + isProviderExhaustedReason, + type ProviderProfile, +} from "../accountFallback.ts"; + +import { FETCH_TIMEOUT_MS, RateLimitReason } from "../../config/constants.ts"; + +import { errorResponse, unavailableResponse } from "../../utils/error.ts"; + +import { clamp01 } from "../../utils/number.ts"; + +import { + recordComboIntent, + recordComboRequest, + recordComboShadowRequest, + getComboMetrics, +} from "../comboMetrics.ts"; + +import { + resolveComboConfig, + getDefaultComboConfig, + resolveComboTargetTimeoutMs, + PRE_SCREEN_CONCURRENCY, +} from "../comboConfig.ts"; + +import { + recordSessionModelUsage, + getLastSessionModel, + getHandoff, +} from "../../../src/lib/db/contextHandoffs.ts"; + +import * as semaphore from "../rateLimitSemaphore.ts"; + +import { getCircuitBreaker } from "../../../src/shared/utils/circuitBreaker"; + +import { fisherYatesShuffle, getNextFromDeck } from "../../../src/shared/utils/shuffleDeck"; + +import { parseModel } from "../model.ts"; + +import { emit } from "../../../src/lib/events/eventBus"; + +import { notifyWebhookEvent } from "../../../src/lib/webhookDispatcher"; + +import { + getResolvedModelCapabilities, + supportsReasoning, + supportsToolCalling, +} from "../modelCapabilities.ts"; + +import { getReasoningTokens } from "../../../src/lib/usage/tokenAccounting.ts"; + +import { orderTargetsByEvalScores } from "../evalRouting.ts"; + +import { getModelContextLimit } from "../../../src/lib/modelCapabilities"; + +import { getProviderConnections } from "../../../src/lib/db/providers"; + +import { getProviderModels } from "../../config/providerModels.ts"; + +import { + getComboModelString, + getComboStepTarget, + getComboStepWeight, + normalizeComboStep, +} from "../../../src/lib/combos/steps.ts"; + +import { + getConnectionRoutingTags, + matchesRoutingTags, + resolveRequestRoutingTags, + type RoutingTagMatchMode, +} from "../../../src/domain/tagRouter.ts"; + +import { normalizeRoutingStrategy } from "../../../src/shared/constants/routingStrategies.ts"; + +import { + isProviderInCooldown, + recordProviderCooldown, + recordProviderSuccess, +} from "../providerCooldownTracker.ts"; + +import { + resolveResilienceSettings, + type ResilienceSettings, +} from "../../../src/lib/resilience/settings"; + +import { HandleRoundRobinOptions, ComboRetryAfter, ComboErrorBody } from "./types.ts"; +import { resolveDelayMs, comboModelNotFoundResponse, MAX_GLOBAL_ATTEMPTS, isAllAccountsRateLimitedResponse, TRANSIENT_FOR_SEMAPHORE, MAX_FALLBACK_WAIT_MS } from "./constants.ts"; +import { resolveComboTargets, applyRequestTagRouting } from "./auto.ts"; +import { filterTargetsByRequestCompatibility } from "./context.ts"; +import { scheduleShadowRouting, resolveShadowTargets } from "./shadow.ts"; +import { rrCounters, MAX_RR_COUNTERS } from "./state.ts"; +import { isRecord, validateResponseQuality, toRecordedTarget, isStreamReadinessFailureErrorBody, isTokenLimitBreachErrorBody, toRetryAfterDisplayValue } from "./utils.ts"; + + + +/** + * Handle round-robin combo: each request goes to the next model in circular order. + * Uses semaphore-based concurrency control with queue + rate-limit awareness. + * + * Flow: + * 1. Pick target model via atomic counter (counter % models.length) + * 2. Acquire semaphore slot (may queue if at max concurrency) + * 3. Send request to target model + * 4. On 429 → mark model rate-limited, try next model in rotation + * 5. On semaphore timeout → fallback to next available model + */ +export async function handleRoundRobinCombo({ + body, + combo, + handleSingleModel, + isModelAvailable, + log, + settings, + allCombos, + signal, +}: HandleRoundRobinOptions): Promise { + const config = settings + ? resolveComboConfig(combo, settings) + : { ...getDefaultComboConfig(), ...(combo.config || {}) }; + const concurrency = config.concurrencyPerModel ?? 3; + const queueTimeout = config.queueTimeoutMs ?? 30000; + const maxRetries = config.maxRetries ?? 1; + const retryDelayMs = resolveDelayMs(config.retryDelayMs, 2000); + const fallbackDelayMs = resolveDelayMs(config.fallbackDelayMs, 0); + + const resilienceSettings: ResilienceSettings = settings + ? resolveResilienceSettings(settings) + : resolveResilienceSettings(null); + + const orderedTargets = resolveComboTargets(combo, allCombos); + const tagFilteredTargets = await applyRequestTagRouting(orderedTargets, body, log); + const evalRankedTargets = orderTargetsByEvalScores(tagFilteredTargets, config.evalRouting, log); + const filteredTargets = filterTargetsByRequestCompatibility( + evalRankedTargets, + body, + log, + "Context-aware round-robin fallback" + ); + const modelCount = filteredTargets.length; + if (modelCount === 0) { + return comboModelNotFoundResponse("Round-robin combo has no executable targets"); + } + + scheduleShadowRouting( + combo, + config, + body, + resolveShadowTargets(combo, config, allCombos), + handleSingleModel, + isModelAvailable, + "round-robin", + log + ); + + // Get and increment atomic counter + const counter = rrCounters.get(combo.name) || 0; + if (!rrCounters.has(combo.name) && rrCounters.size >= MAX_RR_COUNTERS) { + const oldest = rrCounters.keys().next().value; + if (oldest !== undefined) rrCounters.delete(oldest); + } + rrCounters.set(combo.name, counter + 1); + const startIndex = counter % modelCount; + + const clientRequestedStream = body?.stream === true; + const startTime = Date.now(); + let lastError: string | null = null; + let lastStatus: number | null = null; + let earliestRetryAfter: ComboRetryAfter | null = null; + let globalAttempts = 0; + let fallbackCount = 0; + let recordedAttempts = 0; + + // #1731: Per-request in-memory set of providers whose quota is fully exhausted. + // When a target returns a quota-exhausted 429, remaining targets from the same + // provider are skipped to avoid the cascade through N same-provider targets. + const exhaustedProviders = new Set(); + const transientRateLimitedProviders = new Set(); + + // Try each model starting from the round-robin target + for (let offset = 0; offset < modelCount; offset++) { + const modelIndex = (startIndex + offset) % modelCount; + const target = filteredTargets[modelIndex]; + const modelStr = target.modelStr; + const provider = target.provider; + const profile = await getRuntimeProviderProfile(provider); + const semaphoreKey = `combo:${combo.name}:${target.executionKey}`; + const allowRateLimitedConnection = + Boolean(provider && provider !== "unknown") && transientRateLimitedProviders.has(provider); + const targetForAttempt = allowRateLimitedConnection + ? { ...target, allowRateLimitedConnection: true } + : target; + + // Pre-check availability + if (isModelAvailable) { + const available = await isModelAvailable(modelStr, targetForAttempt); + if (!available) { + log.debug?.( + "COMBO-RR", + `Skipping ${modelStr} — no credentials available or model excluded` + ); + if (offset > 0) fallbackCount++; + continue; + } + } + + if ( + resilienceSettings.providerCooldown.enabled && + Boolean(provider && provider !== "unknown") && + isProviderInCooldown(provider, target.connectionId as string | undefined, resilienceSettings) + ) { + log.info("COMBO-RR", `Skipping ${modelStr} — provider ${provider} in global cooldown`); + if (offset > 0) fallbackCount++; + continue; + } + + // #1731: Skip targets from a provider that already signaled full quota exhaustion + // this request. + if (provider && exhaustedProviders.has(provider)) { + log.info( + "COMBO-RR", + `Skipping ${modelStr} — provider ${provider} marked exhausted this request (#1731)` + ); + if (offset > 0) fallbackCount++; + continue; + } + + // Acquire semaphore slot (may wait in queue) + let release: () => void; + try { + release = await semaphore.acquire(semaphoreKey, { + maxConcurrency: concurrency, + timeoutMs: queueTimeout, + }); + } catch (err) { + const errCode = isRecord(err) && typeof err.code === "string" ? err.code : null; + if (errCode === "SEMAPHORE_TIMEOUT" || errCode === "SEMAPHORE_QUEUE_FULL") { + log.warn( + "COMBO-RR", + `Semaphore ${errCode === "SEMAPHORE_QUEUE_FULL" ? "queue full" : "timeout"} for ${modelStr}, trying next model` + ); + if (offset > 0) fallbackCount++; + continue; + } + throw err; + } + + // Retry loop within this model + try { + for (let retry = 0; retry <= maxRetries; retry++) { + globalAttempts++; + if (globalAttempts > MAX_GLOBAL_ATTEMPTS) { + log.warn( + "COMBO-RR", + `Maximum combo attempts (${MAX_GLOBAL_ATTEMPTS}) exceeded. Terminating loop to prevent runaway requests.` + ); + return errorResponse(503, "Maximum combo retry limit reached"); + } + if (retry > 0) { + log.info( + "COMBO-RR", + `Retrying ${modelStr} in ${retryDelayMs}ms (attempt ${retry + 1}/${maxRetries + 1})` + ); + await new Promise((r) => setTimeout(r, retryDelayMs)); + } + + log.info( + "COMBO-RR", + `[RR #${counter}] → ${modelStr}${offset > 0 ? ` (fallback +${offset})` : ""}${retry > 0 ? ` (retry ${retry})` : ""}` + ); + + // Issue #3587: Reasoning models consume ALL max_tokens for reasoning_tokens. + // Add buffer to ensure reasoning + content both fit. Apply the buffer to a + // per-attempt COPY — never mutate the shared `body` — so it does not compound + // across round-robin iterations/retries (otherwise 4096 -> 6144 -> 9216 -> ... + // as each reasoning model re-reads an already-buffered value and overshoots the + // model's real limit, triggering 400s). + let attemptBody = body; + if (supportsReasoning(modelStr)) { + const currentMaxTokens = Number((body as Record).max_tokens) || 0; + if (currentMaxTokens > 0) { + const bufferedMaxTokens = Math.max( + currentMaxTokens + 1000, + Math.ceil(currentMaxTokens * 1.5) + ); + attemptBody = { + ...(body as Record), + max_tokens: bufferedMaxTokens, + } as typeof body; + log.info( + "COMBO-RR", + `Reasoning model ${modelStr}: buffered max_tokens ${currentMaxTokens} -> ${bufferedMaxTokens}` + ); + } + } + + const result = await handleSingleModel(attemptBody, modelStr, { + ...targetForAttempt, + failoverBeforeRetry: config.failoverBeforeRetry, + }); + + // Success — validate response quality before returning + if (result.ok) { + const quality = await validateResponseQuality(result, clientRequestedStream, log); + if (!quality.valid) { + log.warn( + "COMBO-RR", + `${modelStr} returned 200 but failed quality check: ${quality.reason}` + ); + recordComboRequest(combo.name, modelStr, { + success: false, + latencyMs: Date.now() - startTime, + fallbackCount, + strategy: "round-robin", + target: toRecordedTarget(target), + }); + recordedAttempts++; + // Fix #1707: Set terminal state so the fallback doesn't emit + // misleading ALL_ACCOUNTS_INACTIVE when the real issue is quality. + lastError = `Upstream response failed quality validation: ${quality.reason}`; + if (!lastStatus) lastStatus = 502; + if (offset > 0) fallbackCount++; + break; // move to next model + } + const latencyMs = Date.now() - startTime; + log.info( + "COMBO-RR", + `${modelStr} succeeded (${latencyMs}ms, ${fallbackCount} fallbacks)` + ); + recordComboRequest(combo.name, modelStr, { + success: true, + latencyMs, + fallbackCount, + strategy: "round-robin", + target: toRecordedTarget(target), + }); + recordedAttempts++; + + if (provider && provider !== "unknown") { + recordProviderSuccess(provider, target.connectionId ?? undefined); + } + + if (provider) { + const connId = target.connectionId || undefined; + void (async () => { + try { + const { setLKGP } = await import("../../../src/lib/localDb"); + await Promise.all([ + setLKGP(combo.name, target.executionKey, provider, connId), + setLKGP(combo.name, combo.id || combo.name, provider, connId), + ]); + } catch (err) { + log.warn( + "COMBO-RR", + "Failed to record Last Known Good Provider. This is non-fatal.", + { + err, + } + ); + } + })(); + } + return result; + } + + // Extract error info + let errorText = result.statusText || ""; + let retryAfter: ComboRetryAfter | null = null; + let errorBody: ComboErrorBody = null; + try { + const cloned = result.clone(); + try { + const text = await cloned.text(); + if (text) { + errorText = text.substring(0, 500); + errorBody = JSON.parse(text); + const parsedError = errorBody?.error; + errorText = + (typeof parsedError === "object" && parsedError?.message) || + (typeof parsedError === "string" ? parsedError : null) || + errorBody?.message || + errorText; + retryAfter = errorBody?.retryAfter || null; + } + } catch { + /* Clone parse failed */ + } + } catch { + /* Clone failed */ + } + + if (result.status === 499) { + log.info( + "COMBO-RR", + `Client disconnected (499) during ${modelStr} — stopping combo loop` + ); + recordComboRequest(combo.name, modelStr, { + success: false, + latencyMs: Date.now() - startTime, + fallbackCount, + strategy: "round-robin", + target: toRecordedTarget(target), + }); + recordedAttempts++; + return result; + } + + if ( + retryAfter && + (!earliestRetryAfter || new Date(retryAfter) < new Date(earliestRetryAfter)) + ) { + earliestRetryAfter = retryAfter; + } + + if (typeof errorText !== "string") { + try { + errorText = JSON.stringify(errorText); + } catch { + errorText = String(errorText); + } + } + + const isStreamReadinessFailure = + (result.status === 502 || result.status === 504) && + isStreamReadinessFailureErrorBody(errorBody); + + // FIX 5: a local per-API-key token-limit 429 must not cool shared accounts. + const isTokenLimitBreach = result.status === 429 && isTokenLimitBreachErrorBody(errorBody); + + // Round-robin uses the same target-level fallback rule as other combo + // strategies: non-ok target responses fall through to the next target. + // Classification stays here only to support cooldown/semaphore pacing, + // not to decide whether fallback is allowed. + const rawError = errorBody?.error; + const structuredError = + rawError && typeof rawError === "object" + ? { + // Upstream JSON may carry a numeric `code`/`type` (e.g. {"code":40001}). + // Coerce to string if present instead of discarding, so downstream string + // ops (.toLowerCase, .startsWith) can run safely without type crashes. + code: + (rawError as Record).code !== undefined && + (rawError as Record).code !== null + ? String((rawError as Record).code) + : undefined, + type: + (rawError as Record).type !== undefined && + (rawError as Record).type !== null + ? String((rawError as Record).type) + : undefined, + } + : undefined; + const fallbackResult = checkFallbackError( + result.status, + errorText, + 0, + null, + provider, + result.headers, + profile, + structuredError + ); + const { cooldownMs } = fallbackResult; + + const isAllAccountsRateLimited = isAllAccountsRateLimitedResponse( + result.status, + result.headers?.get("content-type") ?? null, + errorText + ); + + // #1731: If the entire provider quota is exhausted, mark it so subsequent + // same-provider targets are skipped immediately. API-key 429s still use + // the short resilience cooldown, but explicit quota text should stop the + // combo from trying another target for the same provider in this request. + const providerExhausted = + Boolean(provider && provider !== "unknown") && + (isProviderExhaustedReason(fallbackResult) || + classifyErrorText(errorText) === RateLimitReason.QUOTA_EXHAUSTED || + isAllAccountsRateLimited); + if (providerExhausted) { + exhaustedProviders.add(provider); + log.debug?.( + "COMBO-RR", + `Provider ${provider} quota exhausted — marking for skip (#1731)` + ); + } else if ( + result.status === 429 && + !isTokenLimitBreach && + provider && + provider !== "unknown" + ) { + transientRateLimitedProviders.add(provider); + } + + // Transient errors → mark in semaphore so round-robin stops stampeding this target. + if ( + !isStreamReadinessFailure && + !isTokenLimitBreach && + TRANSIENT_FOR_SEMAPHORE.includes(result.status) && + cooldownMs > 0 + ) { + semaphore.markRateLimited(semaphoreKey, cooldownMs); + log.warn("COMBO-RR", `${modelStr} error ${result.status}, cooldown ${cooldownMs}ms`); + } + + if (isAllAccountsRateLimited) { + log.info( + "COMBO-RR", + `All accounts rate-limited for ${modelStr}, falling back to next model` + ); + } + + // Transient error → retry same model. + // A token-limit 429 is terminal for the client — never retry it. + const isTransient = + !isStreamReadinessFailure && + !isTokenLimitBreach && + [408, 429, 500, 502, 503, 504].includes(result.status); + if (retry < maxRetries && isTransient && !providerExhausted) { + continue; + } + + // Done with this model + recordComboRequest(combo.name, modelStr, { + success: false, + latencyMs: Date.now() - startTime, + fallbackCount, + strategy: "round-robin", + target: toRecordedTarget(target), + }); + recordedAttempts++; + lastError = errorText || String(result.status); + if (!lastStatus) lastStatus = result.status; + if (offset > 0) fallbackCount++; + log.warn("COMBO-RR", `${modelStr} failed, trying next model`, { status: result.status }); + + if (resilienceSettings.providerCooldown.enabled && provider && provider !== "unknown") { + recordProviderCooldown(provider, target.connectionId ?? undefined, resilienceSettings); + } + + const fallbackWaitMs = + fallbackDelayMs > 0 && cooldownMs > 0 && cooldownMs <= MAX_FALLBACK_WAIT_MS + ? Math.min(cooldownMs, fallbackDelayMs) + : 0; + if ([502, 503, 504].includes(result.status) && fallbackWaitMs > 0) { + log.debug?.("COMBO-RR", `Waiting ${fallbackWaitMs}ms before fallback to next model`); + await new Promise((resolve) => { + const timer = setTimeout(resolve, fallbackWaitMs); + signal?.addEventListener( + "abort", + () => { + clearTimeout(timer); + resolve(undefined); + }, + { once: true } + ); + }); + if (signal?.aborted) { + log.info("COMBO-RR", `Client disconnected during fallback wait — aborting`); + return errorResponse(499, "Client disconnected"); + } + } + + break; + } + } finally { + // ALWAYS release semaphore slot + release(); + } + } + + // All models exhausted + const latencyMs = Date.now() - startTime; + if (recordedAttempts === 0) { + recordComboRequest(combo.name, null, { + success: false, + latencyMs, + fallbackCount, + strategy: "round-robin", + }); + } + + if (!lastStatus) { + return new Response( + JSON.stringify({ + error: { + message: "Service temporarily unavailable: all upstream accounts are inactive", + type: "service_unavailable", + code: "ALL_ACCOUNTS_INACTIVE", + }, + }), + { status: 503, headers: { "Content-Type": "application/json" } } + ); + } + + const status = lastStatus; + const msg = lastError || "All round-robin combo models unavailable"; + + if (earliestRetryAfter) { + const retryHuman = formatRetryAfter(toRetryAfterDisplayValue(earliestRetryAfter)); + log.warn("COMBO-RR", `All models failed | ${msg} (${retryHuman})`); + return unavailableResponse(status, msg, earliestRetryAfter, retryHuman); + } + + log.warn("COMBO-RR", `All models failed | ${msg}`); + return new Response(JSON.stringify({ error: { message: msg } }), { + status, + headers: { "Content-Type": "application/json" }, + }); +} + diff --git a/open-sse/services/combo/shadow.ts b/open-sse/services/combo/shadow.ts new file mode 100644 index 00000000000..109cea4fc4a --- /dev/null +++ b/open-sse/services/combo/shadow.ts @@ -0,0 +1,201 @@ +/** + * Shared combo (model combo) handling with fallback support + * Supports: priority, weighted, round-robin, random, least-used, cost-optimized, + * reset-aware, reset-window, strict-random, auto, fill-first, p2c, lkgp, + * context-optimized, and context-relay strategies + */ + +import { + checkFallbackError, + classifyErrorText, + formatRetryAfter, + getRuntimeProviderProfile, + recordProviderFailure, + isProviderFailureCode, + isProviderExhaustedReason, + type ProviderProfile, +} from "../accountFallback.ts"; + +import { FETCH_TIMEOUT_MS, RateLimitReason } from "../../config/constants.ts"; + +import { errorResponse, unavailableResponse } from "../../utils/error.ts"; + +import { clamp01 } from "../../utils/number.ts"; + +import { + recordComboIntent, + recordComboRequest, + recordComboShadowRequest, + getComboMetrics, +} from "../comboMetrics.ts"; + +import { parseModel } from "../model.ts"; + +import { getProviderModels } from "../../config/providerModels.ts"; + +import { ShadowRoutingConfig, ComboLike, ComboCollectionLike, ResolvedComboTarget, HandleSingleModel, IsModelAvailable, ComboLogger } from "./types.ts"; +import { isRecord, toRecordedTarget } from "./utils.ts"; +import { resolveNestedComboTargets } from "./dag.ts"; + + + +function normalizeShadowRoutingConfig(config: Record): ShadowRoutingConfig { + const raw = isRecord(config.shadowRouting) ? config.shadowRouting : {}; + const sampleRate = Number(raw.sampleRate ?? 1); + const maxTargets = Number(raw.maxTargets ?? 2); + const timeoutMs = Number(raw.timeoutMs ?? 30000); + return { + enabled: raw.enabled === true, + targets: Array.isArray(raw.targets) ? raw.targets : [], + sampleRate: Number.isFinite(sampleRate) ? Math.max(0, Math.min(1, sampleRate)) : 1, + maxTargets: Number.isFinite(maxTargets) ? Math.max(1, Math.min(10, Math.floor(maxTargets))) : 2, + timeoutMs: Number.isFinite(timeoutMs) + ? Math.max(1000, Math.min(120000, Math.floor(timeoutMs))) + : 30000, + }; +} + + + +export function resolveShadowTargets( + combo: ComboLike, + config: Record, + allCombos: ComboCollectionLike +): ResolvedComboTarget[] { + const shadowConfig = normalizeShadowRoutingConfig(config); + if (!shadowConfig.enabled || shadowConfig.targets.length === 0) return []; + if (shadowConfig.sampleRate <= 0 || Math.random() > shadowConfig.sampleRate) return []; + + const shadowCombo: ComboLike = { + ...combo, + name: `${combo.name}:shadow`, + models: shadowConfig.targets, + }; + return resolveNestedComboTargets(shadowCombo, allCombos, new Set([combo.name]), 0, ["shadow"]) + .slice(0, shadowConfig.maxTargets) + .map((target) => ({ + ...target, + trafficType: "shadow" as const, + })); +} + + + +async function drainShadowResponse(response: Response): Promise { + try { + if (!response.body) return; + await response.arrayBuffer(); + } catch { + // Shadow draining is best-effort and must never affect the production response. + } +} + + + +function withTimeout(promise: Promise, timeoutMs: number): Promise { + return new Promise((resolve, reject) => { + const timer = setTimeout(() => reject(new Error("Shadow route timed out")), timeoutMs); + promise.then( + (value) => { + clearTimeout(timer); + resolve(value); + }, + (error) => { + clearTimeout(timer); + reject(error); + } + ); + }); +} + + + +function cloneRequestBodyForShadowRouting(body: Record): Record { + if (typeof structuredClone === "function") { + return structuredClone(body) as Record; + } + + return JSON.parse(JSON.stringify(body)) as Record; +} + + + +export function scheduleShadowRouting( + combo: ComboLike, + config: Record, + body: Record, + targets: ResolvedComboTarget[], + handleSingleModel: HandleSingleModel, + isModelAvailable: IsModelAvailable | undefined, + strategy: string, + log: ComboLogger +): void { + if (targets.length === 0) return; + const shadowConfig = normalizeShadowRoutingConfig(config); + let shadowBaseBody: Record; + try { + shadowBaseBody = cloneRequestBodyForShadowRouting(body); + } catch (error) { + log.warn("COMBO", "Shadow routing skipped: failed to clone request body", { + error: error instanceof Error ? error.message : String(error), + }); + return; + } + const run = async () => { + await Promise.all( + targets.map(async (target) => { + const startedAt = Date.now(); + try { + const shadowBody = { + ...cloneRequestBodyForShadowRouting(shadowBaseBody), + model: target.modelStr, + stream: false, + }; + if (isModelAvailable) { + const available = await isModelAvailable(target.modelStr, target); + if (!available) { + recordComboShadowRequest(combo.name, target.modelStr, { + success: false, + latencyMs: Date.now() - startedAt, + target: toRecordedTarget(target), + }); + log.info("COMBO", `Shadow target skipped (unavailable): ${target.modelStr}`); + return; + } + } + + const response = await withTimeout( + handleSingleModel(shadowBody, target.modelStr, { + ...target, + failoverBeforeRetry: true, + trafficType: "shadow", + }), + shadowConfig.timeoutMs + ); + await drainShadowResponse(response.clone()); + recordComboShadowRequest(combo.name, target.modelStr, { + success: response.ok, + latencyMs: Date.now() - startedAt, + target: toRecordedTarget(target), + }); + log.info( + "COMBO", + `Shadow target ${target.modelStr} completed with status ${response.status} (${strategy})` + ); + } catch (error) { + recordComboShadowRequest(combo.name, target.modelStr, { + success: false, + latencyMs: Date.now() - startedAt, + target: toRecordedTarget(target), + }); + log.warn("COMBO", `Shadow target ${target.modelStr} failed`, { + error: error instanceof Error ? error.message : String(error), + }); + } + }) + ); + }; + + setTimeout(() => void run(), 0); +} + diff --git a/open-sse/services/combo/sorting.ts b/open-sse/services/combo/sorting.ts new file mode 100644 index 00000000000..b544627dd7e --- /dev/null +++ b/open-sse/services/combo/sorting.ts @@ -0,0 +1,361 @@ +/** + * Shared combo (model combo) handling with fallback support + * Supports: priority, weighted, round-robin, random, least-used, cost-optimized, + * reset-aware, reset-window, strict-random, auto, fill-first, p2c, lkgp, + * context-optimized, and context-relay strategies + */ + +import { + checkFallbackError, + classifyErrorText, + formatRetryAfter, + getRuntimeProviderProfile, + recordProviderFailure, + isProviderFailureCode, + isProviderExhaustedReason, + type ProviderProfile, +} from "../accountFallback.ts"; + +import { FETCH_TIMEOUT_MS, RateLimitReason } from "../../config/constants.ts"; + +import { errorResponse, unavailableResponse } from "../../utils/error.ts"; + +import { clamp01 } from "../../utils/number.ts"; + +import { + recordComboIntent, + recordComboRequest, + recordComboShadowRequest, + getComboMetrics, +} from "../comboMetrics.ts"; + +import { + resolveComboConfig, + getDefaultComboConfig, + resolveComboTargetTimeoutMs, + PRE_SCREEN_CONCURRENCY, +} from "../comboConfig.ts"; + +import { + maybeGenerateHandoff, + resolveContextRelayConfig, + maybeGenerateUniversalHandoff, + injectUniversalHandoffBody, + resolveUniversalHandoffConfig, + SKIP_UNIVERSAL_HANDOFF_FLAG, + type MessageLike, +} from "../contextHandoff.ts"; + +import { + recordSessionModelUsage, + getLastSessionModel, + getHandoff, +} from "../../../src/lib/db/contextHandoffs.ts"; + +import { fetchCodexQuota } from "../codexQuotaFetcher.ts"; + +import { getQuotaFetcher } from "../quotaPreflight.ts"; + +import * as semaphore from "../rateLimitSemaphore.ts"; + +import { getCircuitBreaker } from "../../../src/shared/utils/circuitBreaker"; + +import { fisherYatesShuffle, getNextFromDeck } from "../../../src/shared/utils/shuffleDeck"; + +import { parseModel } from "../model.ts"; + +import { applyComboAgentMiddleware } from "../comboAgentMiddleware.ts"; + +import { checkCredentialGate, logCredentialSkip } from "../credentialGate.ts"; + +import { emit } from "../../../src/lib/events/eventBus"; + +import { notifyWebhookEvent } from "../../../src/lib/webhookDispatcher"; + +import { + classifyWithConfig, + DEFAULT_INTENT_CONFIG, + type IntentClassifierConfig, +} from "../intentClassifier.ts"; + +import { selectProvider as selectAutoProvider } from "../autoCombo/engine.ts"; + +import { selectWithStrategy, type SlaRoutingPolicy } from "../autoCombo/routerStrategy.ts"; + +import { getTaskFitness } from "../autoCombo/taskFitness.ts"; + +import { parseAutoPrefix } from "../autoCombo/autoPrefix.ts"; + +import { handlePipelineCombo, buildPipelineResponse } from "../autoCombo/pipelineRouter.ts"; + +import { + calculateFactors, + calculateScore, + DEFAULT_WEIGHTS, + type ProviderCandidate, + type ScoringWeights, +} from "../autoCombo/scoring.ts"; + +import { + getResolvedModelCapabilities, + supportsReasoning, + supportsToolCalling, +} from "../modelCapabilities.ts"; + +import { estimateTokens } from "../contextManager.ts"; + +import { getReasoningTokens } from "../../../src/lib/usage/tokenAccounting.ts"; + +import { getSessionConnection } from "../sessionManager.ts"; + +import { orderTargetsByEvalScores } from "../evalRouting.ts"; + +import type { CompressionMode } from "../compression/types.ts"; + +import { getModelContextLimit } from "../../../src/lib/modelCapabilities"; + +import { getProviderConnections } from "../../../src/lib/db/providers"; + +import { getProviderModels } from "../../config/providerModels.ts"; + +import { + getComboModelString, + getComboStepTarget, + getComboStepWeight, + normalizeComboStep, +} from "../../../src/lib/combos/steps.ts"; + +import { + getConnectionRoutingTags, + matchesRoutingTags, + resolveRequestRoutingTags, + type RoutingTagMatchMode, +} from "../../../src/domain/tagRouter.ts"; + +import { normalizeRoutingStrategy } from "../../../src/shared/constants/routingStrategies.ts"; + +import { + isProviderInCooldown, + recordProviderCooldown, + recordProviderSuccess, +} from "../providerCooldownTracker.ts"; + +import { + resolveResilienceSettings, + type ResilienceSettings, +} from "../../../src/lib/resilience/settings"; + +import { ResolvedComboTarget } from "./types.ts"; + + + +export function selectWeightedTarget(targets: T[]) { + if (targets.length === 0) return null; + + const totalWeight = targets.reduce((sum, target) => sum + (target.weight || 0), 0); + if (totalWeight <= 0) { + return targets[Math.floor(Math.random() * targets.length)]; + } + + let random = Math.random() * totalWeight; + for (const target of targets) { + random -= target.weight || 0; + if (random <= 0) return target; + } + + return targets.at(-1); +} + + + +export function orderTargetsForWeightedFallback( + targets: T[], + selectedExecutionKey: string, + preserveExistingOrder = false +): T[] { + const selected = targets.find((target) => target.executionKey === selectedExecutionKey); + const rest = targets.filter((target) => target.executionKey !== selectedExecutionKey); + if (!preserveExistingOrder) { + rest.sort((a, b) => b.weight - a.weight); + } + return selected ? [selected, ...rest] : rest; +} + + + +// shuffleArray and getNextModelFromDeck moved to src/shared/utils/shuffleDeck.ts +// combo.ts now uses the shared, mutex-protected getNextFromDeck with "combo:" namespace. + +/** + * Sort models by pricing (cheapest first) for cost-optimized strategy + * @param {Array} models - Model strings in "provider/model" format + * @returns {Promise>} Sorted model strings + */ +async function sortModelsByCost(models: string[]): Promise { + try { + const { getPricingForModel } = await import("../../../src/lib/localDb"); + const withCost = await Promise.all( + models.map(async (modelStr) => { + const parsed = parseModel(modelStr); + const provider = parsed.provider || parsed.providerAlias || "unknown"; + const model = parsed.model || modelStr; + try { + const pricing = await getPricingForModel(provider, model); + const cost = Number(pricing?.input); + return { modelStr, cost: Number.isFinite(cost) ? cost : Infinity }; + } catch { + return { modelStr, cost: Infinity }; + } + }) + ); + withCost.sort((a, b) => a.cost - b.cost); + return withCost.map((e) => e.modelStr); + } catch { + // If pricing lookup fails entirely, return original order + return models; + } +} + + + +export async function sortTargetsByCost(targets: ResolvedComboTarget[]) { + const orderedModels = await sortModelsByCost(targets.map((target) => target.modelStr)); + const byModel = new Map(); + for (const target of targets) { + const queue = byModel.get(target.modelStr) || []; + queue.push(target); + byModel.set(target.modelStr, queue); + } + return orderedModels + .map((modelStr) => { + const queue = byModel.get(modelStr); + return queue?.shift() || null; + }) + .filter((target): target is ResolvedComboTarget => target !== null); +} + + + +/** + * Sort models by usage count (least-used first) for least-used strategy + * @param {Array} models - Model strings + * @param {string} comboName - Combo name for metrics lookup + * @returns {Array} Sorted model strings + */ +function sortModelsByUsage(models: string[], comboName: string): string[] { + const metrics = getComboMetrics(comboName); + if (!metrics?.byModel) return models; + + const withUsage = models.map((modelStr) => ({ + modelStr, + requests: metrics.byModel[modelStr]?.requests ?? 0, + })); + withUsage.sort((a, b) => a.requests - b.requests); + return withUsage.map((e) => e.modelStr); +} + + + +export function sortTargetsByUsage(targets: ResolvedComboTarget[], comboName: string) { + const orderedModels = sortModelsByUsage( + targets.map((target) => target.modelStr), + comboName + ); + const byModel = new Map(); + for (const target of targets) { + const queue = byModel.get(target.modelStr) || []; + queue.push(target); + byModel.set(target.modelStr, queue); + } + return orderedModels + .map((modelStr) => { + const queue = byModel.get(modelStr); + return queue?.shift() || null; + }) + .filter((target): target is ResolvedComboTarget => target !== null); +} + + + +/** + * Sort models by context window size (largest first) for context-optimized strategy. + * Uses models.dev synced capabilities to get context limits. + * @param {Array} models - Model strings in "provider/model" format + * @returns {Array} Sorted model strings (largest context first) + */ +function sortModelsByContextSize(models: string[]): string[] { + const withContext = models.map((modelStr) => { + return { modelStr, context: getModelContextLimitForModelString(modelStr) ?? 0 }; + }); + withContext.sort((a, b) => b.context - a.context); + return withContext.map((e) => e.modelStr); +} + + + +export function getModelContextLimitForModelString(modelStr: string) { + const parsed = parseModel(modelStr); + const provider = parsed.provider || parsed.providerAlias || "unknown"; + const model = parsed.model || modelStr; + return getModelContextLimit(provider, model); +} + + + +export function sortTargetsByContextSize(targets: ResolvedComboTarget[]) { + const hasKnownContext = targets.some( + (target) => getModelContextLimitForModelString(target.modelStr) != null + ); + if (!hasKnownContext) return targets; + + const orderedModels = sortModelsByContextSize(targets.map((target) => target.modelStr)); + const byModel = new Map(); + for (const target of targets) { + const queue = byModel.get(target.modelStr) || []; + queue.push(target); + byModel.set(target.modelStr, queue); + } + return orderedModels + .map((modelStr) => { + const queue = byModel.get(modelStr); + return queue?.shift() || null; + }) + .filter((target): target is ResolvedComboTarget => target !== null); +} + + + +function getP2CTargetScore( + target: ResolvedComboTarget, + metrics: ReturnType +): number { + const breakerState = getCircuitBreaker(target.provider)?.getStatus?.()?.state; + if (breakerState === "OPEN") return -Infinity; + const modelMetric = metrics?.byModel?.[target.modelStr] || null; + const successRate = Number(modelMetric?.successRate); + const avgLatency = Number(modelMetric?.avgLatencyMs); + const successScore = Number.isFinite(successRate) ? successRate / 100 : 0.5; + const latencyScore = + Number.isFinite(avgLatency) && avgLatency > 0 ? 1 / Math.log10(avgLatency + 10) : 0.25; + const breakerPenalty = breakerState === "HALF_OPEN" ? 0.25 : 0; + return successScore + latencyScore - breakerPenalty; +} + + + +export function orderTargetsByPowerOfTwoChoices(targets: ResolvedComboTarget[], comboName: string) { + if (targets.length <= 1) return targets; + const metrics = getComboMetrics(comboName); + const firstIndex = Math.floor(Math.random() * targets.length); + let secondIndex = Math.floor(Math.random() * (targets.length - 1)); + if (secondIndex >= firstIndex) secondIndex++; + + const first = targets[firstIndex]; + const second = targets[secondIndex]; + const selectedIndex = + getP2CTargetScore(second, metrics) > getP2CTargetScore(first, metrics) + ? secondIndex + : firstIndex; + return [targets[selectedIndex], ...targets.filter((_, index) => index !== selectedIndex)]; +} + diff --git a/open-sse/services/combo/state.ts b/open-sse/services/combo/state.ts new file mode 100644 index 00000000000..1cdc491f086 --- /dev/null +++ b/open-sse/services/combo/state.ts @@ -0,0 +1,233 @@ +/** + * Shared combo (model combo) handling with fallback support + * Supports: priority, weighted, round-robin, random, least-used, cost-optimized, + * reset-aware, reset-window, strict-random, auto, fill-first, p2c, lkgp, + * context-optimized, and context-relay strategies + */ + +import { + checkFallbackError, + classifyErrorText, + formatRetryAfter, + getRuntimeProviderProfile, + recordProviderFailure, + isProviderFailureCode, + isProviderExhaustedReason, + type ProviderProfile, +} from "../accountFallback.ts"; + +import { FETCH_TIMEOUT_MS, RateLimitReason } from "../../config/constants.ts"; + +import { errorResponse, unavailableResponse } from "../../utils/error.ts"; + +import { clamp01 } from "../../utils/number.ts"; + +import { + recordComboIntent, + recordComboRequest, + recordComboShadowRequest, + getComboMetrics, +} from "../comboMetrics.ts"; + +import { + resolveComboConfig, + getDefaultComboConfig, + resolveComboTargetTimeoutMs, + PRE_SCREEN_CONCURRENCY, +} from "../comboConfig.ts"; + +import { + maybeGenerateHandoff, + resolveContextRelayConfig, + maybeGenerateUniversalHandoff, + injectUniversalHandoffBody, + resolveUniversalHandoffConfig, + SKIP_UNIVERSAL_HANDOFF_FLAG, + type MessageLike, +} from "../contextHandoff.ts"; + +import { + recordSessionModelUsage, + getLastSessionModel, + getHandoff, +} from "../../../src/lib/db/contextHandoffs.ts"; + +import { fetchCodexQuota } from "../codexQuotaFetcher.ts"; + +import { getQuotaFetcher } from "../quotaPreflight.ts"; + +import * as semaphore from "../rateLimitSemaphore.ts"; + +import { parseModel } from "../model.ts"; + +import { applyComboAgentMiddleware } from "../comboAgentMiddleware.ts"; + +import { checkCredentialGate, logCredentialSkip } from "../credentialGate.ts"; + +import { + classifyWithConfig, + DEFAULT_INTENT_CONFIG, + type IntentClassifierConfig, +} from "../intentClassifier.ts"; + +import { selectProvider as selectAutoProvider } from "../autoCombo/engine.ts"; + +import { selectWithStrategy, type SlaRoutingPolicy } from "../autoCombo/routerStrategy.ts"; + +import { getTaskFitness } from "../autoCombo/taskFitness.ts"; + +import { parseAutoPrefix } from "../autoCombo/autoPrefix.ts"; + +import { handlePipelineCombo, buildPipelineResponse } from "../autoCombo/pipelineRouter.ts"; + +import { + calculateFactors, + calculateScore, + DEFAULT_WEIGHTS, + type ProviderCandidate, + type ScoringWeights, +} from "../autoCombo/scoring.ts"; + +import { + getResolvedModelCapabilities, + supportsReasoning, + supportsToolCalling, +} from "../modelCapabilities.ts"; + +import { estimateTokens } from "../contextManager.ts"; + +import { getReasoningTokens } from "../../../src/lib/usage/tokenAccounting.ts"; + +import { getSessionConnection } from "../sessionManager.ts"; + +import { orderTargetsByEvalScores } from "../evalRouting.ts"; + +import type { CompressionMode } from "../compression/types.ts"; + +import { getProviderModels } from "../../config/providerModels.ts"; + +import { + getComboModelString, + getComboStepTarget, + getComboStepWeight, + normalizeComboStep, +} from "../../../src/lib/combos/steps.ts"; + +import { + getConnectionRoutingTags, + matchesRoutingTags, + resolveRequestRoutingTags, + type RoutingTagMatchMode, +} from "../../../src/domain/tagRouter.ts"; + +import { normalizeRoutingStrategy } from "../../../src/shared/constants/routingStrategies.ts"; + +import { + isProviderInCooldown, + recordProviderCooldown, + recordProviderSuccess, +} from "../providerCooldownTracker.ts"; + +import { buildAutoCandidates, scoreAutoTargets } from "./auto.ts"; +import { QUOTA_SOFT_DEPRIORITIZE_FACTOR } from "./constants.ts"; +import { handleComboChat } from "./chat.ts"; + + + +// G2: Module-level registry of active combo execution candidates. +// Maps executionKey → Map. +// Populated by buildAutoCandidates registrations; cleaned up after each execution. +// This allows chatCore.ts to mark a candidate's quotaSoftPenalty flag so that +// subsequent scoring iterations (auto-combo fallback) deprioritize it. +const _activeExecutionCandidates = new Map>(); + + + +/** + * Mark a specific candidate (by comboExecutionKey + stepId) with soft quota penalty. + * Called from chatCore.ts when enforceQuotaShare returns a "soft deprioritize" decision. + * The flag is read on subsequent auto-combo scoring iterations (fallback chain) + * within the same combo execution via scoreAutoTargets → QUOTA_SOFT_DEPRIORITIZE_FACTOR. + * + * Guards: + * - null executionKey or stepId → no-op (non-combo or context not available). + * - unknown executionKey → no-op (candidate not yet registered or already cleaned up). + * - Idempotent: calling twice with the same (key, stepId, true) is safe. + */ +export function setCandidateQuotaSoftPenalty( + comboExecutionKey: string | null, + comboStepId: string | null, + penalty: boolean +): void { + if (!comboExecutionKey || !comboStepId) return; + const byStep = _activeExecutionCandidates.get(comboExecutionKey); + if (!byStep) return; + const candidate = byStep.get(comboStepId); + if (candidate) { + candidate.quotaSoftPenalty = penalty; + } +} + + + +/** + * Register candidates for a combo execution so setCandidateQuotaSoftPenalty can + * locate them by (executionKey, stepId). + * Each candidate object is stored by reference — mutations via setCandidateQuotaSoftPenalty + * propagate back to the original candidate array used by scoreAutoTargets. + * @internal — not exported; only called within combo.ts by buildAutoCandidates callers. + */ +export function _registerExecutionCandidates( + candidates: Array<{ executionKey: string; stepId: string; quotaSoftPenalty?: boolean }> +): void { + for (const candidate of candidates) { + if (!candidate.executionKey) continue; + let byStep = _activeExecutionCandidates.get(candidate.executionKey); + if (!byStep) { + byStep = new Map(); + _activeExecutionCandidates.set(candidate.executionKey, byStep); + } + byStep.set(candidate.stepId, candidate); + } +} + + + +/** + * Unregister all candidates for a given execution key once the execution completes. + * Prevents unbounded memory growth. + * @internal — not exported; called after each handleComboChat iteration. + */ +export function _unregisterExecutionCandidates(executionKeys: string[]): void { + for (const key of executionKeys) { + _activeExecutionCandidates.delete(key); + } +} + + + +// In-memory atomic counter per combo for round-robin distribution +// Resets on server restart (by design — no stale state) +// Eviction limits to prevent unbounded memory growth +export const MAX_RR_COUNTERS = 500; + + +export const MAX_RESET_AWARE_CACHE = 200; + + + +export const rrCounters = new Map(); + + + +export const resetAwareConnectionCache = new Map< + string, + { fetchedAt: number; connections: Array> } +>(); + + +export const resetAwareQuotaCache = new Map< + string, + { fetchedAt: number; quota: unknown; refreshPromise: Promise | null } +>(); + diff --git a/open-sse/services/combo/types.ts b/open-sse/services/combo/types.ts new file mode 100644 index 00000000000..4ee0dd92219 --- /dev/null +++ b/open-sse/services/combo/types.ts @@ -0,0 +1,241 @@ +/** + * Shared combo (model combo) handling with fallback support + * Supports: priority, weighted, round-robin, random, least-used, cost-optimized, + * reset-aware, reset-window, strict-random, auto, fill-first, p2c, lkgp, + * context-optimized, and context-relay strategies + */ + +import { + checkFallbackError, + classifyErrorText, + formatRetryAfter, + getRuntimeProviderProfile, + recordProviderFailure, + isProviderFailureCode, + isProviderExhaustedReason, + type ProviderProfile, +} from "../accountFallback.ts"; + +import { FETCH_TIMEOUT_MS, RateLimitReason } from "../../config/constants.ts"; + +import { errorResponse, unavailableResponse } from "../../utils/error.ts"; + +import { clamp01 } from "../../utils/number.ts"; + +import { parseModel } from "../model.ts"; + +import { + calculateFactors, + calculateScore, + DEFAULT_WEIGHTS, + type ProviderCandidate, + type ScoringWeights, +} from "../autoCombo/scoring.ts"; + +import { getProviderModels } from "../../config/providerModels.ts"; + +import { + getComboModelString, + getComboStepTarget, + getComboStepWeight, + normalizeComboStep, +} from "../../../src/lib/combos/steps.ts"; + +import { + resolveResilienceSettings, + type ResilienceSettings, +} from "../../../src/lib/resilience/settings"; + +import { resolveResetWindowConfig } from "./quota.ts"; +import { QUOTA_SOFT_DEPRIORITIZE_FACTOR } from "./constants.ts"; + + + +export const RESET_WINDOW_NAMES = ["weekly", "session", "monthly"] as const; + + +export type ResetWindowName = (typeof RESET_WINDOW_NAMES)[number]; + + +export type QuotaFetchCacheConfig = { + quotaCacheTtlMs: number; + quotaCacheMaxStaleMs: number; +}; + + +export type ResetWindowConfig = ReturnType; + + +export type ComboRetryAfter = string | number | Date; + + +export type ComboErrorBody = { + error?: { code?: string | null; message?: string | null } | string; + message?: string | null; + retryAfter?: ComboRetryAfter | null; +} | null; + + + +export type ComboLike = { + id?: string; + name: string; + strategy?: string | null; + models: unknown[]; + config?: Record | null; + autoConfig?: Record | null; + context_cache_protection?: boolean | number; + system_message?: string | null; + [key: string]: unknown; +}; + + + +export type ComboInput = ComboLike | Record; + + + +export type ComboCollectionLike = ComboInput[] | { combos?: ComboInput[] } | null | undefined; + + + +export type ComboLogger = { + info: (...args: unknown[]) => void; + warn: (...args: unknown[]) => void; + error?: (...args: unknown[]) => void; + debug: (...args: unknown[]) => void; +}; + + + +export type SingleModelTarget = + | (ResolvedComboTarget & { + allowRateLimitedConnection?: boolean; + modelAbortSignal?: AbortSignal | null; + }) + | { modelAbortSignal: AbortSignal }; + + + +export type HandleSingleModel = ( + body: Record, + modelStr: string, + target?: SingleModelTarget +) => Promise; + + + +export type IsModelAvailable = ( + modelStr: string, + target?: ResolvedComboTarget & { allowRateLimitedConnection?: boolean } +) => Promise | boolean; + + + +type ComboRelayOptions = { + sessionId?: string | null; + config?: Record | null; + [key: string]: unknown; +}; + + + +export type HandleComboChatOptions = { + body: Record; + combo: ComboLike; + handleSingleModel: HandleSingleModel; + isModelAvailable?: IsModelAvailable; + log: ComboLogger; + settings?: Record | null; + allCombos?: ComboCollectionLike; + relayOptions?: ComboRelayOptions | null; + signal?: AbortSignal | null; + apiKeyAllowedConnections?: string[] | null; +}; + + + +export type HandleRoundRobinOptions = Omit< + HandleComboChatOptions, + "relayOptions" | "apiKeyAllowedConnections" +>; + + + +export type HistoricalLatencyStatsEntry = { + totalRequests?: number; + p95LatencyMs?: number; + latencyStdDev?: number; + successRate?: number; +}; + + + +export type AutoProviderCandidate = ProviderCandidate & { + stepId: string; + executionKey: string; + modelStr: string; + /** + * When true, this candidate's auto-combo score is multiplied by + * QUOTA_SOFT_DEPRIORITIZE_FACTOR (B17 soft-policy penalty). + * Set externally when enforceQuotaShare returns deprioritize=true + * for the key routed through this target's connectionId. + */ + quotaSoftPenalty?: boolean; +}; + + + +export type ResolvedComboTarget = { + kind: "model"; + stepId: string; + executionKey: string; + modelStr: string; + provider: string; + providerId: string | null; + connectionId: string | null; + allowedConnectionIds?: string[] | null; + weight: number; + label: string | null; + failoverBeforeRetry?: unknown; + trafficType?: "production" | "shadow"; +}; + + + +export type ShadowRoutingConfig = { + enabled: boolean; + targets: unknown[]; + sampleRate: number; + maxTargets: number; + timeoutMs: number; +}; + + + +export type ComboRuntimeStep = + | ResolvedComboTarget + | { + kind: "combo-ref"; + stepId: string; + executionKey: string; + comboName: string; + weight: number; + label: string | null; + }; + + + +export type RequestCompatibilityRequirements = { + requiresTools: boolean; + requiresVision: boolean; + requiresStructuredOutput: boolean; + estimatedInputTokens: number; + requestedOutputTokens: number; + requiredContextTokens: number; +}; + + + +export type PreScreenResult = { profile: ProviderProfile | null; available: boolean }; + diff --git a/open-sse/services/combo/utils.ts b/open-sse/services/combo/utils.ts new file mode 100644 index 00000000000..4a43322a070 --- /dev/null +++ b/open-sse/services/combo/utils.ts @@ -0,0 +1,400 @@ +/** + * Shared combo (model combo) handling with fallback support + * Supports: priority, weighted, round-robin, random, least-used, cost-optimized, + * reset-aware, reset-window, strict-random, auto, fill-first, p2c, lkgp, + * context-optimized, and context-relay strategies + */ + +import { + checkFallbackError, + classifyErrorText, + formatRetryAfter, + getRuntimeProviderProfile, + recordProviderFailure, + isProviderFailureCode, + isProviderExhaustedReason, + type ProviderProfile, +} from "../accountFallback.ts"; + +import { FETCH_TIMEOUT_MS, RateLimitReason } from "../../config/constants.ts"; + +import { errorResponse, unavailableResponse } from "../../utils/error.ts"; + +import { clamp01 } from "../../utils/number.ts"; + +import { + recordComboIntent, + recordComboRequest, + recordComboShadowRequest, + getComboMetrics, +} from "../comboMetrics.ts"; + +import { + resolveComboConfig, + getDefaultComboConfig, + resolveComboTargetTimeoutMs, + PRE_SCREEN_CONCURRENCY, +} from "../comboConfig.ts"; + +import { + maybeGenerateHandoff, + resolveContextRelayConfig, + maybeGenerateUniversalHandoff, + injectUniversalHandoffBody, + resolveUniversalHandoffConfig, + SKIP_UNIVERSAL_HANDOFF_FLAG, + type MessageLike, +} from "../contextHandoff.ts"; + +import { + recordSessionModelUsage, + getLastSessionModel, + getHandoff, +} from "../../../src/lib/db/contextHandoffs.ts"; + +import { fetchCodexQuota } from "../codexQuotaFetcher.ts"; + +import { getQuotaFetcher } from "../quotaPreflight.ts"; + +import * as semaphore from "../rateLimitSemaphore.ts"; + +import { getCircuitBreaker } from "../../../src/shared/utils/circuitBreaker"; + +import { fisherYatesShuffle, getNextFromDeck } from "../../../src/shared/utils/shuffleDeck"; + +import { parseModel } from "../model.ts"; + +import { applyComboAgentMiddleware } from "../comboAgentMiddleware.ts"; + +import { checkCredentialGate, logCredentialSkip } from "../credentialGate.ts"; + +import { emit } from "../../../src/lib/events/eventBus"; + +import { + classifyWithConfig, + DEFAULT_INTENT_CONFIG, + type IntentClassifierConfig, +} from "../intentClassifier.ts"; + +import { selectProvider as selectAutoProvider } from "../autoCombo/engine.ts"; + +import { selectWithStrategy, type SlaRoutingPolicy } from "../autoCombo/routerStrategy.ts"; + +import { getTaskFitness } from "../autoCombo/taskFitness.ts"; + +import { parseAutoPrefix } from "../autoCombo/autoPrefix.ts"; + +import { handlePipelineCombo, buildPipelineResponse } from "../autoCombo/pipelineRouter.ts"; + +import { + calculateFactors, + calculateScore, + DEFAULT_WEIGHTS, + type ProviderCandidate, + type ScoringWeights, +} from "../autoCombo/scoring.ts"; + +import { + getResolvedModelCapabilities, + supportsReasoning, + supportsToolCalling, +} from "../modelCapabilities.ts"; + +import { estimateTokens } from "../contextManager.ts"; + +import { getReasoningTokens } from "../../../src/lib/usage/tokenAccounting.ts"; + +import { getSessionConnection } from "../sessionManager.ts"; + +import { orderTargetsByEvalScores } from "../evalRouting.ts"; + +import type { CompressionMode } from "../compression/types.ts"; + +import { getProviderModels } from "../../config/providerModels.ts"; + +import { + getComboModelString, + getComboStepTarget, + getComboStepWeight, + normalizeComboStep, +} from "../../../src/lib/combos/steps.ts"; + +import { + getConnectionRoutingTags, + matchesRoutingTags, + resolveRequestRoutingTags, + type RoutingTagMatchMode, +} from "../../../src/domain/tagRouter.ts"; + +import { normalizeRoutingStrategy } from "../../../src/shared/constants/routingStrategies.ts"; + +import { + isProviderInCooldown, + recordProviderCooldown, + recordProviderSuccess, +} from "../providerCooldownTracker.ts"; + +import { ComboRetryAfter, ComboInput, ComboLike, ComboCollectionLike, ResolvedComboTarget } from "./types.ts"; + + + +export function toRetryAfterDisplayValue(value: ComboRetryAfter): string | Date { + if (typeof value !== "number") return value; + if (value > 0 && value < 1_000_000_000) { + return new Date(Date.now() + value * 1000); + } + return new Date(value); +} + + + +export function isRecord(value: unknown): value is Record { + return !!value && typeof value === "object" && !Array.isArray(value); +} + + + +export function toTrimmedString(value: unknown): string | null { + return typeof value === "string" && value.trim().length > 0 ? value.trim() : null; +} + + + +function toComboLike(combo: ComboInput): ComboLike { + return { + ...combo, + id: toTrimmedString(combo.id) || undefined, + name: toTrimmedString(combo.name) || "", + models: Array.isArray(combo.models) ? combo.models : [], + config: isRecord(combo.config) ? combo.config : null, + autoConfig: isRecord(combo.autoConfig) ? combo.autoConfig : null, + context_cache_protection: + typeof combo.context_cache_protection === "boolean" || + typeof combo.context_cache_protection === "number" + ? combo.context_cache_protection + : undefined, + system_message: typeof combo.system_message === "string" ? combo.system_message : null, + }; +} + + + +export function getCombosArray(allCombos: ComboCollectionLike): ComboLike[] { + const combos = Array.isArray(allCombos) ? allCombos : allCombos?.combos || []; + return combos.map((combo) => toComboLike(combo)); +} + + + +/** + * Validate that a successful (HTTP 200) non-streaming response actually contains + * meaningful content. Returns { valid: true } or { valid: false, reason }. + * + * Only inspects non-streaming JSON responses — streaming responses are passed through + * because buffering the full stream would defeat the purpose of streaming. + * + * Checks: + * 1. Body is valid JSON + * 2. Has at least one choice with non-empty content or tool_calls + */ +export async function validateResponseQuality( + response: Response, + isStreaming: boolean, + log: { warn?: (...args: unknown[]) => void } +): Promise<{ valid: boolean; reason?: string; clonedResponse?: Response }> { + if (isStreaming) return { valid: true }; + + const contentType = response.headers.get("content-type") || ""; + if (!contentType.includes("application/json") && !contentType.includes("text/")) { + return { valid: true }; + } + + let cloned: Response; + try { + cloned = response.clone(); + } catch { + return { valid: true }; + } + + let text: string; + try { + text = await cloned.text(); + } catch { + return { valid: true }; + } + + if (!text || text.trim().length === 0) { + return { valid: false, reason: "empty response body" }; + } + + let json: Record; + try { + json = JSON.parse(text); + } catch { + if (text.startsWith("data:") || text.startsWith("event:")) return { valid: true }; + return { valid: false, reason: "response is not valid JSON" }; + } + + const choices = json?.choices; + if (!Array.isArray(choices) || choices.length === 0) { + if (json?.output || json?.result || json?.data || json?.response) return { valid: true }; + if (json?.error) { + const err = json.error as Record; + return { + valid: false, + reason: `upstream error in 200 body: ${err?.message || JSON.stringify(json.error).substring(0, 200)}`, + }; + } + return { valid: true }; + } + + const firstChoice = choices[0]; + const message = firstChoice?.message || firstChoice?.delta; + if (!message) { + return { valid: false, reason: "choice has no message object" }; + } + + const content = message.content; + const toolCalls = message.tool_calls; + // Issue #2341: Reasoning models (Kimi-K2.5-TEE, GLM-5-TEE, etc.) emit their + // output in `reasoning_content` (or `reasoning`) with `content: null`. The + // validator used to flag those as empty and trigger a false-positive 502 + // fallback. Count a non-empty reasoning_content as valid output too. + const reasoningContent = message.reasoning_content ?? message.reasoning; + const hasReasoningContent = + typeof reasoningContent === "string" && reasoningContent.trim().length > 0; + const hasContent = + (content !== null && content !== undefined && content !== "") || hasReasoningContent; + const hasToolCalls = Array.isArray(toolCalls) && toolCalls.length > 0; + + if (!hasContent && !hasToolCalls) { + return { valid: false, reason: "empty content and no tool_calls in response" }; + } + + // Issue #3587: Reasoning models (deepseek-v4-flash, nemotron, etc.) may consume + // ALL max_tokens for reasoning_tokens, leaving content empty. When content is + // empty but reasoning_content exists, and usage shows reasoning consumed nearly + // all completion tokens, treat as invalid so the combo loop retries with more + // tokens or falls back to a non-reasoning model. + const contentIsEmpty = content === null || content === undefined || content === ""; + if (contentIsEmpty && hasReasoningContent && !hasToolCalls) { + const usage = json?.usage as Record | undefined; + if (usage) { + const completionTokens = Number(usage.completion_tokens) || 0; + const reasoningTokens = getReasoningTokens(usage); + // If reasoning consumed 90%+ of completion tokens, the model ran out of + // budget before producing any content output. + if (completionTokens > 0 && reasoningTokens >= completionTokens * 0.9) { + return { + valid: false, + reason: `reasoning consumed ${reasoningTokens}/${completionTokens} tokens — no content output`, + }; + } + } + } + + return { + valid: true, + clonedResponse: new Response(text, { + status: response.status, + statusText: response.statusText, + headers: response.headers, + }), + }; +} + + + +/** + * Normalize a model entry to { model, weight } + * Supports both legacy string format and new object format + */ +export function normalizeModelEntry(entry: unknown): { model: string; weight: number } { + return { + model: getComboStepTarget(entry) || "", + weight: getComboStepWeight(entry), + }; +} + + + +export function getTargetProvider(modelStr: string, providerId?: string | null): string { + const parsed = parseModel(modelStr); + return providerId || parsed.provider || parsed.providerAlias || "unknown"; +} + + + +export function isStreamReadinessFailureErrorBody(errorBody: unknown): boolean { + if (!errorBody || typeof errorBody !== "object") return false; + const error = (errorBody as Record).error; + if (!error || typeof error !== "object") return false; + const code = (error as Record).code; + return code === "STREAM_READINESS_TIMEOUT" || code === "STREAM_EARLY_EOF"; +} + + + +/** + * A local per-API-key token-limit breach surfaces as a 429 tagged with + * errorCode "TOKEN_LIMIT_EXCEEDED" (see chatCore.ts Tier 2 early return). This + * is NOT an upstream rate limit, so the combo loop must not cool the shared + * account/provider, must not add it to transientRateLimitedProviders, and must + * not retry it transiently — it propagates to the client as a terminal 429. + */ +export function isTokenLimitBreachErrorBody(errorBody: unknown): boolean { + if (!errorBody || typeof errorBody !== "object") return false; + const error = (errorBody as Record).error; + if (!error || typeof error !== "object") return false; + return (error as Record).code === "TOKEN_LIMIT_EXCEEDED"; +} + + + +export function toRecordedTarget(target: ResolvedComboTarget) { + return { + executionKey: target.executionKey, + stepId: target.stepId, + provider: target.provider, + providerId: target.providerId, + connectionId: target.connectionId, + label: target.label, + }; +} + + + +export function finiteNumberOrNull(value: unknown): number | null { + const numericValue = Number(value); + return Number.isFinite(numericValue) ? numericValue : null; +} + + + +export function toTextContent(content: unknown): string { + if (typeof content === "string") return content; + if (!Array.isArray(content)) return ""; + return content + .map((part) => { + if (!isRecord(part)) return ""; + if (typeof part.text === "string") return part.text; + return ""; + }) + .join("\n"); +} + + + +export function toStringArray(input: unknown): string[] { + if (Array.isArray(input)) { + return input.map((v) => (typeof v === "string" ? v.trim() : "")).filter(Boolean); + } + if (typeof input === "string") { + return input + .split(",") + .map((v) => v.trim()) + .filter(Boolean); + } + return []; +} + diff --git a/open-sse/services/comboConfig.ts b/open-sse/services/comboConfig.ts index e8c9c33a52e..5afe81f6aa2 100644 --- a/open-sse/services/comboConfig.ts +++ b/open-sse/services/comboConfig.ts @@ -26,6 +26,7 @@ const DEFAULT_COMBO_CONFIG = { maxMessagesForSummary: 30, maxComboDepth: 3, trackMetrics: true, + reasoningTokenBufferEnabled: true, manifestRouting: false, resetAwareSessionWeight: 0.35, resetAwareWeeklyWeight: 0.65, diff --git a/open-sse/services/geminiRateLimitTracker.ts b/open-sse/services/geminiRateLimitTracker.ts new file mode 100644 index 00000000000..7ede8a4f883 --- /dev/null +++ b/open-sse/services/geminiRateLimitTracker.ts @@ -0,0 +1,145 @@ +/** + * In-memory request counters for Gemini models — tracks both RPD (daily) + * and RPM (sliding 60s window) so that 429 responses can be classified + * as either quota_exhausted (RPD hit) or rate_limit_exceeded (RPM hit). + * + * Gemini returns identical error bodies for both types, so we rely on + * published per-model limits from geminiRateLimits.json to distinguish them. + * + * Counters are incremented on every Gemini request so that once usage + * reaches the published limit, subsequent 429s are correctly classified. + */ + +import geminiLimits from "../config/geminiRateLimits.json"; + +// ── RPD (daily) state ──────────────────────────────────────────────────────── + +interface DailyCount { + date: string; // "YYYY-MM-DD" + count: number; +} + +const dailyCounts = new Map(); + +// ── RPM (sliding 60s window) state ─────────────────────────────────────────── + +const minuteWindows = new Map(); + +// ── Helpers ────────────────────────────────────────────────────────────────── + +function toDateKey(): string { + return new Date().toISOString().slice(0, 10); +} + +function stripModelPrefix(modelId: string): string { + // Only strip the "gemini/" provider prefix, never "gemini-" which is part + // of the actual model name (e.g. "gemini-2.5-flash", "gemini-3.5-live-translate"). + return modelId.replace(/^gemini\//, "").trim(); +} + +function lookupValue(modelId: string, field: "rpm" | "rpd"): number { + if (!modelId) return 0; + const key = stripModelPrefix(modelId); + const entry = (geminiLimits as Record>)[key]; + if (!entry) { + for (const [knownKey, knownEntry] of Object.entries(geminiLimits)) { + if (key.endsWith(knownKey) || knownKey.endsWith(key)) { + const val = knownEntry[field]; + return typeof val === "number" && val > 0 ? val : 0; + } + } + return 0; + } + const val = entry[field]; + return typeof val === "number" && val > 0 ? val : 0; +} + +// ── RPD exports ────────────────────────────────────────────────────────────── + +export function getModelRpd(modelId: string): number { + return lookupValue(modelId, "rpd"); +} + +export function incrementDailyRequestCount(modelId: string): void { + if (!modelId) return; + const key = stripModelPrefix(modelId); + const today = toDateKey(); + const existing = dailyCounts.get(key); + if (existing && existing.date === today) { + existing.count++; + } else { + dailyCounts.set(key, { date: today, count: 1 }); + } +} + +export function getDailyRequestCount(modelId: string): number { + if (!modelId) return 0; + const key = stripModelPrefix(modelId); + const today = toDateKey(); + const entry = dailyCounts.get(key); + if (entry && entry.date === today) return entry.count; + return 0; +} + +export function isRpdExhausted(modelId: string): boolean { + const rpd = getModelRpd(modelId); + if (rpd <= 0) return false; + return getDailyRequestCount(modelId) >= rpd; +} + +// ── RPM exports ────────────────────────────────────────────────────────────── + +export function getModelRpm(modelId: string): number { + return lookupValue(modelId, "rpm"); +} + +/** Prune timestamps older than 60 seconds from a model's window. */ +function pruneMinuteWindow(key: string): void { + const now = Date.now(); + const cutoff = now - 60_000; + const timestamps = minuteWindows.get(key); + if (!timestamps) return; + // Keep only timestamps >= cutoff + let i = 0; + while (i < timestamps.length && timestamps[i] < cutoff) i++; + if (i > 0) { + minuteWindows.set(key, timestamps.slice(i)); + } +} + +export function incrementMinuteRequestCount(modelId: string): void { + if (!modelId) return; + const key = stripModelPrefix(modelId); + pruneMinuteWindow(key); + const timestamps = minuteWindows.get(key) ?? []; + timestamps.push(Date.now()); + minuteWindows.set(key, timestamps); +} + +export function getMinuteRequestCount(modelId: string): number { + if (!modelId) return 0; + const key = stripModelPrefix(modelId); + pruneMinuteWindow(key); + return minuteWindows.get(key)?.length ?? 0; +} + +export function isRpmExhausted(modelId: string): boolean { + const rpm = getModelRpm(modelId); + if (rpm <= 0) return false; + return getMinuteRequestCount(modelId) >= rpm; +} + +// ── Increment both (convenience) ───────────────────────────────────────────── + +/** Increment both daily and minute counters for a Gemini request. */ +export function incrementRequestCount(modelId: string): void { + incrementDailyRequestCount(modelId); + incrementMinuteRequestCount(modelId); +} + +// ── Reset (testing) ────────────────────────────────────────────────────────── + +export function resetCounters(): void { + dailyCounts.clear(); + minuteWindows.clear(); +} diff --git a/open-sse/services/reasoningTokenBuffer.ts b/open-sse/services/reasoningTokenBuffer.ts new file mode 100644 index 00000000000..23290de713a --- /dev/null +++ b/open-sse/services/reasoningTokenBuffer.ts @@ -0,0 +1,40 @@ +import { getResolvedModelCapabilities } from "../../src/lib/modelCapabilities.ts"; +import { MODEL_SPECS } from "../../src/shared/constants/modelSpecs.ts"; + +const DEFAULT_MAX_OUTPUT_TOKENS = MODEL_SPECS.__default__.maxOutputTokens; + +export function toPositiveInteger(value: unknown): number | null { + const numericValue = + typeof value === "number" + ? value + : typeof value === "string" && value.trim() !== "" + ? Number(value) + : null; + if (numericValue === null || !Number.isFinite(numericValue)) return null; + const normalized = Math.floor(numericValue); + return normalized > 0 ? normalized : null; +} + +export function resolveReasoningBufferedMaxTokens( + modelStr: string, + currentMaxTokens: unknown, + options: { enabled?: boolean } = {} +): number | null { + if (options.enabled === false) return null; + + const current = toPositiveInteger(currentMaxTokens); + if (current === null) return null; + + const capabilities = getResolvedModelCapabilities(modelStr); + if (capabilities.supportsThinking !== true) return null; + + const maxOutputTokens = toPositiveInteger(capabilities.maxOutputTokens); + if (maxOutputTokens === null || maxOutputTokens === DEFAULT_MAX_OUTPUT_TOKENS) return null; + if (current > maxOutputTokens) return maxOutputTokens; + if (current === maxOutputTokens) return current; + + const buffered = Math.max(current + 1000, Math.ceil(current * 1.5)); + if (buffered > maxOutputTokens) return current; + + return buffered; +} diff --git a/open-sse/services/tokenExtractionConfig.ts b/open-sse/services/tokenExtractionConfig.ts index d9f6f0a86c6..f82804e2e0d 100644 --- a/open-sse/services/tokenExtractionConfig.ts +++ b/open-sse/services/tokenExtractionConfig.ts @@ -168,16 +168,24 @@ const RAW_CONFIGS: TokenExtractionConfig[] = [ ), // ── Qwen Web ────────────────────────────────────────────── + // The v2 API sits behind Alibaba's "baxia" WAF, which needs the full browser + // cookie jar (cna + ssxmod_itna/itna2 + token), not just the bearer token. + // Capture the WAF cookies alongside the localStorage token (#3288). config( "qwen-web", "Qwen Web (Tongyi)", "https://chat.qwen.ai/", "https://chat.qwen.ai", [ - { type: "cookie", name: "XSRF_TOKEN", domain: ".chat.qwen.ai" }, { type: "localStorage", key: "token" }, + { type: "cookie", name: "token", domain: ".chat.qwen.ai" }, + { type: "cookie", name: "cna", domain: ".chat.qwen.ai" }, + { type: "cookie", name: "ssxmod_itna", domain: ".chat.qwen.ai" }, + { type: "cookie", name: "ssxmod_itna2", domain: ".chat.qwen.ai" }, + { type: "cookie", name: "XSRF_TOKEN", domain: ".chat.qwen.ai" }, ], - "Log in to Qwen at chat.qwen.ai using your Alibaba account. The session token will be extracted.", + "Log in to Qwen at chat.qwen.ai using your Alibaba account. The session token and the " + + "Alibaba WAF cookies (cna, ssxmod_itna) will be extracted — all are required by the v2 API.", { cookieDomain: ".chat.qwen.ai" } ), diff --git a/open-sse/services/tokenRefresh.ts b/open-sse/services/tokenRefresh.ts index 4abb7acbee0..0581a77bf6b 100755 --- a/open-sse/services/tokenRefresh.ts +++ b/open-sse/services/tokenRefresh.ts @@ -181,6 +181,93 @@ function getRefreshCacheKey(provider, refreshToken) { return `${provider}:${tokenHash}`; } +/** + * OAuth2 error codes that mean the refresh token is permanently dead and + * retrying will never succeed → callers must emit the unrecoverable sentinel + * so the HealthCheck deactivates the account instead of looping every 60s. + * Deliberately EXCLUDES transient codes (server_error, temporarily_unavailable, + * slow_down) so we never deactivate an account over a recoverable blip. + */ +const UNRECOVERABLE_OAUTH_ERROR_CODES = new Set([ + "invalid_grant", + "invalid_request", + "refresh_token_reused", + "invalid_token", + "expired_token", + "unauthorized_client", + "access_denied", +]); + +/** + * Extract a canonical OAuth error code from a refresh-endpoint error body of + * ANY shape. Production proxies/MITMs deliver the same `invalid_grant` 400 in + * several shapes — a plain object `{error:"invalid_grant"}`, a nested + * `{error:{code:"invalid_grant"}}`, a JSON **string** (double-encoded body), + * or the raw JSON text wrapped as `{error:""}` by a catch branch. + * The old `errorBody.error === "invalid_grant"` only matched the first shape, + * so the others returned `null` → the HealthCheck refresh loop (root cause of + * the 1352× claude/aa5dd5cf invalidation storm). + * + * Returns the matched code (only if it is in UNRECOVERABLE_OAUTH_ERROR_CODES) + * or null. Never matches loosely — a known code is accepted only when it is a + * bare code string or the value of an `"error"`/`"error_code"` field, so a 502 + * HTML page or a `server_error` body never becomes a false positive. + */ +export function extractOAuthErrorCode(raw: unknown, depth = 0): string | null { + if (raw == null || depth > 6) return null; + + if (typeof raw === "string") { + const s = raw.trim(); + if (!s) return null; + if (UNRECOVERABLE_OAUTH_ERROR_CODES.has(s)) return s; + // The string may itself be JSON (a double-encoded body, or the raw text). + if (s[0] === "{" || s[0] === "[" || s[0] === '"') { + try { + const nested = extractOAuthErrorCode(JSON.parse(s), depth + 1); + if (nested) return nested; + } catch { + // not valid JSON — fall through to the field scan + } + } + // Safety net: a known code appearing as the value of an "error"/"error_code" + // field inside otherwise-unparsed text. Scoped to avoid false positives. + const m = s.match(/"error(?:_code)?"\s*:\s*"([a-z_]+)"/i); + if (m && UNRECOVERABLE_OAUTH_ERROR_CODES.has(m[1])) return m[1]; + return null; + } + + if (typeof raw === "object") { + const o = raw as Record; + return ( + extractOAuthErrorCode(o.error, depth + 1) ?? + extractOAuthErrorCode(o.code, depth + 1) ?? + extractOAuthErrorCode(o.error_code, depth + 1) + ); + } + + return null; +} + +/** + * Read an error response body ONCE and classify it. Returns the raw text (for + * logging) and the extracted unrecoverable OAuth code (or null). Reading once + * avoids the double-read bug where `response.json()` consumes the stream and a + * later `response.text()` returns empty. + */ +async function readRefreshErrorBody( + response: Response +): Promise<{ rawText: string; code: string | null }> { + const rawText = await response.text().catch(() => ""); + let parsed: unknown = rawText; + try { + parsed = JSON.parse(rawText); + } catch { + // keep rawText as-is + } + const code = extractOAuthErrorCode(parsed) ?? extractOAuthErrorCode(rawText); + return { rawText, code }; +} + /** * Refresh OAuth access token using refresh token */ @@ -229,6 +316,10 @@ export async function refreshAccessToken( status: response.status, error: errorText, }); + const code = extractOAuthErrorCode(errorText); + if (code === "invalid_grant" || code === "invalid_request") { + return { error: "unrecoverable_refresh_error", code }; + } return null; } @@ -401,6 +492,10 @@ export async function refreshClineToken(refreshToken, log, proxyConfig: unknown status: response.status, error: errorText, }); + const code = extractOAuthErrorCode(errorText); + if (code === "invalid_grant" || code === "invalid_request") { + return { error: "unrecoverable_refresh_error", code }; + } return null; } @@ -664,19 +759,17 @@ export async function refreshClaudeOAuthToken(refreshToken, log, proxyConfig: un ); if (!response.ok) { - let errorBody: { error?: string; error_description?: string } = {}; - try { - errorBody = await response.json(); - } catch { - const text = await response.text().catch(() => "unknown"); - errorBody = { error: text }; - } + // Read + classify the body ONCE, shape-agnostic. A proxy/MITM can deliver + // the invalid_grant 400 as a JSON string, a double-encoded string, a + // nested {error:{code}}, or raw text — all must yield the sentinel so the + // HealthCheck deactivates instead of looping every 60s. + const { rawText, code } = await readRefreshErrorBody(response); log?.error?.("TOKEN_REFRESH", "Failed to refresh Claude OAuth token", { status: response.status, - error: errorBody, + error: rawText.slice(0, 300), }); - if (errorBody.error === "invalid_grant" || errorBody.error === "invalid_request") { - return { error: "unrecoverable_refresh_error", code: errorBody.error }; + if (code === "invalid_grant" || code === "invalid_request") { + return { error: "unrecoverable_refresh_error", code }; } return null; } @@ -1186,6 +1279,10 @@ export async function refreshQoderToken(refreshToken, log, proxyConfig: unknown status: response.status, error: errorText, }); + const code = extractOAuthErrorCode(errorText); + if (code === "invalid_grant" || code === "invalid_request") { + return { error: "unrecoverable_refresh_error", code }; + } return null; } @@ -1230,6 +1327,10 @@ export async function refreshGitHubToken(refreshToken, log, proxyConfig: unknown status: response.status, error: errorText, }); + const code = extractOAuthErrorCode(errorText); + if (code === "invalid_grant" || code === "invalid_request") { + return { error: "unrecoverable_refresh_error", code }; + } return null; } diff --git a/open-sse/services/usage.ts b/open-sse/services/usage.ts index d33cbcea4ab..6a69590d606 100644 --- a/open-sse/services/usage.ts +++ b/open-sse/services/usage.ts @@ -1,3288 +1 @@ -/** - * Usage Fetcher - Get usage data from provider APIs - */ - -import { PROVIDERS } from "../config/constants.ts"; -import { - getAntigravityFetchAvailableModelsUrls, - ANTIGRAVITY_BASE_URLS, -} from "../config/antigravityUpstream.ts"; -import { - isUserCallableAntigravityModelId, - toClientAntigravityQuotaModelId, -} from "../config/antigravityModelAliases.ts"; -import { isUserCallableAgyModelId } from "../config/agyModels.ts"; -import { getGlmQuotaUrl } from "../config/glmProvider.ts"; -import { getGitHubCopilotInternalUserHeaders } from "../config/providerHeaderProfiles.ts"; -import { safePercentage } from "@/shared/utils/formatting"; -import { getDbInstance } from "@/lib/db/core"; -import { fetchBailianQuota, type BailianTripleWindowQuota } from "./bailianQuotaFetcher.ts"; -import { fetchDeepseekQuota, type DeepseekQuota } from "./deepseekQuotaFetcher.ts"; -import { fetchOpencodeQuota, type OpencodeTripleWindowQuota } from "./opencodeQuotaFetcher.ts"; -import { - applyAntigravityClientProfileHeaders, - getAntigravityBootstrapHeaders, - getAntigravityClientProfile, -} from "./antigravityClientProfile.ts"; -import { - antigravityUserAgent, - getAntigravityHeaders, - getAntigravityLoadCodeAssistMetadata, -} from "./antigravityHeaders.ts"; -import { - getAntigravityRemainingCredits, - updateAntigravityRemainingCredits, -} from "../executors/antigravity.ts"; -import { getCreditsMode } from "./antigravityCredits.ts"; -import { CLAUDE_CODE_VERSION, fetchClaudeBootstrap } from "../executors/claudeIdentity.ts"; -import { generateAntigravityRequestId, getAntigravitySessionId } from "./antigravityIdentity.ts"; -import { - extractCodeAssistOnboardTierId, - extractCodeAssistSubscriptionTier, -} from "./codeAssistSubscription.ts"; -import { sanitizeErrorMessage } from "../utils/error.ts"; - -// Quota / usage upstream URLs (overridable for testing or relays). -const CROF_USAGE_URL = process.env.OMNIROUTE_CROF_USAGE_URL ?? "https://crof.ai/usage_api/"; -const GEMINI_CLI_USAGE_URL = - process.env.OMNIROUTE_GEMINI_CLI_USAGE_URL ?? - "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist"; -const CODEWHISPERER_BASE_URL = - process.env.OMNIROUTE_CODEWHISPERER_BASE_URL ?? "https://codewhisperer.us-east-1.amazonaws.com"; - -// Antigravity API config (credentials from PROVIDERS via credential loader) -const ANTIGRAVITY_CONFIG = { - quotaApiUrls: getAntigravityFetchAvailableModelsUrls(), - loadProjectApiUrl: "https://daily-cloudcode-pa.sandbox.googleapis.com/v1internal:loadCodeAssist", - tokenUrl: "https://oauth2.googleapis.com/token", - get clientId() { - return PROVIDERS.antigravity.clientId; - }, - get clientSecret() { - return PROVIDERS.antigravity.clientSecret; - }, - get userAgent() { - return antigravityUserAgent(); - }, -}; - -// Codex (OpenAI) API config -const CODEX_CONFIG = { - usageUrl: "https://chatgpt.com/backend-api/wham/usage", -}; - -// Claude API config -const CLAUDE_CONFIG = { - oauthUsageUrl: "https://api.anthropic.com/api/oauth/usage", - usageUrl: "https://api.anthropic.com/v1/organizations/{org_id}/usage", - settingsUrl: "https://api.anthropic.com/v1/settings", - apiVersion: "2023-06-01", -}; - -// Kimi Coding API config -const KIMI_CONFIG = { - baseUrl: "https://api.kimi.com/coding/v1", - usageUrl: "https://api.kimi.com/coding/v1/usages", - apiVersion: "2023-06-01", -}; - -const NANOGPT_CONFIG = { - usageUrl: "https://nano-gpt.com/api/subscription/v1/usage", -}; - -const OPENCODE_GO_QUOTA_URL = - process.env.OMNIROUTE_OPENCODE_GO_QUOTA_URL ?? "https://api.z.ai/api/monitor/usage/quota/limit"; -const OPENCODE_GO_QUOTA_TOTALS = { - session: 12, - weekly: 30, - mcp_monthly: 60, -} as const; -const OPENCODE_GO_QUOTA_ORDER = ["session", "weekly", "mcp_monthly"] as const; -type OpenCodeGoQuotaName = (typeof OPENCODE_GO_QUOTA_ORDER)[number]; - -// Cursor dashboard usage API config -// The endpoint that powers https://cursor.com/dashboard/spending. Validates the WorkOS -// session via the WorkosCursorSessionToken cookie (format: `${userId}::${jwt}`) and -// rejects requests without a matching Origin/Referer (Invalid origin for state-changing request). -const CURSOR_USAGE_CONFIG = { - usageUrl: "https://cursor.com/api/dashboard/get-current-period-usage", - origin: "https://cursor.com", - referer: "https://cursor.com/dashboard/spending", - userAgent: - "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36", -}; - -const MINIMAX_USAGE_CONFIG = { - minimax: { - usageUrls: [ - "https://www.minimax.io/v1/token_plan/remains", - "https://api.minimax.io/v1/api/openplatform/coding_plan/remains", - ], - }, - "minimax-cn": { - usageUrls: [ - "https://www.minimaxi.com/v1/api/openplatform/coding_plan/remains", - "https://api.minimaxi.com/v1/api/openplatform/coding_plan/remains", - ], - }, -} as const; - -type JsonRecord = Record; -type UsageQuota = { - used: number; - total: number; - remaining?: number; - remainingPercentage?: number; - resetAt: string | null; - unlimited: boolean; - /** - * True when the upstream provider reported the remaining fraction. False - * means the API didn't include the field and the 0 value here is a sentinel, - * NOT a confirmed-exhausted state. Antigravity-specific. - */ - fractionReported?: boolean; - quotaSource?: "retrieveUserQuota" | "fetchAvailableModels" | "localUsageHistory"; - displayName?: string; - details?: Array<{ - name: string; - used: number; - }>; - currency?: string; - grantedBalance?: number; - toppedUpBalance?: number; -}; -type UsageProviderConnection = JsonRecord & { - id?: string; - provider?: string; - accessToken?: string; - apiKey?: string; - providerSpecificData?: JsonRecord; - projectId?: string; - email?: string; -}; -type SubscriptionCacheEntry = { - data: unknown; - fetchedAt: number; -}; - -function toRecord(value: unknown): JsonRecord { - return value && typeof value === "object" && !Array.isArray(value) ? (value as JsonRecord) : {}; -} - -function toNumber(value: unknown, fallback = 0): number { - const parsed = - typeof value === "number" - ? value - : typeof value === "string" && value.trim().length > 0 - ? Number(value) - : Number.NaN; - return Number.isFinite(parsed) ? parsed : fallback; -} - -function toPercentage(value: unknown): number { - return Math.max(0, Math.min(100, toNumber(value, 0))); -} - -function toTitleCase(value: string): string { - return value - .trim() - .split(/[\s_-]+/) - .filter(Boolean) - .map((part) => part.charAt(0).toUpperCase() + part.slice(1).toLowerCase()) - .join(" "); -} - -function getGlmTokenQuotaName( - limit: JsonRecord, - existingQuotas: Record -): string { - const unit = toNumber(limit.unit, 0); - const number = toNumber(limit.number, 0); - - if (unit === 3 && number === 5) return "session"; - if ((unit === 4 && number === 7) || (unit === 3 && number >= 24 * 7)) return "weekly"; - - return existingQuotas.session ? "weekly" : "session"; -} - -function getGlmQuotaDisplayName(quotaName: string): string { - if (quotaName === "session") return "5 Hours Quota"; - if (quotaName === "weekly") return "Weekly Quota"; - return quotaName; -} - -function getOpenCodeGoTokenQuotaName( - limit: JsonRecord, - existingQuotas: Record -): "session" | "weekly" { - const unit = toNumber(limit.unit, 0); - const number = toNumber(limit.number, 0); - - if (unit === 3 && number === 5) return "session"; - if (unit === 6 && number === 1) return "weekly"; - if ((unit === 4 && number === 7) || (unit === 3 && number >= 24 * 7)) return "weekly"; - - return existingQuotas.session ? "weekly" : "session"; -} - -function getOpenCodeGoQuotaDisplayName(quotaName: OpenCodeGoQuotaName): string { - if (quotaName === "session") return "5-hour rolling"; - if (quotaName === "weekly") return "Weekly"; - return "Monthly"; -} - -function normalizeOpenCodeGoQuotaToken(apiKey: string): string { - return apiKey.trim().replace(/^Bearer\s+/i, ""); -} - -function buildOpenCodeGoDollarQuota( - quotaName: OpenCodeGoQuotaName, - percentage: unknown, - resetAt: string | null, - usedOverride?: unknown, - details?: UsageQuota["details"] -): UsageQuota { - const total = OPENCODE_GO_QUOTA_TOTALS[quotaName]; - const percentUsed = toPercentage(percentage); - const rawUsed = toNumber(usedOverride, Number.NaN); - const used = roundCurrency( - Number.isFinite(rawUsed) ? Math.max(0, Math.min(total, rawUsed)) : (total * percentUsed) / 100 - ); - const remaining = roundCurrency(Math.max(0, total - used)); - const remainingPercentage = - total > 0 - ? clampPercentage(Math.round((remaining / total) * 100)) - : clampPercentage(100 - percentUsed); - - return { - used, - total, - remaining, - remainingPercentage, - resetAt, - unlimited: false, - displayName: getOpenCodeGoQuotaDisplayName(quotaName), - currency: "USD", - details, - }; -} - -function orderOpenCodeGoQuotas(quotas: Record): Record { - const ordered: Record = {}; - - for (const key of OPENCODE_GO_QUOTA_ORDER) { - if (quotas[key]) ordered[key] = quotas[key]; - } - - for (const [key, quota] of Object.entries(quotas)) { - if (!ordered[key]) ordered[key] = quota; - } - - return ordered; -} - -function getFieldValue(source: unknown, snakeKey: string, camelKey: string): unknown { - const obj = toRecord(source); - return obj[snakeKey] ?? obj[camelKey] ?? null; -} - -function clampPercentage(value: number): number { - return Math.max(0, Math.min(100, value)); -} - -function roundCurrency(value: number): number { - return Math.round(value * 100) / 100; -} - -function toDisplayLabel(value: string): string { - return value - .replace(/^copilot[_\s-]*/i, "") - .split(/[\s_-]+/) - .filter(Boolean) - .map((part) => { - if (/^pro\+$/i.test(part)) return "Pro+"; - if (/^[a-z]{2,}$/.test(part)) - return part.charAt(0).toUpperCase() + part.slice(1).toLowerCase(); - return part; - }) - .join(" ") - .trim(); -} - -function shouldDisplayGitHubQuota(quota: UsageQuota | null): quota is UsageQuota { - if (!quota) return false; - if (quota.unlimited && quota.total <= 0) return false; - return quota.total > 0 || quota.remainingPercentage !== undefined; -} - -function pickFirstNonEmptyString(...values: unknown[]): string | undefined { - for (const value of values) { - if (typeof value !== "string") continue; - const trimmed = value.trim(); - if (trimmed) return trimmed; - } - return undefined; -} - -function inferMiniMaxPlanLabelFromTotals(models: JsonRecord[]): string | null { - const maxSessionTotal = models.reduce( - (maxTotal, model) => Math.max(maxTotal, getMiniMaxSessionTotal(model)), - 0 - ); - - if (maxSessionTotal >= 15_000) return "Max"; - if (maxSessionTotal >= 4_500) return "Plus"; - if (maxSessionTotal >= 1_500) return "Starter"; - return null; -} - -function getMiniMaxPlanLabel(payload: JsonRecord, models: JsonRecord[] = []): string { - const raw = pickFirstNonEmptyString( - getFieldValue(payload, "current_subscribe_title", "currentSubscribeTitle"), - getFieldValue(payload, "plan_name", "planName"), - getFieldValue(payload, "plan", "plan"), - getFieldValue(payload, "current_plan_title", "currentPlanTitle"), - getFieldValue(payload, "combo_title", "comboTitle") - ); - - if (!raw) return inferMiniMaxPlanLabelFromTotals(models) || "Coding Plan"; - - const cleaned = raw - .replace(/^minimax\s+/i, "") - .replace(/\bcoding\s+plan\b/gi, "") - .replace(/\s{2,}/g, " ") - .trim(); - - return cleaned || inferMiniMaxPlanLabelFromTotals(models) || "Coding Plan"; -} - -function getClaudePlanLabel(...candidates: Array): string | null { - for (const candidate of candidates) { - if (typeof candidate !== "string") continue; - const trimmed = candidate.trim(); - if ( - !trimmed || - trimmed.toLowerCase() === "claude code" || - trimmed.toLowerCase() === "unknown" - ) { - continue; - } - return trimmed; - } - return null; -} - -function createQuotaFromUsage( - usedValue: unknown, - totalValue: unknown, - resetValue: unknown -): UsageQuota { - const total = Math.max(0, toNumber(totalValue, 0)); - const used = total > 0 ? Math.min(Math.max(0, toNumber(usedValue, 0)), total) : 0; - const remaining = total > 0 ? Math.max(total - used, 0) : 0; - - return { - used, - total, - remaining, - remainingPercentage: total > 0 ? clampPercentage((remaining / total) * 100) : 0, - resetAt: parseResetTime(resetValue), - unlimited: false, - }; -} - -function getMiniMaxQuotaResetAt( - model: JsonRecord, - capturedAtMs: number, - remainsTimeSnakeKey: string, - remainsTimeCamelKey: string, - endTimeSnakeKey: string, - endTimeCamelKey: string -): string | null { - const remainsMs = toNumber(getFieldValue(model, remainsTimeSnakeKey, remainsTimeCamelKey), 0); - if (remainsMs > 0) { - return new Date(capturedAtMs + remainsMs).toISOString(); - } - - return parseResetTime(getFieldValue(model, endTimeSnakeKey, endTimeCamelKey)); -} - -function isMiniMaxTextQuotaModel(modelName: string): boolean { - const normalized = modelName.trim().toLowerCase(); - return ( - normalized.startsWith("minimax-m") || - normalized.startsWith("coding-plan") || - // MiniMax Coding Plan surfaces the text/coding quota under model "general" - // (media buckets like "video"/"image"/"music" are excluded). - normalized === "general" - ); -} - -function getMiniMaxSessionTotal(model: JsonRecord): number { - return Math.max( - 0, - toNumber(getFieldValue(model, "current_interval_total_count", "currentIntervalTotalCount"), 0) - ); -} - -function getMiniMaxWeeklyTotal(model: JsonRecord): number { - return Math.max( - 0, - toNumber(getFieldValue(model, "current_weekly_total_count", "currentWeeklyTotalCount"), 0) - ); -} - -function pickMiniMaxRepresentativeModel( - models: JsonRecord[], - getTotal: (model: JsonRecord) => number -): JsonRecord | null { - const withQuota = models.filter((model) => getTotal(model) > 0); - const pool = withQuota.length > 0 ? withQuota : models; - if (pool.length === 0) return null; - - return pool.reduce((best, current) => (getTotal(current) > getTotal(best) ? current : best)); -} - -function createMiniMaxQuotaFromCount( - total: number, - count: number, - resetAt: string | null, - countMeansRemaining: boolean -): UsageQuota { - const used = countMeansRemaining ? Math.max(total - count, 0) : count; - return createQuotaFromUsage(used, total, resetAt); -} - -/** - * MiniMax Coding Plan exposes per-window remaining as a 0–100 percent - * (`current_interval_remaining_percent` / `current_weekly_remaining_percent`) - * with zero request counts. Read it defensively (string-encoded numbers ok). - */ -function getMiniMaxRemainingPercent( - model: JsonRecord, - snakeKey: string, - camelKey: string -): number | null { - const raw = getFieldValue(model, snakeKey, camelKey); - if (raw === null || raw === undefined || raw === "") return null; - const parsed = toNumber(raw, NaN); - return Number.isFinite(parsed) ? Math.max(0, Math.min(100, parsed)) : null; -} - -/** Build a 0–100 percent-based window quota (used = 100 − remaining). */ -function createMiniMaxQuotaFromPercent( - remainingPercent: number, - resetAt: string | null -): UsageQuota { - const clamped = Math.max(0, Math.min(100, remainingPercent)); - return createQuotaFromUsage(100 - clamped, 100, resetAt); -} - -/** - * Build one MiniMax usage window (session or weekly) from the representative - * model. Token Plan keys report request counts (`*_total_count`); Coding Plan - * keys report zero counts and a `*_remaining_percent` instead — fall back to - * that so the Coding Plan still surfaces a quota. The percent signal is keyed - * off "counts == 0 + percent present", NOT the endpoint URL, because the - * `token_plan/remains` and `coding_plan/remains` endpoints return identical - * Coding-Plan payloads for a Coding Plan key. - */ -function buildMiniMaxWindow( - models: JsonRecord[], - getTotal: (model: JsonRecord) => number, - usageCountKeys: [string, string], - percentKeys: [string, string], - resetKeys: [string, string, string, string], - capturedAtMs: number, - countMeansRemaining: boolean -): UsageQuota | null { - const model = pickMiniMaxRepresentativeModel(models, getTotal); - if (!model) return null; - - const resetAt = getMiniMaxQuotaResetAt(model, capturedAtMs, ...resetKeys); - const total = getTotal(model); - - if (total > 0) { - const count = Math.max(0, toNumber(getFieldValue(model, ...usageCountKeys), 0)); - return createMiniMaxQuotaFromCount(total, count, resetAt, countMeansRemaining); - } - - const remainingPercent = getMiniMaxRemainingPercent(model, ...percentKeys); - return remainingPercent !== null - ? createMiniMaxQuotaFromPercent(remainingPercent, resetAt) - : null; -} - -function getMiniMaxAuthErrorMessage(message: string): string { - const normalized = message.toLowerCase(); - if ( - normalized.includes("token plan") || - normalized.includes("coding plan") || - normalized.includes("active period") || - normalized.includes("invalid api key") || - normalized.includes("invalid key") || - normalized.includes("subscription") - ) { - return "MiniMax Token Plan API key invalid or inactive. Use an active Token Plan key."; - } - - return "MiniMax access denied. Confirm the key is an active Token Plan API key."; -} - -function getMiniMaxErrorSummary(status: number, message: string): string { - const compact = message.replace(/\s+/g, " ").trim(); - if (!compact) { - return `MiniMax usage endpoint error (${status}).`; - } - if (compact.length <= 160) { - return `MiniMax usage endpoint error (${status}): ${compact}`; - } - return `MiniMax usage endpoint error (${status}): ${compact.slice(0, 157)}...`; -} - -async function getMiniMaxUsage(apiKey: string, provider: "minimax" | "minimax-cn") { - if (!apiKey) { - return { message: "MiniMax API key not available. Add a Token Plan API key." }; - } - - const usageUrls = MINIMAX_USAGE_CONFIG[provider].usageUrls; - let lastErrorMessage = ""; - - for (let index = 0; index < usageUrls.length; index += 1) { - const usageUrl = usageUrls[index]; - const canFallback = index < usageUrls.length - 1; - - try { - const response = await fetch(usageUrl, { - method: "GET", - headers: { - Authorization: `Bearer ${apiKey}`, - Accept: "application/json", - "Content-Type": "application/json", - }, - }); - - const rawText = await response.text(); - let payload: JsonRecord = {}; - if (rawText) { - try { - payload = toRecord(JSON.parse(rawText)); - } catch { - payload = {}; - } - } - - const baseResp = toRecord(getFieldValue(payload, "base_resp", "baseResp")); - const apiStatusCode = toNumber(getFieldValue(baseResp, "status_code", "statusCode"), 0); - const apiStatusMessage = String( - getFieldValue(baseResp, "status_msg", "statusMsg") ?? "" - ).trim(); - const combinedMessage = `${apiStatusMessage} ${rawText}`.trim(); - const authLikeStatusMessage = - /token plan|coding plan|invalid api key|invalid key|unauthorized|inactive/i; - - if ( - response.status === 401 || - response.status === 403 || - apiStatusCode === 1004 || - authLikeStatusMessage.test(apiStatusMessage) - ) { - return { message: getMiniMaxAuthErrorMessage(apiStatusMessage || combinedMessage) }; - } - - if (!response.ok) { - lastErrorMessage = getMiniMaxErrorSummary(response.status, combinedMessage); - if ( - (response.status === 404 || response.status === 405 || response.status >= 500) && - canFallback - ) { - continue; - } - return { message: `MiniMax connected. ${lastErrorMessage}` }; - } - - if (rawText && Object.keys(payload).length === 0) { - return { message: "MiniMax connected. Unable to parse usage response." }; - } - - if (apiStatusCode !== 0) { - if (apiStatusMessage) { - return { message: `MiniMax connected. ${apiStatusMessage}` }; - } - return { message: "MiniMax connected. Upstream quota API returned an error." }; - } - - const capturedAtMs = Date.now(); - const modelRemains = getFieldValue(payload, "model_remains", "modelRemains"); - const allModels = Array.isArray(modelRemains) - ? modelRemains.map((item) => toRecord(item)) - : []; - const textModels = allModels.filter((model) => { - const modelName = String(getFieldValue(model, "model_name", "modelName") ?? ""); - return isMiniMaxTextQuotaModel(modelName); - }); - - if (textModels.length === 0) { - return { message: "MiniMax connected. No text quota data was returned." }; - } - - const countMeansRemaining = usageUrl.includes("/coding_plan/remains"); - const quotas: Record = {}; - - const sessionQuota = buildMiniMaxWindow( - textModels, - getMiniMaxSessionTotal, - ["current_interval_usage_count", "currentIntervalUsageCount"], - ["current_interval_remaining_percent", "currentIntervalRemainingPercent"], - ["remains_time", "remainsTime", "end_time", "endTime"], - capturedAtMs, - countMeansRemaining - ); - if (sessionQuota) { - quotas["session (5h)"] = sessionQuota; - } - - const weeklyQuota = buildMiniMaxWindow( - textModels, - getMiniMaxWeeklyTotal, - ["current_weekly_usage_count", "currentWeeklyUsageCount"], - ["current_weekly_remaining_percent", "currentWeeklyRemainingPercent"], - ["weekly_remains_time", "weeklyRemainsTime", "weekly_end_time", "weeklyEndTime"], - capturedAtMs, - countMeansRemaining - ); - if (weeklyQuota) { - quotas["weekly (7d)"] = weeklyQuota; - } - - if (Object.keys(quotas).length === 0) { - return { message: "MiniMax connected. Unable to extract text quota usage." }; - } - - return { plan: getMiniMaxPlanLabel(payload, textModels), quotas }; - } catch (error) { - lastErrorMessage = (error as Error).message; - if (!canFallback) { - break; - } - } - } - - return { - message: lastErrorMessage - ? `MiniMax connected. Unable to fetch usage: ${lastErrorMessage}` - : "MiniMax connected. Unable to fetch usage.", - }; -} - -// CrofAI surfaces a tiny endpoint with two signals: -// GET https://crof.ai/usage_api/ → { usable_requests: number|null, credits: number } -// `usable_requests` is the daily request bucket on a subscription plan; `null` -// for pay-as-you-go. `credits` is the USD credit balance. We surface both as -// quotas so the Limits & Quotas page can render whichever the account uses. -async function getCrofUsage(apiKey: string) { - if (!apiKey) { - return { message: "CrofAI API key not available. Add a key to view usage." }; - } - - let response: Response; - try { - response = await fetch(CROF_USAGE_URL, { - method: "GET", - headers: { - Authorization: `Bearer ${apiKey}`, - Accept: "application/json", - }, - }); - } catch (error) { - return { message: `CrofAI connected. Unable to fetch usage: ${(error as Error).message}` }; - } - - const rawText = await response.text(); - - if (response.status === 401 || response.status === 403) { - return { message: "CrofAI connected. The API key was rejected by /usage_api/." }; - } - - if (!response.ok) { - return { message: `CrofAI connected. /usage_api/ returned HTTP ${response.status}.` }; - } - - let payload: JsonRecord = {}; - if (rawText) { - try { - payload = toRecord(JSON.parse(rawText)); - } catch { - return { message: "CrofAI connected. Unable to parse /usage_api/ response." }; - } - } - - const usableRequestsRaw = payload["usable_requests"]; - const usableRequests = - usableRequestsRaw === null || usableRequestsRaw === undefined - ? null - : toNumber(usableRequestsRaw, 0); - const credits = toNumber(payload["credits"], 0); - - const quotas: Record = {}; - - if (usableRequests !== null) { - // CrofAI's /usage_api/ returns only the remaining count; the daily - // allotment is not exposed. CrofAI Pro plan = 1,000 requests/day per - // their pricing page, so use that as the baseline total. If the user - // is on a plan with a higher cap we widen the total to whatever they - // currently report so we never compute a negative `used`. - // Without this, total=0 makes the dashboard's percentage formula read - // 0% (interpreted as "depleted" → red) even on a fresh bucket. - const CROF_DAILY_BASELINE = 1000; - const remaining = Math.max(0, usableRequests); - const total = Math.max(CROF_DAILY_BASELINE, remaining); - const used = Math.max(0, total - remaining); - - // CrofAI also does not return a reset timestamp and the docs only say - // "requests left today". The Crof.ai dashboard shows the daily bucket - // resetting at ~05:00 UTC (verified against the live countdown on - // 2026-04-25), so synthesize the next 05:00 UTC instant to match. - // Swap for a real field if Crof ever exposes one. - const now = new Date(); - const RESET_HOUR_UTC = 5; - const todayResetMs = Date.UTC( - now.getUTCFullYear(), - now.getUTCMonth(), - now.getUTCDate(), - RESET_HOUR_UTC - ); - const nextResetMs = - todayResetMs > now.getTime() ? todayResetMs : todayResetMs + 24 * 60 * 60 * 1000; - const nextResetIso = new Date(nextResetMs).toISOString(); - - quotas["Requests Today"] = { - used, - total, - remaining, - resetAt: nextResetIso, - unlimited: false, - displayName: `Requests Today: ${remaining} left`, - }; - } - - // Credits are an open balance — render as unlimited so the UI shows the - // dollar value rather than a misleading 0/0 bar. - quotas["Credits"] = { - used: 0, - total: 0, - remaining: 0, - resetAt: null, - unlimited: true, - displayName: `Credits: $${credits.toFixed(4)}`, - }; - - return { quotas }; -} - -const GLM_QUOTA_ORDER = ["5 Hours Quota", "Weekly Quota", "Monthly Tools", "Tokens", "Time Limit"]; - -function getGlmQuotaLabel(type: unknown, unit: unknown): string | null { - const normalized = typeof type === "string" ? type.trim().toUpperCase() : ""; - const unitValue = toNumber(unit, -1); - - switch (normalized) { - case "TOKENS_LIMIT": - case "TOKEN_LIMIT": - if (unitValue === 3) return "5 Hours Quota"; - if (unitValue === 6) return "Weekly Quota"; - return "Tokens"; - case "TIME_LIMIT": - case "TIME_USAGE_LIMIT": - if (unitValue === 5) return "Monthly Tools"; - return "Time Limit"; - default: - return null; - } -} - -function orderGlmQuotas(quotas: Record): Record { - const ordered: Record = {}; - - for (const key of GLM_QUOTA_ORDER) { - if (quotas[key]) ordered[key] = quotas[key]; - } - - for (const [key, quota] of Object.entries(quotas)) { - if (!ordered[key]) ordered[key] = quota; - } - - return ordered; -} - -/** - * Remaining-percentage for a GLM/z.ai TIME_LIMIT ("Monthly") quota. With an absolute - * monthly cap (`total > 0`) it is `remaining / total`. Coding plans that have no - * monthly cap (only 5-hour windows) report `total = 0`; in that case fall back to the - * percentage-derived remaining so "no monthly cap" renders as full/100% instead of a - * misleading 0% (#3580). - */ -export function glmMonthlyRemainingPercentage(total: number, remaining: number): number { - if (total > 0) { - return Math.max(0, Math.min(100, Math.round((remaining / total) * 100))); - } - return Math.max(0, Math.min(100, Math.round(remaining))); -} - -async function getGlmUsage(apiKey: string, providerSpecificData?: Record) { - if (!apiKey) { - return { message: "API key not available. Add a coding plan API key to view usage." }; - } - - const quotaUrl = getGlmQuotaUrl(providerSpecificData); - - const res = await fetch(quotaUrl, { - headers: { - Authorization: `Bearer ${apiKey}`, - Accept: "application/json", - }, - }); - - if (!res.ok) { - if (res.status === 401) throw new Error("Invalid API key"); - throw new Error(`GLM quota API error (${res.status})`); - } - - const json = await res.json(); - if (toNumber(json.code, 200) === 401 || json.success === false) { - throw new Error("Invalid API key"); - } - - const data = toRecord(json.data); - const limits: unknown[] = Array.isArray(data.limits) ? data.limits : []; - const quotas: Record = {}; - - for (const limit of limits) { - const src = toRecord(limit); - const type = String(src.type || "").toUpperCase(); - const resetMs = toNumber(src.nextResetTime, 0); - const resetAt = resetMs > 0 ? new Date(resetMs).toISOString() : null; - - if (type === "TOKENS_LIMIT") { - const quotaName = getGlmTokenQuotaName(src, quotas); - const usedPercent = toPercentage(src.percentage); - const remaining = Math.max(0, 100 - usedPercent); - - quotas[quotaName] = { - used: usedPercent, - total: 100, - remaining, - remainingPercentage: remaining, - resetAt, - displayName: getGlmQuotaDisplayName(quotaName), - details: Array.isArray(src.models) - ? (src.models as unknown[]).map((m) => { - const modelInfo = toRecord(m); - return { - name: String(modelInfo.model || ""), - used: toNumber(modelInfo.percentage, 0), - }; - }) - : [], - unlimited: false, - }; - continue; - } - - if (type === "TIME_LIMIT") { - const total = toNumber(src.usage, toNumber(src.total, 0)); - const remaining = toNumber(src.remaining, Math.max(0, 100 - toPercentage(src.percentage))); - const used = toNumber(src.currentValue, Math.max(0, total - remaining)); - const remainingPercentage = glmMonthlyRemainingPercentage(total, remaining); - - quotas["mcp_monthly"] = { - used, - total, - remaining, - remainingPercentage, - resetAt, - unlimited: false, - displayName: "Monthly", - details: Array.isArray(src.usageDetails) - ? src.usageDetails.map((item) => { - const detail = toRecord(item); - return { - name: String(detail.modelCode || detail.name || "usage"), - used: toNumber(detail.usage, 0), - }; - }) - : undefined, - }; - } - } - - const levelRaw = - typeof data.planName === "string" - ? data.planName - : typeof data.level === "string" - ? data.level - : ""; - const plan = levelRaw ? toTitleCase(levelRaw.replace(/\s*plan$/i, "")) : null; - - return { plan, quotas: orderGlmQuotas(quotas) }; -} - -async function getOpenCodeGoUsage(apiKey: string) { - const token = normalizeOpenCodeGoQuotaToken(apiKey); - - if (!token) { - return { message: "API key not available. Add an OpenCode Go API key to view usage." }; - } - - const res = await fetch(OPENCODE_GO_QUOTA_URL, { - headers: { - Authorization: token, - "Accept-Language": "en-US,en", - "Content-Type": "application/json", - Accept: "application/json", - }, - }); - - if (!res.ok) { - if (res.status === 401 || res.status === 403) { - return { - message: "OpenCode Go quota endpoint rejected this API key. Chat requests still work.", - }; - } - return { message: `OpenCode Go quota API error (${res.status})` }; - } - - let json: unknown; - try { - json = await res.json(); - } catch { - return { message: "OpenCode Go quota response parsing failed." }; - } - - const code = toNumber((json as Record).code, 200); - if (code === 401 || code === 403 || (json as Record).success === false) { - return { - message: "OpenCode Go quota endpoint rejected this API key. Chat requests still work.", - }; - } - - const data = toRecord((json as Record).data); - const limits: unknown[] = Array.isArray(data.limits) ? data.limits : []; - const quotas: Record = {}; - - for (const limit of limits) { - const src = toRecord(limit); - const type = String(src.type || "").toUpperCase(); - const resetAt = parseResetTime(src.nextResetTime); - - if (type === "TOKENS_LIMIT" || type === "TOKEN_LIMIT") { - const quotaName = getOpenCodeGoTokenQuotaName(src, quotas); - - quotas[quotaName] = buildOpenCodeGoDollarQuota( - quotaName, - src.percentage, - resetAt, - undefined, - Array.isArray(src.models) - ? (src.models as unknown[]).map((model) => { - const modelInfo = toRecord(model); - return { - name: String(modelInfo.model || modelInfo.modelCode || "usage"), - used: toNumber(modelInfo.percentage, 0), - }; - }) - : undefined - ); - continue; - } - - if (type === "TIME_LIMIT" || type === "TIME_USAGE_LIMIT") { - quotas.mcp_monthly = buildOpenCodeGoDollarQuota( - "mcp_monthly", - src.percentage, - resetAt, - src.currentValue, - Array.isArray(src.usageDetails) - ? src.usageDetails.map((item) => { - const detail = toRecord(item); - return { - name: String(detail.modelCode || detail.name || "usage"), - used: toNumber(detail.usage, 0), - }; - }) - : undefined - ); - } - } - - const levelRaw = - typeof data.planName === "string" - ? data.planName - : typeof data.level === "string" - ? data.level - : ""; - const planLabel = toTitleCase(levelRaw.replace(/\s*plan$/i, "")); - const plan = planLabel - ? /^opencode\s+go\b/i.test(planLabel) - ? planLabel - : `OpenCode Go ${planLabel}` - : null; - - return { plan, quotas: orderOpenCodeGoQuotas(quotas) }; -} - -/** - * Bailian (Alibaba Coding Plan) Usage - * Fetches triple-window quota (5h, weekly, monthly) and returns worst-case. - */ -async function getBailianCodingPlanUsage( - connectionId: string, - apiKey: string, - providerSpecificData?: Record -) { - try { - const connection = { apiKey, providerSpecificData }; - const quota = await fetchBailianQuota(connectionId, connection); - - if (!quota) { - return { message: "Bailian Coding Plan connected. Unable to fetch quota." }; - } - - const bailianQuota = quota as BailianTripleWindowQuota; - const used = bailianQuota.used; - const total = bailianQuota.total; - const remaining = Math.max(0, total - used); - const remainingPercentage = Math.round(remaining); - - return { - plan: "Alibaba Coding Plan", - used, - total, - remaining, - remainingPercentage, - resetAt: bailianQuota.resetAt, - unlimited: false, - displayName: "Alibaba Coding Plan", - }; - } catch (error) { - return { message: `Bailian Coding Plan error: ${(error as Error).message}` }; - } -} - -/** - * DeepSeek Usage - * Fetches balance from the DeepSeek balance API. - * Returns all balances (USD and CNY) as "credits" for credits-style UI display. - */ -async function getDeepseekUsage(connectionId: string, apiKey: string) { - try { - const connection = { apiKey }; - const quota = await fetchDeepseekQuota(connectionId, connection); - - if (!quota) { - return { message: "DeepSeek API key not available. Add a key to view usage." }; - } - - const deepseekQuota = quota as DeepseekQuota; - const { balances, isAvailable, limitReached } = deepseekQuota; - - const quotas: Record = {}; - - // Show all balances as credits-style entries (e.g., credits_usd, credits_cny) - // The UI will display them as "🪙 Balance (USD) $50.00" - for (const balanceInfo of balances) { - const key = `credits_${balanceInfo.currency.toLowerCase()}`; - quotas[key] = { - used: 0, - total: 0, - remaining: balanceInfo.balance, - remainingPercentage: 100, - resetAt: null, - unlimited: true, - currency: balanceInfo.currency, - grantedBalance: balanceInfo.grantedBalance, - toppedUpBalance: balanceInfo.toppedUpBalance, - }; - } - - const plan = isAvailable ? "DeepSeek" : "DeepSeek (Insufficient Balance)"; - - return { - plan, - quotas, - isAvailable, - limitReached, - }; - } catch (error) { - return { message: `DeepSeek error: ${(error as Error).message}` }; - } -} - -// Xiaomi MiMo Token Plan monthly limit (tokens). Keep in sync with the -// "xiaomi-mimo" preset in src/lib/quota/planRegistry.ts. -const XIAOMI_MIMO_MONTHLY_TOKEN_LIMIT = 4_100_000_000; - -/** - * Xiaomi MiMo — SELF-TRACKED monthly quota. - * - * Xiaomi exposes plan usage only behind the console session cookie (the API key - * cannot reach the `tokenPlan/usage` endpoint), so there is no upstream usage - * API to call. Instead we count the tokens OmniRoute itself routed to this - * connection in the current UTC month (from `usage_history`) and compare them - * to the known Token Plan monthly limit. This reflects only traffic that went - * through OmniRoute, not the provider's own dashboard figure. - */ -async function getXiaomiMimoUsage(connectionId: string) { - if (!connectionId) { - return { message: "Xiaomi MiMo: connection id unavailable for self-tracked quota." }; - } - try { - const { getMonthlyProviderTokensForConnection } = await import("@/lib/usage/usageStats"); - const used = getMonthlyProviderTokensForConnection("xiaomi-mimo", connectionId); - const total = XIAOMI_MIMO_MONTHLY_TOKEN_LIMIT; - const now = new Date(); - const resetAt = new Date( - Date.UTC(now.getUTCFullYear(), now.getUTCMonth() + 1, 1) - ).toISOString(); - return { - plan: "Xiaomi MiMo Token Plan (OmniRoute-tracked)", - quotas: { - monthly: createQuotaFromUsage(used, total, resetAt), - }, - }; - } catch (error) { - return { message: `Xiaomi MiMo self-tracked usage error: ${(error as Error).message}` }; - } -} - -/** - * OpenCode Go / OpenCode / OpenCode Zen Usage - * Delegates to the dedicated opencodeQuotaFetcher and shapes the result into - * the standard `{ plan, quotas }` usage response expected by the limits page. - * - * Three rolling windows are surfaced: $12/5h, $30/wk, $60/mo. - */ -async function getOpencodeUsage(connectionId: string, apiKey: string) { - if (!apiKey) { - return { message: "OpenCode API key not available. Add a key to view usage." }; - } - - try { - const quota = (await fetchOpencodeQuota(connectionId, { - apiKey, - })) as OpencodeTripleWindowQuota | null; - - if (!quota) { - return { message: "OpenCode connected. Unable to fetch quota data." }; - } - - const { window5h, windowWeekly, windowMonthly, limitReached } = quota; - - const quotas: Record = {}; - - // $12 / 5-hour rolling window - quotas["window_5h"] = { - used: window5h.percentUsed * 12, - total: 12, - remaining: (1 - window5h.percentUsed) * 12, - remainingPercentage: (1 - window5h.percentUsed) * 100, - resetAt: window5h.resetAt, - unlimited: false, - displayName: "$12 / 5-hour", - currency: "USD", - }; - - // $30 / weekly window - quotas["window_weekly"] = { - used: windowWeekly.percentUsed * 30, - total: 30, - remaining: (1 - windowWeekly.percentUsed) * 30, - remainingPercentage: (1 - windowWeekly.percentUsed) * 100, - resetAt: windowWeekly.resetAt, - unlimited: false, - displayName: "$30 / week", - currency: "USD", - }; - - // $60 / monthly window - quotas["window_monthly"] = { - used: windowMonthly.percentUsed * 60, - total: 60, - remaining: (1 - windowMonthly.percentUsed) * 60, - remainingPercentage: (1 - windowMonthly.percentUsed) * 100, - resetAt: windowMonthly.resetAt, - unlimited: false, - displayName: "$60 / month", - currency: "USD", - }; - - return { - plan: "OpenCode Go", - quotas, - limitReached, - }; - } catch (error) { - return { message: `OpenCode error: ${sanitizeErrorMessage(error)}` }; - } -} - -/** - * NanoGPT Usage - * Fetches subscription-level quota from the NanoGPT API. - * Returns daily/weekly token limits and daily image limits for PRO accounts. - */ -async function getNanoGptUsage(apiKey: string) { - if (!apiKey) { - return { message: "NanoGPT API key not available. Add a key to view usage." }; - } - - try { - const res = await fetch(NANOGPT_CONFIG.usageUrl, { - headers: { Authorization: `Bearer ${apiKey}` }, - }); - - if (!res.ok) { - if (res.status === 401) return { message: "Invalid NanoGPT API key." }; - return { message: `NanoGPT quota API error (${res.status})` }; - } - - const data = toRecord(await res.json()); - const quotas: Record = {}; - - // active -> PRO, otherwise FREE - const plan = data.active ? "PRO" : "FREE"; - - if (data.active) { - // 1. Tokens limit - // dailyInputTokens if exists, else weeklyInputTokens - let tokenQuota = toRecord(data.dailyInputTokens); - let tokenLabel = "Daily Tokens"; - if (!tokenQuota.resetAt) { - const weeklyQuota = toRecord(data.weeklyInputTokens); - if (weeklyQuota.remaining !== undefined) { - tokenQuota = weeklyQuota; - tokenLabel = "Weekly Tokens"; - } - } - - if (tokenQuota.remaining !== undefined) { - const used = toNumber(tokenQuota.used, 0); - const remaining = toNumber(tokenQuota.remaining, 0); - const total = used + remaining; - quotas[tokenLabel] = { - used, - total, - remaining, - remainingPercentage: clampPercentage(100 - toNumber(tokenQuota.percentUsed, 0) * 100), - resetAt: parseResetTime(tokenQuota.resetAt), - unlimited: false, - }; - } - - // 2. Images limit - const imageQuota = toRecord(data.dailyImages); - if (imageQuota.remaining !== undefined) { - const used = toNumber(imageQuota.used, 0); - const remaining = toNumber(imageQuota.remaining, 0); - const total = used + remaining; - quotas["Daily Images"] = { - used, - total, - remaining, - remainingPercentage: clampPercentage(100 - toNumber(imageQuota.percentUsed, 0) * 100), - resetAt: parseResetTime(imageQuota.resetAt), - unlimited: false, - }; - } - - if (Object.keys(quotas).length === 0) { - return { plan, message: "NanoGPT connected, but no active limits found." }; - } - } - - return { plan, quotas }; - } catch (error) { - return { message: `NanoGPT connected. Unable to fetch usage: ${(error as Error).message}` }; - } -} - -/** - * Decode the `sub` claim of a Cursor JWT (the WorkOS user id). - * Returns null if the token is not a parseable JWT. - */ -function decodeCursorJwtSub(token: string): string | null { - if (!token || typeof token !== "string") return null; - const parts = token.split("."); - if (parts.length !== 3) return null; - try { - let payload = parts[1].replace(/-/g, "+").replace(/_/g, "/"); - while (payload.length % 4 !== 0) payload += "="; - const decoded = JSON.parse(Buffer.from(payload, "base64").toString("utf8")); - const sub = decoded?.sub; - return typeof sub === "string" && sub.length > 0 ? sub : null; - } catch { - return null; - } -} - -/** - * Cursor Pro Plan Usage - * Fetches current-billing-cycle spend from the cursor.com dashboard API and exposes three - * windows that mirror the cursor.com/dashboard/spending UI: Total / Auto + Composer / API. - */ -async function getCursorUsage(accessToken: string, providerSpecificData?: unknown) { - if (!accessToken) { - return { message: "Cursor access token missing. Re-import the connection from Cursor IDE." }; - } - - const storedUserId = (() => { - const raw = toRecord(providerSpecificData).userId; - return typeof raw === "string" && raw.length > 0 ? raw : null; - })(); - const userId = storedUserId || decodeCursorJwtSub(accessToken); - - if (!userId) { - return { - message: "Cursor token missing user id. Re-import the connection from Cursor IDE.", - }; - } - - try { - const response = await fetch(CURSOR_USAGE_CONFIG.usageUrl, { - method: "POST", - redirect: "manual", - headers: { - Cookie: `WorkosCursorSessionToken=${userId}::${accessToken}`, - Origin: CURSOR_USAGE_CONFIG.origin, - Referer: CURSOR_USAGE_CONFIG.referer, - "Content-Type": "application/json", - Accept: "application/json", - "User-Agent": CURSOR_USAGE_CONFIG.userAgent, - }, - body: "{}", - }); - - // 3xx redirect to WorkOS authkit means the session cookie was rejected. - if (response.status >= 300 && response.status < 400) { - return { - plan: "Cursor", - message: "Cursor session expired. Re-import the token from Cursor IDE.", - }; - } - - if (!response.ok) { - const errorText = (await response.text()).slice(0, 200); - if (response.status === 401 || response.status === 403) { - return { - plan: "Cursor", - message: "Cursor session unauthorized. Re-import the token from Cursor IDE.", - }; - } - return { - plan: "Cursor", - message: `Cursor usage endpoint error (${response.status}): ${errorText}`, - }; - } - - const data = toRecord(await response.json()); - const planUsage = toRecord(data.planUsage); - - if (Object.keys(planUsage).length === 0) { - return { - plan: "Cursor", - message: "Cursor connected. No active plan usage returned.", - }; - } - - const limitCents = Math.max(0, toNumber(planUsage.limit, 0)); - const totalSpendCents = Math.max(0, toNumber(planUsage.totalSpend, 0)); - const autoPercentUsed = clampPercentage(toNumber(planUsage.autoPercentUsed, 0)); - const apiPercentUsed = clampPercentage(toNumber(planUsage.apiPercentUsed, 0)); - const totalPercentUsed = clampPercentage(toNumber(planUsage.totalPercentUsed, 0)); - - // billingCycleEnd is a numeric-string in ms; coerce so parseResetTime sees a number. - const billingCycleEndMs = toNumber(data.billingCycleEnd, 0); - const resetAt = billingCycleEndMs > 0 ? parseResetTime(billingCycleEndMs) : null; - - // Convert cents → dollars rounded to 2 decimal places. - const toDollars = (cents: number) => Math.round(cents) / 100; - - const limitDollars = toDollars(limitCents); - const buildWindow = (percentUsed: number, usedCentsOverride?: number): UsageQuota => { - const usedCents = - typeof usedCentsOverride === "number" - ? usedCentsOverride - : Math.round((limitCents * percentUsed) / 100); - const used = toDollars(Math.min(usedCents, limitCents)); - const remaining = toDollars(Math.max(limitCents - Math.min(usedCents, limitCents), 0)); - return { - used, - total: limitDollars, - remaining, - remainingPercentage: clampPercentage(100 - percentUsed), - resetAt, - unlimited: false, - }; - }; - - const quotas: Record = { - Total: buildWindow(totalPercentUsed, totalSpendCents), - "Auto + Composer": buildWindow(autoPercentUsed), - API: buildWindow(apiPercentUsed), - }; - - return { - plan: "Cursor Pro", - quotas, - }; - } catch (error) { - return { - plan: "Cursor", - message: `Cursor connected. Unable to fetch usage: ${(error as Error).message}`, - }; - } -} - -/** - * Single source of truth for which providers have a `getUsageForProvider` - * implementation. Consumers like `genericQuotaFetcher.ts` reference this so - * the registration list can't drift from the switch statement below. - * - * If you add a new provider to the switch, add it here too. - */ -export const USAGE_FETCHER_PROVIDERS = [ - "github", - "gemini-cli", - "antigravity", - "agy", - "claude", - "codex", - "cursor", - "kiro", - "amazon-q", - "kimi-coding", - "qwen", - "qoder", - "glm", - "glm-cn", - "zai", - "glmt", - "opencode-go", - "minimax", - "minimax-cn", - "crof", - "bailian-coding-plan", - "nanogpt", - "deepseek", - "opencode", - "opencode-zen", - "xiaomi-mimo", -] as const; - -export type UsageFetcherProvider = (typeof USAGE_FETCHER_PROVIDERS)[number]; - -/** - * Get usage data for a provider connection - * @param {Object} connection - Provider connection with accessToken - * @returns {Promise} Usage data with quotas - */ -export async function getUsageForProvider( - connection: UsageProviderConnection, - options: { forceRefresh?: boolean } = {} -) { - const { id, provider, accessToken, apiKey, providerSpecificData, projectId, email } = connection; - - switch (provider) { - case "github": - return await getGitHubUsage(accessToken, providerSpecificData); - case "gemini-cli": - return await getGeminiUsage(accessToken, providerSpecificData, projectId); - case "antigravity": - case "agy": - return await getAntigravityUsage( - provider, - accessToken, - providerSpecificData, - projectId, - id, - options - ); - case "claude": - return await getClaudeUsage(accessToken); - case "codex": - return await getCodexUsage(accessToken, providerSpecificData); - case "cursor": - return await getCursorUsage(accessToken || "", providerSpecificData); - case "kiro": - case "amazon-q": - return await getKiroUsage(accessToken, providerSpecificData); - case "kimi-coding": - return await getKimiUsage(accessToken); - case "qwen": - return await getQwenUsage(accessToken, providerSpecificData); - case "qoder": - return await getQoderUsage(accessToken); - case "glm": - case "glm-cn": - case "zai": - case "glmt": - return await getGlmUsage(apiKey || "", { - ...(providerSpecificData || {}), - ...(provider === "glm-cn" ? { apiRegion: "china" } : {}), - }); - case "opencode-go": - return await getOpenCodeGoUsage(apiKey || ""); - case "minimax": - case "minimax-cn": - return await getMiniMaxUsage(apiKey || "", provider); - case "crof": - return await getCrofUsage(apiKey || ""); - case "bailian-coding-plan": - return await getBailianCodingPlanUsage(id || "", apiKey || "", providerSpecificData); - case "nanogpt": - return await getNanoGptUsage(apiKey || ""); - case "deepseek": - return await getDeepseekUsage(id || "", apiKey || ""); - case "opencode": - case "opencode-zen": - return await getOpencodeUsage(id || "", apiKey || ""); - case "xiaomi-mimo": - return await getXiaomiMimoUsage(id || ""); - default: - return { message: `Usage API not implemented for ${provider}` }; - } -} - -/** - * Parse reset date/time to ISO string - * Handles multiple formats: Unix timestamp (ms), ISO date string, etc. - */ -function parseResetTime(resetValue: unknown): string | null { - if (!resetValue) return null; - - try { - let date: Date; - if (resetValue instanceof Date) { - date = resetValue; - } else if (typeof resetValue === "number") { - date = new Date(resetValue < 1e12 ? resetValue * 1000 : resetValue); - } else if (typeof resetValue === "string") { - date = new Date(resetValue); - } else { - return null; - } - - // Epoch-zero (1970-01-01) means no scheduled reset — treat as null - if (date.getTime() <= 0) return null; - - return date.toISOString(); - } catch (error) { - return null; - } -} - -/** - * GitHub Copilot Usage - * Uses GitHub accessToken (not copilotToken) to call copilot_internal/user API - */ -async function getGitHubUsage(accessToken?: string, providerSpecificData?: JsonRecord) { - try { - if (!accessToken) { - throw new Error("No GitHub access token available. Please re-authorize the connection."); - } - - // copilot_internal/user API requires GitHub OAuth token, not copilotToken - const response = await fetch("https://api.github.com/copilot_internal/user", { - headers: getGitHubCopilotInternalUserHeaders(`token ${accessToken}`), - }); - - if (!response.ok) { - const error = await response.text(); - if (response.status === 401 || response.status === 403) { - return { - message: `GitHub token expired or permission denied. Please re-authenticate the connection.`, - }; - } - throw new Error(`GitHub API error: ${error}`); - } - - const data = await response.json(); - const dataRecord = toRecord(data); - - // Handle different response formats (paid vs free) - if (dataRecord.quota_snapshots) { - // Paid plan format - const snapshots = toRecord(dataRecord.quota_snapshots); - const resetAt = parseResetTime( - getFieldValue(dataRecord, "quota_reset_date", "quotaResetDate") - ); - const premiumQuota = formatGitHubQuotaSnapshot(snapshots.premium_interactions, resetAt); - const chatQuota = formatGitHubQuotaSnapshot(snapshots.chat, resetAt); - const completionsQuota = formatGitHubQuotaSnapshot(snapshots.completions, resetAt); - const quotas: Record = {}; - - if (shouldDisplayGitHubQuota(premiumQuota)) { - quotas.premium_interactions = premiumQuota; - } - if (shouldDisplayGitHubQuota(chatQuota)) { - quotas.chat = chatQuota; - } - if (shouldDisplayGitHubQuota(completionsQuota)) { - quotas.completions = completionsQuota; - } - - return { - plan: inferGitHubPlanName(dataRecord, premiumQuota), - resetDate: getFieldValue(dataRecord, "quota_reset_date", "quotaResetDate"), - quotas, - }; - } else if (dataRecord.monthly_quotas || dataRecord.limited_user_quotas) { - // Free/limited plan format. NOTE (#2876): the upstream field - // `limited_user_quotas[name]` is the *remaining* count for the month - // (it counts down toward 0 and resets on `limited_user_reset_date`), - // NOT the used count. The pre-3.8.6 implementation inverted this and - // showed "0% when not used / 100% when fully used" on the dashboard. - // Confirmed against three independent upstream parsers: - // - robinebers/openusage docs/providers/copilot.md (Free Tier table) - // - raycast/extensions agent-usage/src/copilot/fetcher.ts (inline comment) - // - looplj/axonhub frontend/src/components/quota-badges.tsx - const monthlyQuotas = toRecord(dataRecord.monthly_quotas); - const remainingQuotas = toRecord(dataRecord.limited_user_quotas); - const resetDate = getFieldValue( - dataRecord, - "limited_user_reset_date", - "limitedUserResetDate" - ); - const resetAt = parseResetTime(resetDate); - const quotas: Record = {}; - - const addLimitedQuota = (name: string) => { - const total = toNumber(getFieldValue(monthlyQuotas, name, name), 0); - if (total <= 0) return null; - const remainingRaw = Math.max(0, toNumber(getFieldValue(remainingQuotas, name, name), 0)); - const remaining = Math.min(remainingRaw, total); - const used = Math.max(total - remaining, 0); - quotas[name] = { - used, - total, - remaining, - remainingPercentage: clampPercentage((remaining / total) * 100), - unlimited: false, - resetAt, - }; - return quotas[name]; - }; - - const premiumQuota = addLimitedQuota("premium_interactions"); - addLimitedQuota("chat"); - addLimitedQuota("completions"); - - return { - plan: inferGitHubPlanName(dataRecord, premiumQuota), - resetDate, - quotas, - }; - } - - return { message: "GitHub Copilot connected. Unable to parse quota data." }; - } catch (error) { - throw new Error(`Failed to fetch GitHub usage: ${error.message}`); - } -} - -function formatGitHubQuotaSnapshot( - quota: unknown, - resetAt: string | null = null -): UsageQuota | null { - const source = toRecord(quota); - if (Object.keys(source).length === 0) return null; - - const unlimited = source.unlimited === true; - const entitlement = toNumber(source.entitlement, Number.NaN); - const totalValue = toNumber(source.total, Number.NaN); - const remainingValue = toNumber(source.remaining, Number.NaN); - const usedValue = toNumber(source.used, Number.NaN); - const percentRemainingValue = toNumber( - getFieldValue(source, "percent_remaining", "percentRemaining"), - Number.NaN - ); - - let total = Number.isFinite(totalValue) - ? Math.max(0, totalValue) - : Number.isFinite(entitlement) - ? Math.max(0, entitlement) - : 0; - let remaining = Number.isFinite(remainingValue) ? Math.max(0, remainingValue) : undefined; - let used = Number.isFinite(usedValue) ? Math.max(0, usedValue) : undefined; - let remainingPercentage = Number.isFinite(percentRemainingValue) - ? clampPercentage(percentRemainingValue) - : undefined; - - if (used === undefined && total > 0 && remaining !== undefined) { - used = Math.max(total - remaining, 0); - } - - if (remaining === undefined && total > 0 && used !== undefined) { - remaining = Math.max(total - used, 0); - } - - if (remainingPercentage === undefined && total > 0 && remaining !== undefined) { - remainingPercentage = clampPercentage((remaining / total) * 100); - } - - if (total <= 0 && remainingPercentage !== undefined) { - total = 100; - used = 100 - remainingPercentage; - remaining = remainingPercentage; - } - - return { - used: Math.max(0, used ?? 0), - total, - remaining, - remainingPercentage, - resetAt, - unlimited, - }; -} - -function inferGitHubPlanName(data: JsonRecord, premiumQuota: UsageQuota | null): string { - const rawPlan = getFieldValue(data, "copilot_plan", "copilotPlan"); - const rawSku = getFieldValue(data, "access_type_sku", "accessTypeSku"); - const planText = typeof rawPlan === "string" ? rawPlan.trim() : ""; - const skuText = typeof rawSku === "string" ? rawSku.trim() : ""; - const combined = `${skuText} ${planText}`.trim().toUpperCase(); - const monthlyQuotas = toRecord(getFieldValue(data, "monthly_quotas", "monthlyQuotas")); - const premiumTotal = - premiumQuota?.total || - toNumber(getFieldValue(monthlyQuotas, "premium_interactions", "premiumInteractions"), 0); - const chatTotal = toNumber(getFieldValue(monthlyQuotas, "chat", "chat"), 0); - - if (combined.includes("PRO+") || combined.includes("PRO_PLUS") || combined.includes("PROPLUS")) { - return "Copilot Pro+"; - } - if (combined.includes("ENTERPRISE")) return "Copilot Enterprise"; - if (combined.includes("BUSINESS")) return "Copilot Business"; - if (combined.includes("STUDENT")) return "Copilot Student"; - if (combined.includes("FREE")) return "Copilot Free"; - if (combined.includes("PRO")) return "Copilot Pro"; - - if (premiumTotal >= 1400) return "Copilot Pro+"; - if (premiumTotal >= 900) return "Copilot Enterprise"; - if (premiumTotal >= 250) { - if (combined.includes("INDIVIDUAL")) return "Copilot Pro"; - return "Copilot Business"; - } - if (premiumTotal > 0 || chatTotal === 50) return "Copilot Free"; - - if (skuText) { - const label = toDisplayLabel(skuText); - return label ? `Copilot ${label}` : "GitHub Copilot"; - } - if (planText) { - const label = toDisplayLabel(planText); - return label ? `Copilot ${label}` : "GitHub Copilot"; - } - return "GitHub Copilot"; -} - -// ── Gemini CLI subscription info cache ────────────────────────────────────── -// Prevents duplicate loadCodeAssist calls within the same quota cycle. -// Key: accessToken → { data, fetchedAt } -const _geminiCliSubCache = new Map(); -const GEMINI_CLI_CACHE_TTL_MS = 5 * 60 * 1000; // 5 minutes - -/** - * Gemini CLI Usage — fetch per-model quota from Cloud Code Assist API. - * Gemini CLI and Antigravity share the same upstream (cloudcode-pa.googleapis.com), - * so this follows the same pattern as getAntigravityUsage(). - */ -async function getGeminiUsage( - accessToken?: string, - providerSpecificData?: JsonRecord, - connectionProjectId?: string -) { - if (!accessToken) { - return { plan: "Free", message: "Gemini CLI access token not available." }; - } - - try { - const subscriptionInfo = await getGeminiCliSubscriptionInfoCached(accessToken); - const projectId = - connectionProjectId || - providerSpecificData?.projectId || - toRecord(subscriptionInfo).cloudaicompanionProject || - null; - - const plan = getGeminiCliPlanLabel(subscriptionInfo); - - if (!projectId) { - return { plan, message: "Gemini CLI project ID not available." }; - } - - // Use retrieveUserQuota (same endpoint as Gemini CLI /stats command). - // Returns per-model buckets with remainingFraction and resetTime. - const response = await fetch( - "https://cloudcode-pa.googleapis.com/v1internal:retrieveUserQuota", - { - method: "POST", - headers: { - Authorization: `Bearer ${accessToken}`, - "Content-Type": "application/json", - }, - body: JSON.stringify({ project: projectId }), - signal: AbortSignal.timeout(10000), - } - ); - - if (!response.ok) { - return { plan, message: `Gemini CLI quota error (${response.status}).` }; - } - - const data = await response.json(); - const quotas: Record = {}; - - const dataRecord = toRecord(data); - if (Array.isArray(dataRecord.buckets)) { - for (const bucketValue of dataRecord.buckets) { - const bucket = toRecord(bucketValue); - if (!bucket.modelId || bucket.remainingFraction == null) continue; - - const remainingFraction = toNumber(bucket.remainingFraction, 0); - const remainingPercentage = remainingFraction * 100; - const QUOTA_NORMALIZED_BASE = 1000; - const total = QUOTA_NORMALIZED_BASE; - const remaining = Math.round(total * remainingFraction); - const used = Math.max(0, total - remaining); - - quotas[String(bucket.modelId)] = { - used, - total, - resetAt: parseResetTime(bucket.resetTime), - remainingPercentage, - unlimited: false, - }; - } - } - - return { plan, quotas }; - } catch (error) { - return { message: `Gemini CLI error: ${(error as Error).message}` }; - } -} - -/** - * Get Gemini CLI subscription info (cached, 5 min TTL) - */ -async function getGeminiCliSubscriptionInfoCached(accessToken: string): Promise { - const cacheKey = accessToken; - const cached = _geminiCliSubCache.get(cacheKey); - - if (cached && Date.now() - cached.fetchedAt < GEMINI_CLI_CACHE_TTL_MS) { - return cached.data; - } - - const data = await getGeminiCliSubscriptionInfo(accessToken); - _geminiCliSubCache.set(cacheKey, { data, fetchedAt: Date.now() }); - return data; -} - -/** - * Get Gemini CLI subscription info using correct headers. - */ -async function getGeminiCliSubscriptionInfo(accessToken: string): Promise { - try { - const response = await fetch(GEMINI_CLI_USAGE_URL, { - method: "POST", - headers: { - Authorization: `Bearer ${accessToken}`, - "Content-Type": "application/json", - }, - body: JSON.stringify({ - metadata: { - ideType: "IDE_UNSPECIFIED", - platform: "PLATFORM_UNSPECIFIED", - pluginType: "GEMINI", - }, - }), - }); - - if (!response.ok) return null; - - return await response.json(); - } catch { - return null; - } -} - -/** - * Map Gemini CLI subscription tier to display label (same tiers as Antigravity). - */ -function getGeminiCliPlanLabel(subscriptionInfo: unknown): string { - return mapCodeAssistSubscriptionToPlanLabel(subscriptionInfo); -} - -// ── Antigravity subscription info cache ────────────────────────────────────── -// Prevents duplicate loadCodeAssist calls within the same quota cycle. -// Key: truncated accessToken → { data, fetchedAt } -const _antigravitySubCache = new Map(); -const ANTIGRAVITY_CACHE_TTL_MS = 5 * 60 * 1000; // 5 minutes -const ANTIGRAVITY_MODELS_CACHE_TTL_MS = 60 * 1000; -const ANTIGRAVITY_CREDIT_PROBE_TTL_MS = 5 * 60 * 1000; -const _antigravityAvailableModelsCache = new Map(); -const _antigravityAvailableModelsInflight = new Map>(); -const _antigravityUserQuotaCache = new Map(); -const _antigravityUserQuotaInflight = new Map>(); -const _antigravityCreditProbeCache = new Map(); -const _antigravityCreditProbeInflight = new Map>(); - -// ── Proactive TTL purging for module-level caches ────────────────────────── -// All 4 data caches only evict on read (passive TTL). This interval proactively -// purges stale entries so keys accessed once and never again don't leak memory. -// The 2 inflight Maps (availableModelsInflight, creditProbeInflight) self-clean -// when the Promise resolves/rejects, so they are NOT touched here. -const _usageCacheCleanupTimer = setInterval( - () => { - const now = Date.now(); - for (const [key, entry] of _geminiCliSubCache) { - if (now - entry.fetchedAt > GEMINI_CLI_CACHE_TTL_MS) _geminiCliSubCache.delete(key); - } - for (const [key, entry] of _antigravitySubCache) { - if (now - entry.fetchedAt > ANTIGRAVITY_CACHE_TTL_MS) _antigravitySubCache.delete(key); - } - for (const [key, entry] of _antigravityAvailableModelsCache) { - if (now - entry.fetchedAt > ANTIGRAVITY_MODELS_CACHE_TTL_MS) - _antigravityAvailableModelsCache.delete(key); - } - for (const [key, entry] of _antigravityUserQuotaCache) { - if (now - entry.fetchedAt > ANTIGRAVITY_MODELS_CACHE_TTL_MS) - _antigravityUserQuotaCache.delete(key); - } - for (const [key, entry] of _antigravityCreditProbeCache) { - if (now - entry.fetchedAt > ANTIGRAVITY_CREDIT_PROBE_TTL_MS) - _antigravityCreditProbeCache.delete(key); - } - }, - 5 * 60 * 1000 -); // every 5 minutes -_usageCacheCleanupTimer.unref?.(); // Don't prevent process exit - -interface AntigravityUsageOptions { - forceRefresh?: boolean; -} - -const ANTIGRAVITY_LOCAL_USAGE_WINDOW_MS = 5 * 60 * 60 * 1000; -const ANTIGRAVITY_LOCAL_USAGE_TOKENS_PER_UNIT = 1000; - -// `toClientAntigravityQuotaModelId` was an inline if-ladder here; it is now the single -// source of truth in open-sse/config/antigravityModelAliases.ts (imported above), shared -// with the provider-limits cache sanitizer. (#3821-review LEDGER-5) - -function getAntigravityLocalUsageUnits( - provider: "antigravity" | "agy", - connectionId: string | undefined, - modelId: string, - resetAt: string | null -): number { - if (!connectionId || !modelId || !resetAt) return 0; - - const resetMs = Date.parse(resetAt); - if (!Number.isFinite(resetMs)) return 0; - - const windowStart = new Date(resetMs - ANTIGRAVITY_LOCAL_USAGE_WINDOW_MS).toISOString(); - const windowEnd = new Date(resetMs).toISOString(); - - try { - const db = getDbInstance() as unknown as { - prepare: (sql: string) => { get: (...params: unknown[]) => unknown }; - }; - const row = db - .prepare( - `SELECT COALESCE(SUM( - COALESCE(tokens_input, 0) + COALESCE(tokens_output, 0) + COALESCE(tokens_reasoning, 0) - ), 0) AS tokens - FROM usage_history - WHERE provider = ? - AND connection_id = ? - AND model = ? - AND success = 1 - AND timestamp >= ? - AND timestamp < ?` - ) - .get(provider, connectionId, modelId, windowStart, windowEnd) as - | { tokens?: unknown } - | undefined; - - const tokens = Number(row?.tokens || 0); - if (!Number.isFinite(tokens) || tokens <= 0) return 0; - return Math.max(1, Math.ceil(tokens / ANTIGRAVITY_LOCAL_USAGE_TOKENS_PER_UNIT)); - } catch { - return 0; - } -} - -function applyLocalUsageFallback( - quota: UsageQuota, - provider: "antigravity" | "agy", - connectionId: string | undefined, - modelId: string -): UsageQuota { - if (quota.quotaSource !== "fetchAvailableModels" || quota.used > 0 || quota.unlimited) { - return quota; - } - - const localUsed = getAntigravityLocalUsageUnits(provider, connectionId, modelId, quota.resetAt); - if (localUsed <= 0 || quota.total <= 0) return quota; - - const used = Math.min(quota.total, localUsed); - return { - ...quota, - used, - remainingPercentage: Math.max(0, ((quota.total - used) / quota.total) * 100), - quotaSource: "localUsageHistory", - }; -} - -function buildAntigravityUsageCacheKey(accessToken: string, projectId?: string | null): string { - return `${accessToken.substring(0, 16)}:${projectId || "default"}`; -} - -async function fetchAntigravityAvailableModelsCached( - accessToken: string, - projectId?: string | null, - options: AntigravityUsageOptions = {} -): Promise { - if (!accessToken) throw new Error("Access token is required"); - - const cacheKey = buildAntigravityUsageCacheKey(accessToken, projectId); - const cached = _antigravityAvailableModelsCache.get(cacheKey); - if ( - !options.forceRefresh && - cached && - Date.now() - cached.fetchedAt < ANTIGRAVITY_MODELS_CACHE_TTL_MS - ) { - return cached.data; - } - - const inflight = _antigravityAvailableModelsInflight.get(cacheKey); - if (inflight) return inflight; - - const promise = (async () => { - let response: Response | null = null; - let lastError: Error | null = null; - - for (const quotaApiUrl of ANTIGRAVITY_CONFIG.quotaApiUrls) { - try { - response = await fetch(quotaApiUrl, { - method: "POST", - headers: getAntigravityHeaders("fetchAvailableModels", accessToken), - body: JSON.stringify(projectId ? { project: projectId } : {}), - signal: AbortSignal.timeout(10000), - }); - - if (response.ok || response.status === 401 || response.status === 403) { - break; - } - } catch (error) { - lastError = error as Error; - } - } - - if (!response) { - throw lastError || new Error("Antigravity API unavailable"); - } - - if (response.status === 403) { - return { __antigravityForbidden: true }; - } - - if (!response.ok) { - throw new Error(`Antigravity API error: ${response.status}`); - } - - const data = await response.json(); - _antigravityAvailableModelsCache.set(cacheKey, { data, fetchedAt: Date.now() }); - return data; - })().finally(() => { - _antigravityAvailableModelsInflight.delete(cacheKey); - }); - - _antigravityAvailableModelsInflight.set(cacheKey, promise); - return promise; -} - -async function fetchAntigravityUserQuotaCached( - accessToken: string, - projectId?: string | null, - options: AntigravityUsageOptions = {} -): Promise { - if (!accessToken || !projectId) return null; - - const cacheKey = buildAntigravityUsageCacheKey(accessToken, projectId); - const cached = _antigravityUserQuotaCache.get(cacheKey); - if ( - !options.forceRefresh && - cached && - Date.now() - cached.fetchedAt < ANTIGRAVITY_MODELS_CACHE_TTL_MS - ) { - return cached.data; - } - - const inflight = _antigravityUserQuotaInflight.get(cacheKey); - if (inflight) return inflight; - - const promise = (async () => { - try { - const response = await fetch( - "https://cloudcode-pa.googleapis.com/v1internal:retrieveUserQuota", - { - method: "POST", - headers: { - Authorization: `Bearer ${accessToken}`, - "Content-Type": "application/json", - }, - body: JSON.stringify({ project: projectId }), - signal: AbortSignal.timeout(10000), - } - ); - - if (!response.ok) return null; - - const data = await response.json(); - _antigravityUserQuotaCache.set(cacheKey, { data, fetchedAt: Date.now() }); - return data; - } catch { - return null; - } - })().finally(() => { - _antigravityUserQuotaInflight.delete(cacheKey); - }); - - _antigravityUserQuotaInflight.set(cacheKey, promise); - return promise; -} - -function extractCodeAssistTierId(subscription: JsonRecord): string { - const tierId = extractCodeAssistOnboardTierId(subscription); - if (tierId === "legacy-tier") return ""; - const upper = tierId.toUpperCase(); - return mapCodeAssistTierIdToLabel(upper) ? upper : ""; -} - -function mapCodeAssistTierIdToLabel(tierId: string): string | null { - const upper = tierId.toUpperCase(); - if (upper.includes("ULTRA")) return "Ultra"; - if ( - upper.includes("PRO") || - upper.includes("PREMIUM") || - upper.includes("GOOGLE_ONE") || - upper.includes("ONE_AI") - ) - return "Pro"; - if (upper.includes("ENTERPRISE")) return "Enterprise"; - if (upper.includes("BUSINESS") || upper.includes("STANDARD")) return "Business"; - if (upper.includes("PLUS")) return "Plus"; - if (upper.includes("LITE") || upper.includes("LIGHT")) return "Lite"; - if (upper.includes("FREE") || upper.includes("INDIVIDUAL") || upper.includes("LEGACY")) - return "Free"; - return null; -} - -function mapSubscriptionTierStringToPlanLabel(tierText: string): string | null { - const upper = tierText.toUpperCase(); - if (upper.includes("ULTRA")) return "Ultra"; - if (upper.includes("PRO") || upper.includes("PREMIUM") || upper.includes("GOOGLE ONE")) - return "Pro"; - if (upper.includes("ENTERPRISE")) return "Enterprise"; - if (upper.includes("STANDARD") || upper.includes("BUSINESS")) return "Business"; - if (upper.includes("PLUS")) return "Plus"; - if (upper.includes("LITE")) return "Lite"; - if (upper.includes("INDIVIDUAL") || upper.includes("FREE")) return "Free"; - // Strip a trailing "(RESTRICTED)" marker. Match the fixed literal anywhere then - // trim, instead of /\s*\(RESTRICTED\)\s*$/ whose overlapping \s* runs backtrack - // polynomially on whitespace-heavy upstream input (js/polynomial-redos). - const normalizedId = upper.replace(/\(RESTRICTED\)/i, "").trim(); - if (normalizedId) { - const mapped = mapCodeAssistTierIdToLabel(normalizedId); - if (mapped) return mapped; - } - return null; -} - -function mapCodeAssistSubscriptionToPlanLabel(subscriptionInfo: unknown): string { - const subscription = toRecord(subscriptionInfo); - if (Object.keys(subscription).length === 0) return "Free"; - - const subscriptionTier = extractCodeAssistSubscriptionTier(subscriptionInfo); - if (subscriptionTier) { - const mapped = mapSubscriptionTierStringToPlanLabel(subscriptionTier); - if (mapped) return mapped; - if (subscriptionTier.toLowerCase() !== "free") { - return subscriptionTier.charAt(0).toUpperCase() + subscriptionTier.slice(1).toLowerCase(); - } - } - - const currentTier = toRecord(subscription.currentTier); - const tierName = String( - getFieldValue(currentTier, "name", "displayName") || - subscription.subscriptionType || - subscription.tier || - "" - ); - const mappedName = tierName ? mapSubscriptionTierStringToPlanLabel(tierName) : null; - if (mappedName) return mappedName; - - const tierId = extractCodeAssistTierId(subscription); - if (tierId) { - const mapped = mapCodeAssistTierIdToLabel(tierId); - if (mapped) return mapped; - } - if (currentTier.upgradeSubscriptionType) return "Free"; - if (tierName) return tierName.charAt(0).toUpperCase() + tierName.slice(1).toLowerCase(); - return "Free"; -} - -const KNOWN_ANTIGRAVITY_PLAN_LABELS = new Set([ - "Ultra", - "Pro", - "Enterprise", - "Business", - "Plus", - "Lite", -]); - -/** - * Map raw loadCodeAssist tier data to short display labels (Antigravity Manager parity). - */ -function getAntigravityPlanLabel(subscriptionInfo: unknown, fallbackInfo?: unknown): string { - const livePlan = mapCodeAssistSubscriptionToPlanLabel(subscriptionInfo); - const fallbackPlan = mapCodeAssistSubscriptionToPlanLabel(fallbackInfo); - - if (KNOWN_ANTIGRAVITY_PLAN_LABELS.has(livePlan)) return livePlan; - if (KNOWN_ANTIGRAVITY_PLAN_LABELS.has(fallbackPlan)) return fallbackPlan; - if (livePlan !== "Free") return livePlan; - return fallbackPlan !== "Free" ? fallbackPlan : livePlan; -} - -/** - * Proactive credit balance probe for Antigravity. - * - * Fires a minimal streamGenerateContent request with GOOGLE_ONE_AI credits enabled - * and maxOutputTokens=1 to extract the `remainingCredits` field from the SSE stream. - * This uses ~1 credit but lets us show the balance on the dashboard without waiting - * for a real user request. - * - * Returns the credit balance, or null if the probe failed. - */ -async function probeAntigravityCreditBalance( - accessToken: string, - accountId: string, - projectId?: string | null, - options: AntigravityUsageOptions = {}, - providerSpecificData: JsonRecord = {} -): Promise { - if (!accessToken) return null; - - const cacheKey = buildAntigravityUsageCacheKey(accessToken, projectId || accountId); - const cached = _antigravityCreditProbeCache.get(cacheKey); - if ( - !options.forceRefresh && - cached && - Date.now() - cached.fetchedAt < ANTIGRAVITY_CREDIT_PROBE_TTL_MS - ) { - return cached.data; - } - - const inflight = _antigravityCreditProbeInflight.get(cacheKey); - if (inflight) return inflight; - - const promise = probeAntigravityCreditBalanceUncached( - accessToken, - accountId, - projectId, - providerSpecificData - ) - .then( - (data) => { - _antigravityCreditProbeCache.set(cacheKey, { data, fetchedAt: Date.now() }); - return data; - }, - (error) => { - _antigravityCreditProbeCache.set(cacheKey, { data: null, fetchedAt: Date.now() }); - throw error; - } - ) - .finally(() => { - _antigravityCreditProbeInflight.delete(cacheKey); - }); - - _antigravityCreditProbeInflight.set(cacheKey, promise); - return promise; -} - -async function probeAntigravityCreditBalanceUncached( - accessToken: string, - accountId: string, - projectId?: string | null, - providerSpecificData: JsonRecord = {} -): Promise { - try { - if (!projectId) return null; - - // Try all base URLs (some accounts only work with specific endpoints) - for (const baseUrl of ANTIGRAVITY_BASE_URLS) { - const url = `${baseUrl}/v1internal:streamGenerateContent?alt=sse`; - - const sessionId = getAntigravitySessionId({ connectionId: accountId, projectId }); - const body = { - project: projectId, - model: "gemini-2-flash", - userAgent: "antigravity", - requestType: "agent", - requestId: generateAntigravityRequestId(), - enabledCreditTypes: ["GOOGLE_ONE_AI"], - request: { - model: "gemini-2-flash", - contents: [{ role: "user", parts: [{ text: "hi" }] }], - generationConfig: { maxOutputTokens: 1 }, - sessionId, - }, - }; - - const headers: Record = { - "Content-Type": "application/json", - Authorization: `Bearer ${accessToken}`, - Accept: "text/event-stream", - }; - applyAntigravityClientProfileHeaders( - headers, - { connectionId: accountId, projectId, providerSpecificData }, - body - ); - - try { - const res = await fetch(url, { - method: "POST", - headers, - body: JSON.stringify(body), - signal: AbortSignal.timeout(10_000), - }); - - if (!res.ok) continue; - - // Read the full SSE response and scan for remainingCredits - const rawSSE = await res.text(); - const lines = rawSSE.split("\n"); - - for (const line of lines) { - const trimmed = line.trim(); - if (!trimmed.startsWith("data:")) continue; - const payload = trimmed.slice(5).trim(); - if (payload === "[DONE]") break; - try { - const parsed = JSON.parse(payload); - if (Array.isArray(parsed?.remainingCredits)) { - const googleCredit = parsed.remainingCredits.find( - (c: { creditType?: string }) => c?.creditType === "GOOGLE_ONE_AI" - ); - if (googleCredit) { - const balance = parseInt(googleCredit.creditAmount, 10); - if (!isNaN(balance)) { - updateAntigravityRemainingCredits(accountId, balance); - return balance; - } - } - } - } catch { - // Skip malformed SSE lines - } - } - } catch { - // Individual endpoint failure; try next - } - } - - return null; - } catch { - // Probe is best-effort — don't let it break the usage fetch - return null; - } -} - -/** - * Antigravity Usage - Fetch quota from Google Cloud Code API. - * fetchAvailableModels is catalog/eligibility data and may keep reporting full buckets - * after real usage. retrieveUserQuota is the consumption signal for Gemini-family - * buckets, so prefer it when present and fall back to fetchAvailableModels only for - * models that have no retrieveUserQuota entry (for example Claude/GPT OSS buckets). - */ -async function getAntigravityUsage( - provider: "antigravity" | "agy", - accessToken?: string, - providerSpecificData?: JsonRecord, - connectionProjectId?: string, - connectionId?: string, - options: AntigravityUsageOptions = {} -) { - if (!accessToken) { - return { plan: "Free", message: "Antigravity access token not available." }; - } - - let subscriptionInfo: unknown = null; - try { - subscriptionInfo = await getAntigravitySubscriptionInfoCached( - accessToken, - providerSpecificData, - options - ); - const savedProjectId = - typeof providerSpecificData?.projectId === "string" && providerSpecificData.projectId.trim() - ? providerSpecificData.projectId.trim() - : null; - const subscriptionProject = toRecord(subscriptionInfo).cloudaicompanionProject; - const projectId = - savedProjectId || - connectionProjectId || - (typeof subscriptionProject === "string" - ? subscriptionProject - : typeof toRecord(subscriptionProject).id === "string" - ? (toRecord(subscriptionProject).id as string) - : null); - - // Derive accountId for credit balance cache. - // Must match executor key: credentials.connectionId - const accountId: string = connectionId || "unknown"; - - // Read cached credit balance (hydrated from DB on first access) - let creditBalance = getAntigravityRemainingCredits(accountId); - - // If no cached balance and credits mode is enabled, fire a minimal probe - const creditsMode = getCreditsMode(); - if ((options.forceRefresh || creditBalance === null) && creditsMode !== "off") { - creditBalance = await probeAntigravityCreditBalance( - accessToken, - accountId, - projectId, - options, - providerSpecificData || {} - ); - } - - const [data, userQuotaData] = await Promise.all([ - fetchAntigravityAvailableModelsCached(accessToken, projectId, options), - fetchAntigravityUserQuotaCached(accessToken, projectId, options), - ]); - const dataObj = toRecord(data); - if (dataObj.__antigravityForbidden === true) { - return { message: "Antigravity access forbidden. Check subscription." }; - } - const modelEntries = toRecord(dataObj.models); - const userQuotaEntries = new Map(); - const userQuotaObj = toRecord(userQuotaData); - if (Array.isArray(userQuotaObj.buckets)) { - for (const bucketValue of userQuotaObj.buckets) { - const bucket = toRecord(bucketValue); - const modelId = toClientAntigravityQuotaModelId(String(bucket.modelId || "").trim()); - if (!modelId) continue; - userQuotaEntries.set(modelId, bucket); - } - } - const quotas: Record = {}; - - // Parse per-model quota info from fetchAvailableModels response. - for (const [rawModelKey, infoValue] of Object.entries(modelEntries)) { - const info = toRecord(infoValue); - const quotaInfo = toRecord(info.quotaInfo); - const modelKey = toClientAntigravityQuotaModelId(rawModelKey); - - // Skip internal, excluded, and models without quota info - if ( - !modelKey || - info.isInternal === true || - !(provider === "agy" - ? isUserCallableAgyModelId(modelKey) - : isUserCallableAntigravityModelId(modelKey)) || - Object.keys(quotaInfo).length === 0 - ) { - continue; - } - - const liveQuota = userQuotaEntries.get(modelKey); - const quotaSource = liveQuota || quotaInfo; - const rawFraction = toNumber(quotaSource.remainingFraction, -1); - const resetAt = parseResetTime(quotaSource.resetTime); - // Distinguish "upstream did not report remainingFraction" from "remaining is 0%". - // fetchAvailableModels is a catalog view and can be stale/full; retrieveUserQuota is - // the source of truth for actual Gemini consumption when it includes the model. - const fractionReported = rawFraction >= 0; - if (!fractionReported) { - console.warn( - `[Antigravity] model ${modelKey} returned no remainingFraction — quota unknown` - ); - } - const remainingFraction = fractionReported ? Math.max(0, Math.min(1, rawFraction)) : 0; - // Models with no resetTime AND a reported full fraction are unlimited - // (e.g. tab-completion models). Unreported fraction is NEVER unlimited. - const isUnlimited = fractionReported && !resetAt && remainingFraction >= 1; - const remainingPercentage = remainingFraction * 100; - const QUOTA_NORMALIZED_BASE = 1000; - const total = QUOTA_NORMALIZED_BASE; - const remaining = Math.round(total * remainingFraction); - const used = isUnlimited ? 0 : Math.max(0, total - remaining); - - quotas[modelKey] = applyLocalUsageFallback( - { - used, - total: isUnlimited ? 0 : total, - resetAt, - remainingPercentage: isUnlimited ? 100 : remainingPercentage, - unlimited: isUnlimited, - fractionReported, - quotaSource: liveQuota ? "retrieveUserQuota" : "fetchAvailableModels", - }, - provider, - connectionId, - modelKey - ); - } - - // Include retrieveUserQuota buckets not listed in the static/public Antigravity catalog yet. - // This keeps Provider Limits honest when Google adds a new Gemini tier before our catalog is - // updated. Hidden/internal catalog entries above are still filtered by the public pass. - for (const [modelKey, bucket] of userQuotaEntries) { - if ( - quotas[modelKey] || - !(provider === "agy" - ? isUserCallableAgyModelId(modelKey) - : isUserCallableAntigravityModelId(modelKey)) - ) { - continue; - } - const rawFraction = toNumber(bucket.remainingFraction, -1); - if (rawFraction < 0) continue; - const remainingFraction = Math.max(0, Math.min(1, rawFraction)); - const resetAt = parseResetTime(bucket.resetTime); - const isUnlimited = !resetAt && remainingFraction >= 1; - const QUOTA_NORMALIZED_BASE = 1000; - const total = QUOTA_NORMALIZED_BASE; - const remaining = Math.round(total * remainingFraction); - quotas[modelKey] = { - used: isUnlimited ? 0 : Math.max(0, total - remaining), - total: isUnlimited ? 0 : total, - resetAt, - remainingPercentage: isUnlimited ? 100 : remainingFraction * 100, - unlimited: isUnlimited, - fractionReported: true, - quotaSource: "retrieveUserQuota", - }; - } - - return { - plan: getAntigravityPlanLabel(subscriptionInfo, providerSpecificData), - quotas: { - ...quotas, - ...(creditBalance !== null && { - credits: { - used: 0, - total: 0, - remaining: creditBalance, - unlimited: false, - resetAt: null, - }, - }), - }, - subscriptionInfo, - }; - } catch (error) { - return { - plan: getAntigravityPlanLabel(subscriptionInfo, providerSpecificData), - subscriptionInfo, - message: `Antigravity error: ${(error as Error).message}`, - }; - } -} - -/** - * Get Antigravity subscription info (cached, 5 min TTL) - * Prevents duplicate loadCodeAssist calls within the same quota cycle. - */ -async function getAntigravitySubscriptionInfoCached( - accessToken: string, - providerSpecificData?: JsonRecord, - options: AntigravityUsageOptions = {} -): Promise { - const profile = getAntigravityClientProfile({ providerSpecificData }); - const cacheKey = `${accessToken.substring(0, 16)}:${profile}`; - - if (options.forceRefresh) { - _antigravitySubCache.delete(cacheKey); - } else { - const cached = _antigravitySubCache.get(cacheKey); - if (cached && Date.now() - cached.fetchedAt < ANTIGRAVITY_CACHE_TTL_MS) { - return cached.data; - } - } - - const data = await getAntigravitySubscriptionInfo(accessToken, providerSpecificData); - if (data != null) { - _antigravitySubCache.set(cacheKey, { data, fetchedAt: Date.now() }); - } - return data; -} - -/** - * Get Antigravity subscription info using correct Antigravity headers. - * Must match the headers used in providers.js postExchange (not CLI headers). - */ -async function getAntigravitySubscriptionInfo( - accessToken: string, - providerSpecificData?: JsonRecord -): Promise { - try { - const profile = getAntigravityClientProfile({ providerSpecificData }); - const response = await fetch(ANTIGRAVITY_CONFIG.loadProjectApiUrl, { - method: "POST", - headers: - profile === "harness" - ? getAntigravityBootstrapHeaders(profile, accessToken) - : getAntigravityHeaders("loadCodeAssist", accessToken), - body: JSON.stringify({ metadata: getAntigravityLoadCodeAssistMetadata() }), - }); - - if (!response.ok) return null; - - return await response.json(); - } catch { - return null; - } -} - -/** - * Claude Usage - Try to fetch from Anthropic API - */ -async function getClaudeUsage(accessToken?: string) { - if (!accessToken) { - return { message: "Claude connected. Access token not available.", bootstrap: null }; - } - - // Refresh bootstrap in parallel; best-effort, failure non-fatal. - const bootstrapPromise = fetchClaudeBootstrap(accessToken).catch(() => null); - try { - // Real CLI uses axios here, not Stainless — UA is `claude-code/` - // (not `claude-cli/...`) and the shape is simpler than /v1/messages. - const ctrl = new AbortController(); - const timer = setTimeout(() => ctrl.abort(), 10_000); - let oauthResponse; - try { - oauthResponse = await fetch(CLAUDE_CONFIG.oauthUsageUrl, { - method: "GET", - headers: { - Accept: "application/json, text/plain, */*", - "Accept-Encoding": "gzip, compress, deflate, br", - Authorization: `Bearer ${accessToken}`, - "Content-Type": "application/json", - "User-Agent": `claude-code/${CLAUDE_CODE_VERSION}`, - "anthropic-beta": "oauth-2025-04-20", - }, - signal: ctrl.signal, - }); - } finally { - clearTimeout(timer); - } - - if (oauthResponse.ok) { - const data = toRecord(await oauthResponse.json()); - const quotas: Record = {}; - - // utilization = percentage USED (e.g., 90 means 90% used, 10% remaining) - // Confirmed via user report #299: Claude.ai shows 87% used = OmniRoute must show 13% remaining. - const hasUtilization = (window: JsonRecord) => - window && typeof window === "object" && safePercentage(window.utilization) !== undefined; - - const createQuotaObject = (window: JsonRecord) => { - const used = safePercentage(window.utilization) as number; // utilization = % used - const remaining = Math.max(0, 100 - used); - return { - used, - total: 100, - remaining, - resetAt: parseResetTime(window.resets_at), - remainingPercentage: remaining, - unlimited: false, - }; - }; - - const fiveHour = toRecord(data.five_hour); - if (hasUtilization(fiveHour)) { - quotas["session (5h)"] = createQuotaObject(fiveHour); - } - - const sevenDay = toRecord(data.seven_day); - if (hasUtilization(sevenDay)) { - quotas["weekly (7d)"] = createQuotaObject(sevenDay); - } - - // Map Anthropic's internal codenames (e.g., omelette → Designer) for display. - const MODEL_DISPLAY_NAMES: Record = { - omelette: "designer", - }; - for (const [key, value] of Object.entries(data)) { - const valueRecord = toRecord(value); - if (key.startsWith("seven_day_") && key !== "seven_day" && hasUtilization(valueRecord)) { - const codename = key.replace("seven_day_", ""); - const modelName = MODEL_DISPLAY_NAMES[codename] || codename; - quotas[`weekly ${modelName} (7d)`] = createQuotaObject(valueRecord); - } - } - - const bootstrap = await bootstrapPromise; - const plan = - getClaudePlanLabel( - typeof data.tier === "string" ? data.tier : null, - typeof data.plan === "string" ? data.plan : null, - typeof data.subscription_type === "string" ? data.subscription_type : null, - bootstrap?.organization_rate_limit_tier - ) ?? undefined; - - return { - ...(plan ? { plan } : {}), - quotas, - extraUsage: data.extra_usage ?? null, - bootstrap, - }; - } - - // Fallback: OAuth endpoint returned non-OK, try legacy settings/org endpoint - console.warn( - `[Claude Usage] OAuth endpoint returned ${oauthResponse.status}, falling back to legacy` - ); - const legacy = await getClaudeUsageLegacy(accessToken); - return { ...legacy, bootstrap: await bootstrapPromise }; - } catch (error) { - return { - message: `Claude connected. Unable to fetch usage: ${(error as Error).message}`, - bootstrap: await bootstrapPromise, - }; - } -} - -/** - * Legacy Claude usage fetcher for API key / org admin users. - * Uses /v1/settings + /v1/organizations/{org_id}/usage endpoints. - */ -async function getClaudeUsageLegacy(accessToken?: string) { - try { - const settingsResponse = await fetch(CLAUDE_CONFIG.settingsUrl, { - method: "GET", - headers: { - Authorization: `Bearer ${accessToken}`, - "anthropic-version": CLAUDE_CONFIG.apiVersion, - }, - }); - - if (settingsResponse.ok) { - const settings = toRecord(await settingsResponse.json()); - - const organizationId = - typeof settings.organization_id === "string" ? settings.organization_id : ""; - if (organizationId) { - const usageResponse = await fetch( - CLAUDE_CONFIG.usageUrl.replace("{org_id}", organizationId), - { - method: "GET", - headers: { - Authorization: `Bearer ${accessToken}`, - "anthropic-version": CLAUDE_CONFIG.apiVersion, - }, - } - ); - - if (usageResponse.ok) { - const usage = await usageResponse.json(); - return { - plan: settings.plan || "Unknown", - organization: settings.organization_name, - quotas: usage, - }; - } - } - - return { - plan: settings.plan || "Unknown", - organization: settings.organization_name, - message: "Claude connected. Usage details require admin access.", - }; - } - - return { message: "Claude connected. Usage API requires admin permissions." }; - } catch (error) { - return { message: `Claude connected. Unable to fetch usage: ${(error as Error).message}` }; - } -} - -/** - * Codex (OpenAI) Usage - Fetch from ChatGPT backend API - * IMPORTANT: Uses persisted workspaceId from OAuth to ensure correct workspace binding. - * No fallback to other workspaces - strict binding to user's selected workspace. - */ -async function getCodexUsage( - accessToken?: string, - providerSpecificData: Record = {} -) { - try { - // Use persisted workspace ID from OAuth - NO FALLBACK - const accountId = - typeof providerSpecificData.workspaceId === "string" - ? providerSpecificData.workspaceId - : null; - - const headers: Record = { - Authorization: `Bearer ${accessToken}`, - "Content-Type": "application/json", - Accept: "application/json", - }; - if (accountId) { - headers["chatgpt-account-id"] = accountId; - } - - const response = await fetch(CODEX_CONFIG.usageUrl, { - method: "GET", - headers, - }); - - if (!response.ok) { - if (response.status === 401 || response.status === 403) { - return { - message: `Codex token expired or access denied. Please re-authenticate the connection.`, - }; - } - throw new Error(`Codex API error: ${response.status}`); - } - - const data = await response.json(); - - // Parse rate limit info (supports both snake_case and camelCase) - const rateLimit = toRecord(getFieldValue(data, "rate_limit", "rateLimit")); - const primaryWindow = toRecord(getFieldValue(rateLimit, "primary_window", "primaryWindow")); - const secondaryWindow = toRecord( - getFieldValue(rateLimit, "secondary_window", "secondaryWindow") - ); - - // Parse reset times (reset_at is Unix timestamp in seconds) - const parseWindowReset = (window: unknown) => { - const resetAt = toNumber(getFieldValue(window, "reset_at", "resetAt"), 0); - const resetAfterSeconds = toNumber( - getFieldValue(window, "reset_after_seconds", "resetAfterSeconds"), - 0 - ); - if (resetAt > 0) return parseResetTime(resetAt * 1000); - if (resetAfterSeconds > 0) return parseResetTime(Date.now() + resetAfterSeconds * 1000); - return null; - }; - - // Build quota windows - const quotas: Record = {}; - - // Primary window (5-hour) - if (Object.keys(primaryWindow).length > 0) { - const usedPercent = toNumber(getFieldValue(primaryWindow, "used_percent", "usedPercent"), 0); - quotas.session = { - used: usedPercent, - total: 100, - remaining: 100 - usedPercent, - resetAt: parseWindowReset(primaryWindow), - unlimited: false, - }; - } - - // Secondary window (weekly) - if (Object.keys(secondaryWindow).length > 0) { - const usedPercent = toNumber( - getFieldValue(secondaryWindow, "used_percent", "usedPercent"), - 0 - ); - quotas.weekly = { - used: usedPercent, - total: 100, - remaining: 100 - usedPercent, - resetAt: parseWindowReset(secondaryWindow), - unlimited: false, - }; - } - - // Code review rate limit (3rd window — differs per plan: Plus/Pro/Team) - const codeReviewRateLimit = toRecord( - getFieldValue(data, "code_review_rate_limit", "codeReviewRateLimit") - ); - const codeReviewWindow = toRecord( - getFieldValue(codeReviewRateLimit, "primary_window", "primaryWindow") - ); - - // Only include code review quota if the API returned data for it - const codeReviewUsedRaw = getFieldValue(codeReviewWindow, "used_percent", "usedPercent"); - const codeReviewRemainingRaw = getFieldValue( - codeReviewWindow, - "remaining_count", - "remainingCount" - ); - if (codeReviewUsedRaw !== null || codeReviewRemainingRaw !== null) { - const codeReviewUsedPercent = toNumber(codeReviewUsedRaw, 0); - quotas.code_review = { - used: codeReviewUsedPercent, - total: 100, - remaining: 100 - codeReviewUsedPercent, - resetAt: parseWindowReset(codeReviewWindow), - unlimited: false, - }; - } - - return { - plan: String(getFieldValue(data, "plan_type", "planType") || "unknown"), - limitReached: Boolean(getFieldValue(rateLimit, "limit_reached", "limitReached")), - quotas, - }; - } catch (error) { - return { message: `Failed to fetch Codex usage: ${(error as Error).message}` }; - } -} - -/** - * Build the Kiro usage result from a GetUsageLimits response. When the account returns no - * usage breakdown (some AWS IAM / Builder ID accounts don't expose per-resource quota via - * GetUsageLimits), return an informative message instead of empty `quotas:{}` — otherwise the - * dashboard renders a blank quota card with no explanation (#3506). Exported for testing. - */ -export function buildKiroUsageResult( - data: JsonRecord -): { plan: string; quotas: Record } | { message: string } { - const usageList = Array.isArray(data.usageBreakdownList) ? data.usageBreakdownList : []; - const quotaInfo: Record = {}; - const resetAt = parseResetTime(data.nextDateReset || data.resetDate); - - usageList.forEach((breakdownValue: unknown) => { - const breakdown = toRecord(breakdownValue); - const resourceType = - typeof breakdown.resourceType === "string" ? breakdown.resourceType.toLowerCase() : "unknown"; - const used = toNumber(breakdown.currentUsageWithPrecision, 0); - const total = toNumber(breakdown.usageLimitWithPrecision, 0); - - quotaInfo[resourceType] = { used, total, remaining: total - used, resetAt, unlimited: false }; - - const freeTrialInfo = toRecord(breakdown.freeTrialInfo); - if (Object.keys(freeTrialInfo).length > 0) { - const freeUsed = toNumber(freeTrialInfo.currentUsageWithPrecision, 0); - const freeTotal = toNumber(freeTrialInfo.usageLimitWithPrecision, 0); - quotaInfo[`${resourceType}_freetrial`] = { - used: freeUsed, - total: freeTotal, - remaining: freeTotal - freeUsed, - resetAt, - unlimited: false, - }; - } - }); - - if (Object.keys(quotaInfo).length === 0) { - return { - message: - "Kiro connected, but the account returned no usage breakdown. Some AWS IAM / Builder ID accounts don't expose per-resource quota via GetUsageLimits.", - }; - } - - return { - plan: String(toRecord(data.subscriptionInfo).subscriptionTitle || "").trim() || "Kiro", - quotas: quotaInfo, - }; -} - -/** - * Kiro (AWS CodeWhisperer) Usage - */ -async function getKiroUsage(accessToken?: string, providerSpecificData?: JsonRecord) { - try { - const profileArn = providerSpecificData?.profileArn; - if (!profileArn) { - return { message: "Kiro connected. Profile ARN not available for quota tracking." }; - } - - // Enterprise IAM Identity Center accounts are region-bound: the profileArn, token and - // endpoint must all match the region. Derive the region from the stored region (preferred) - // or the profileArn, then route to the regional Amazon Q endpoint (us-east-1 keeps the - // legacy codewhisperer host; codewhisperer.{region} does not resolve for other regions). - const regionFromArn = - typeof profileArn === "string" - ? profileArn.toLowerCase().match(/^arn:aws:codewhisperer:([a-z0-9-]+):/)?.[1] - : undefined; - const region = - (typeof providerSpecificData?.region === "string" && - providerSpecificData.region.trim().toLowerCase()) || - regionFromArn || - "us-east-1"; - const usageBaseUrl = - region === "us-east-1" ? CODEWHISPERER_BASE_URL : `https://q.${region}.amazonaws.com`; - - // Kiro uses AWS CodeWhisperer GetUsageLimits API - const payload = { - origin: "AI_EDITOR", - profileArn: profileArn, - resourceType: "AGENTIC_REQUEST", - }; - - const response = await fetch(usageBaseUrl, { - method: "POST", - headers: { - Authorization: `Bearer ${accessToken}`, - "Content-Type": "application/x-amz-json-1.0", - "x-amz-target": "AmazonCodeWhispererService.GetUsageLimits", - Accept: "application/json", - }, - body: JSON.stringify(payload), - }); - - if (!response.ok) { - const errorText = await response.text(); - throw new Error(`Kiro API error (${response.status}): ${errorText}`); - } - - const data = toRecord(await response.json()); - return buildKiroUsageResult(data); - } catch (error) { - throw new Error(`Failed to fetch Kiro usage: ${error.message}`); - } -} - -/** - * Map Kimi membership level to display name - * LEVEL_BASIC = Moderato, LEVEL_INTERMEDIATE = Allegretto, - * LEVEL_ADVANCED = Allegro, LEVEL_STANDARD = Vivace - */ -function getKimiPlanName(level: unknown): string { - if (!level) return ""; - const normalizedLevel = String(level); - - const levelMap = { - LEVEL_BASIC: "Moderato", - LEVEL_INTERMEDIATE: "Allegretto", - LEVEL_ADVANCED: "Allegro", - LEVEL_STANDARD: "Vivace", - }; - - return ( - levelMap[normalizedLevel as keyof typeof levelMap] || - normalizedLevel.replace("LEVEL_", "").toLowerCase() - ); -} - -/** - * Kimi Coding Usage - Fetch quota from Kimi API - * Uses the official /v1/usages endpoint with custom X-Msh-* headers - */ -async function getKimiUsage(accessToken?: string) { - // Generate device info for headers (same as OAuth flow) - const deviceId = "kimi-usage-" + Date.now(); - const platform = "omniroute"; - const version = "2.1.2"; - const deviceModel = - typeof process !== "undefined" ? `${process.platform} ${process.arch}` : "unknown"; - - try { - const response = await fetch(KIMI_CONFIG.usageUrl, { - method: "GET", - headers: { - Authorization: `Bearer ${accessToken}`, - "Content-Type": "application/json", - "X-Msh-Platform": platform, - "X-Msh-Version": version, - "X-Msh-Device-Model": deviceModel, - "X-Msh-Device-Id": deviceId, - }, - }); - - const responseText = await response.text(); - - if (!response.ok) { - return { - plan: "Kimi Coding", - message: `Kimi Coding connected. API Error ${response.status}: ${responseText.slice(0, 100)}`, - }; - } - - let data; - try { - data = JSON.parse(responseText); - } catch { - return { - plan: "Kimi Coding", - message: "Kimi Coding connected. Invalid JSON response from API.", - }; - } - - const quotas: Record = {}; - const dataObj = toRecord(data); - - // Parse Kimi usage response format - // Format: { user: {...}, usage: { limit: "100", used: "92", remaining: "8", resetTime: "..." }, limits: [...] } - const usageObj = toRecord(dataObj.usage); - - // Check for Kimi's actual usage fields (strings, not numbers) - const usageLimit = toNumber(usageObj.limit || usageObj.Limit, 0); - const usageUsed = toNumber(usageObj.used || usageObj.Used, 0); - const usageRemaining = toNumber(usageObj.remaining || usageObj.Remaining, 0); - const usageResetTime = - usageObj.resetTime || usageObj.ResetTime || usageObj.reset_at || usageObj.resetAt; - - if (usageLimit > 0) { - const percentRemaining = usageLimit > 0 ? (usageRemaining / usageLimit) * 100 : 0; - - quotas["Weekly"] = { - used: usageUsed, - total: usageLimit, - remaining: usageRemaining, - remainingPercentage: percentRemaining, - resetAt: parseResetTime(usageResetTime), - unlimited: false, - }; - } - - // Also parse limits array for rate limits - const limitsArray = Array.isArray(dataObj.limits) ? dataObj.limits : []; - for (let i = 0; i < limitsArray.length; i++) { - const limitItem = toRecord(limitsArray[i]); - const window = toRecord(limitItem.window); - const detail = toRecord(limitItem.detail); - - const limit = toNumber(detail.limit || detail.Limit, 0); - const remaining = toNumber(detail.remaining || detail.Remaining, 0); - const resetTime = detail.resetTime || detail.reset_at || detail.resetAt; - - if (limit > 0) { - quotas["Ratelimit"] = { - used: limit - remaining, - total: limit, - remaining, - remainingPercentage: limit > 0 ? (remaining / limit) * 100 : 0, - resetAt: parseResetTime(resetTime), - unlimited: false, - }; - } - } - - // Check for quota windows (Claude-like format with utilization) as fallback - const hasUtilization = (window: JsonRecord) => - window && typeof window === "object" && safePercentage(window.utilization) !== undefined; - - const createQuotaObject = (window: JsonRecord) => { - const remaining = safePercentage(window.utilization) as number; - const used = 100 - remaining; - return { - used, - total: 100, - remaining, - resetAt: parseResetTime(window.resets_at), - remainingPercentage: remaining, - unlimited: false, - }; - }; - - if (hasUtilization(toRecord(dataObj.five_hour))) { - quotas["session (5h)"] = createQuotaObject(toRecord(dataObj.five_hour)); - } - - if (hasUtilization(toRecord(dataObj.seven_day))) { - quotas["weekly (7d)"] = createQuotaObject(toRecord(dataObj.seven_day)); - } - - // Check for model-specific quotas - for (const [key, value] of Object.entries(dataObj)) { - const valueRecord = toRecord(value); - if (key.startsWith("seven_day_") && key !== "seven_day" && hasUtilization(valueRecord)) { - const modelName = key.replace("seven_day_", ""); - quotas[`weekly ${modelName} (7d)`] = createQuotaObject(valueRecord); - } - } - - if (Object.keys(quotas).length > 0) { - const userRecord = toRecord(dataObj.user); - const membershipLevel = toRecord(userRecord.membership).level; - const planName = getKimiPlanName(membershipLevel); - return { - plan: planName || "Kimi Coding", - quotas, - }; - } - - // No quota data in response - const userRecord = toRecord(dataObj.user); - const membershipLevel = toRecord(userRecord.membership).level; - const planName = getKimiPlanName(membershipLevel); - return { - plan: planName || "Kimi Coding", - message: "Kimi Coding connected. Usage tracked per request.", - }; - } catch (error) { - return { - message: `Kimi Coding connected. Unable to fetch usage: ${(error as Error).message}`, - }; - } -} - -/** - * Qwen Usage - */ -async function getQwenUsage(accessToken?: string, providerSpecificData?: JsonRecord) { - void accessToken; - try { - const resourceUrl = providerSpecificData?.resourceUrl; - if (!resourceUrl) { - return { message: "Qwen connected. No resource URL available." }; - } - - // Qwen may have usage endpoint at resource URL - return { message: "Qwen connected. Usage tracked per request." }; - } catch (error) { - return { message: "Unable to fetch Qwen usage." }; - } -} - -/** - * Qoder Usage - */ -async function getQoderUsage(accessToken?: string) { - void accessToken; - try { - // Qoder may have usage endpoint - return { message: "Qoder connected. Usage tracked per request." }; - } catch (error) { - return { message: "Unable to fetch Qoder usage." }; - } -} - -export const __testing = { - parseResetTime, - formatGitHubQuotaSnapshot, - inferGitHubPlanName, - getGeminiCliPlanLabel, - getAntigravityPlanLabel, - extractCodeAssistSubscriptionTier, - extractCodeAssistOnboardTierId, - getMiniMaxPlanLabel, - inferMiniMaxPlanLabelFromTotals, - getOpencodeUsage, - getClaudePlanLabel, - createQuotaFromUsage, - getMiniMaxQuotaResetAt, - isMiniMaxTextQuotaModel, - getMiniMaxSessionTotal, - getMiniMaxWeeklyTotal, - createMiniMaxQuotaFromCount, - createMiniMaxQuotaFromPercent, - getMiniMaxRemainingPercent, - getMiniMaxUsage, - getXiaomiMimoUsage, - getMiniMaxAuthErrorMessage, - getMiniMaxErrorSummary, - mapCodeAssistSubscriptionToPlanLabel, - mapCodeAssistTierIdToLabel, - mapSubscriptionTierStringToPlanLabel, - toDisplayLabel, -}; +export * from "./usage/index.ts"; diff --git a/open-sse/services/usage/antigravity.ts b/open-sse/services/usage/antigravity.ts new file mode 100644 index 00000000000..3b4d7fe47b2 --- /dev/null +++ b/open-sse/services/usage/antigravity.ts @@ -0,0 +1,867 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { + getAntigravityFetchAvailableModelsUrls, + ANTIGRAVITY_BASE_URLS, +} from "../../config/antigravityUpstream.ts"; + +import { + isUserCallableAntigravityModelId, + toClientAntigravityModelId, +} from "../../config/antigravityModelAliases.ts"; + +import { isUserCallableAgyModelId } from "../../config/agyModels.ts"; + +import { getDbInstance } from "@/lib/db/core"; + +import { + applyAntigravityClientProfileHeaders, + getAntigravityBootstrapHeaders, + getAntigravityClientProfile, +} from "../antigravityClientProfile.ts"; + +import { + antigravityUserAgent, + getAntigravityHeaders, + getAntigravityLoadCodeAssistMetadata, +} from "../antigravityHeaders.ts"; + +import { + getAntigravityRemainingCredits, + updateAntigravityRemainingCredits, +} from "../../executors/antigravity.ts"; + +import { getCreditsMode } from "../antigravityCredits.ts"; + +import { generateAntigravityRequestId, getAntigravitySessionId } from "../antigravityIdentity.ts"; + +import { + extractCodeAssistOnboardTierId, + extractCodeAssistSubscriptionTier, +} from "../codeAssistSubscription.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { SubscriptionCacheEntry, UsageQuota, AntigravityUsageOptions, JsonRecord } from "./types.ts"; +import { _geminiCliSubCache, GEMINI_CLI_CACHE_TTL_MS } from "./gemini.ts"; +import { ANTIGRAVITY_CONFIG } from "./constants.ts"; +import { toRecord, getFieldValue, toNumber, parseResetTime } from "./utils.ts"; + + + +// ── Antigravity subscription info cache ────────────────────────────────────── +// Prevents duplicate loadCodeAssist calls within the same quota cycle. +// Key: truncated accessToken → { data, fetchedAt } +const _antigravitySubCache = new Map(); + + +const ANTIGRAVITY_CACHE_TTL_MS = 5 * 60 * 1000; + + // 5 minutes +const ANTIGRAVITY_MODELS_CACHE_TTL_MS = 60 * 1000; + + +const ANTIGRAVITY_CREDIT_PROBE_TTL_MS = 5 * 60 * 1000; + + +const _antigravityAvailableModelsCache = new Map(); + + +const _antigravityAvailableModelsInflight = new Map>(); + + +const _antigravityUserQuotaCache = new Map(); + + +const _antigravityUserQuotaInflight = new Map>(); + + +const _antigravityCreditProbeCache = new Map(); + + +const _antigravityCreditProbeInflight = new Map>(); + + + +// ── Proactive TTL purging for module-level caches ────────────────────────── +// All 4 data caches only evict on read (passive TTL). This interval proactively +// purges stale entries so keys accessed once and never again don't leak memory. +// The 2 inflight Maps (availableModelsInflight, creditProbeInflight) self-clean +// when the Promise resolves/rejects, so they are NOT touched here. +const _usageCacheCleanupTimer = setInterval( + () => { + const now = Date.now(); + for (const [key, entry] of _geminiCliSubCache) { + if (now - entry.fetchedAt > GEMINI_CLI_CACHE_TTL_MS) _geminiCliSubCache.delete(key); + } + for (const [key, entry] of _antigravitySubCache) { + if (now - entry.fetchedAt > ANTIGRAVITY_CACHE_TTL_MS) _antigravitySubCache.delete(key); + } + for (const [key, entry] of _antigravityAvailableModelsCache) { + if (now - entry.fetchedAt > ANTIGRAVITY_MODELS_CACHE_TTL_MS) + _antigravityAvailableModelsCache.delete(key); + } + for (const [key, entry] of _antigravityUserQuotaCache) { + if (now - entry.fetchedAt > ANTIGRAVITY_MODELS_CACHE_TTL_MS) + _antigravityUserQuotaCache.delete(key); + } + for (const [key, entry] of _antigravityCreditProbeCache) { + if (now - entry.fetchedAt > ANTIGRAVITY_CREDIT_PROBE_TTL_MS) + _antigravityCreditProbeCache.delete(key); + } + }, + 5 * 60 * 1000 +); + + // every 5 minutes +_usageCacheCleanupTimer.unref?.(); + + + +const ANTIGRAVITY_LOCAL_USAGE_WINDOW_MS = 5 * 60 * 60 * 1000; + + +const ANTIGRAVITY_LOCAL_USAGE_TOKENS_PER_UNIT = 1000; + + + +const ANTIGRAVITY_QUOTA_MODEL_ALIASES: Record = { + "gemini-3.5-flash-preview": null, + "gemini-3-flash-preview": null, +}; + + + +function normalizeAntigravityQuotaModelId(modelId: string): string | null { + if (!modelId) return null; + return Object.prototype.hasOwnProperty.call(ANTIGRAVITY_QUOTA_MODEL_ALIASES, modelId) + ? ANTIGRAVITY_QUOTA_MODEL_ALIASES[modelId] + : modelId; +} + + + +function toClientAntigravityQuotaModelId(modelId: string): string | null { + if (!modelId) return null; + if (normalizeAntigravityQuotaModelId(modelId) === null) return null; + if (modelId === "gemini-3.5-flash-extra-low") return "gemini-3.5-flash-low"; + if (modelId === "gemini-3.5-flash-low") return "gemini-3.5-flash-medium"; + if (modelId === "gemini-3-flash-agent") return "gemini-3.5-flash-high"; + return toClientAntigravityModelId(modelId); +} + + + +function getAntigravityLocalUsageUnits( + provider: "antigravity" | "agy", + connectionId: string | undefined, + modelId: string, + resetAt: string | null +): number { + if (!connectionId || !modelId || !resetAt) return 0; + + const resetMs = Date.parse(resetAt); + if (!Number.isFinite(resetMs)) return 0; + + const windowStart = new Date(resetMs - ANTIGRAVITY_LOCAL_USAGE_WINDOW_MS).toISOString(); + const windowEnd = new Date(resetMs).toISOString(); + + try { + const db = getDbInstance() as unknown as { + prepare: (sql: string) => { get: (...params: unknown[]) => unknown }; + }; + const row = db + .prepare( + `SELECT COALESCE(SUM( + COALESCE(tokens_input, 0) + COALESCE(tokens_output, 0) + COALESCE(tokens_reasoning, 0) + ), 0) AS tokens + FROM usage_history + WHERE provider = ? + AND connection_id = ? + AND model = ? + AND success = 1 + AND timestamp >= ? + AND timestamp < ?` + ) + .get(provider, connectionId, modelId, windowStart, windowEnd) as + | { tokens?: unknown } + | undefined; + + const tokens = Number(row?.tokens || 0); + if (!Number.isFinite(tokens) || tokens <= 0) return 0; + return Math.max(1, Math.ceil(tokens / ANTIGRAVITY_LOCAL_USAGE_TOKENS_PER_UNIT)); + } catch { + return 0; + } +} + + + +function applyLocalUsageFallback( + quota: UsageQuota, + provider: "antigravity" | "agy", + connectionId: string | undefined, + modelId: string +): UsageQuota { + if (quota.quotaSource !== "fetchAvailableModels" || quota.used > 0 || quota.unlimited) { + return quota; + } + + const localUsed = getAntigravityLocalUsageUnits(provider, connectionId, modelId, quota.resetAt); + if (localUsed <= 0 || quota.total <= 0) return quota; + + const used = Math.min(quota.total, localUsed); + return { + ...quota, + used, + remainingPercentage: Math.max(0, ((quota.total - used) / quota.total) * 100), + quotaSource: "localUsageHistory", + }; +} + + + +function buildAntigravityUsageCacheKey(accessToken: string, projectId?: string | null): string { + return `${accessToken.substring(0, 16)}:${projectId || "default"}`; +} + + + +async function fetchAntigravityAvailableModelsCached( + accessToken: string, + projectId?: string | null, + options: AntigravityUsageOptions = {} +): Promise { + if (!accessToken) throw new Error("Access token is required"); + + const cacheKey = buildAntigravityUsageCacheKey(accessToken, projectId); + const cached = _antigravityAvailableModelsCache.get(cacheKey); + if ( + !options.forceRefresh && + cached && + Date.now() - cached.fetchedAt < ANTIGRAVITY_MODELS_CACHE_TTL_MS + ) { + return cached.data; + } + + const inflight = _antigravityAvailableModelsInflight.get(cacheKey); + if (inflight) return inflight; + + const promise = (async () => { + let response: Response | null = null; + let lastError: Error | null = null; + + for (const quotaApiUrl of ANTIGRAVITY_CONFIG.quotaApiUrls) { + try { + response = await fetch(quotaApiUrl, { + method: "POST", + headers: getAntigravityHeaders("fetchAvailableModels", accessToken), + body: JSON.stringify(projectId ? { project: projectId } : {}), + signal: AbortSignal.timeout(10000), + }); + + if (response.ok || response.status === 401 || response.status === 403) { + break; + } + } catch (error) { + lastError = error as Error; + } + } + + if (!response) { + throw lastError || new Error("Antigravity API unavailable"); + } + + if (response.status === 403) { + return { __antigravityForbidden: true }; + } + + if (!response.ok) { + throw new Error(`Antigravity API error: ${response.status}`); + } + + const data = await response.json(); + _antigravityAvailableModelsCache.set(cacheKey, { data, fetchedAt: Date.now() }); + return data; + })().finally(() => { + _antigravityAvailableModelsInflight.delete(cacheKey); + }); + + _antigravityAvailableModelsInflight.set(cacheKey, promise); + return promise; +} + + + +async function fetchAntigravityUserQuotaCached( + accessToken: string, + projectId?: string | null, + options: AntigravityUsageOptions = {} +): Promise { + if (!accessToken || !projectId) return null; + + const cacheKey = buildAntigravityUsageCacheKey(accessToken, projectId); + const cached = _antigravityUserQuotaCache.get(cacheKey); + if ( + !options.forceRefresh && + cached && + Date.now() - cached.fetchedAt < ANTIGRAVITY_MODELS_CACHE_TTL_MS + ) { + return cached.data; + } + + const inflight = _antigravityUserQuotaInflight.get(cacheKey); + if (inflight) return inflight; + + const promise = (async () => { + try { + const response = await fetch( + "https://cloudcode-pa.googleapis.com/v1internal:retrieveUserQuota", + { + method: "POST", + headers: { + Authorization: `Bearer ${accessToken}`, + "Content-Type": "application/json", + }, + body: JSON.stringify({ project: projectId }), + signal: AbortSignal.timeout(10000), + } + ); + + if (!response.ok) return null; + + const data = await response.json(); + _antigravityUserQuotaCache.set(cacheKey, { data, fetchedAt: Date.now() }); + return data; + } catch { + return null; + } + })().finally(() => { + _antigravityUserQuotaInflight.delete(cacheKey); + }); + + _antigravityUserQuotaInflight.set(cacheKey, promise); + return promise; +} + + + +function extractCodeAssistTierId(subscription: JsonRecord): string { + const tierId = extractCodeAssistOnboardTierId(subscription); + if (tierId === "legacy-tier") return ""; + const upper = tierId.toUpperCase(); + return mapCodeAssistTierIdToLabel(upper) ? upper : ""; +} + + + +export function mapCodeAssistTierIdToLabel(tierId: string): string | null { + const upper = tierId.toUpperCase(); + if (upper.includes("ULTRA")) return "Ultra"; + if ( + upper.includes("PRO") || + upper.includes("PREMIUM") || + upper.includes("GOOGLE_ONE") || + upper.includes("ONE_AI") + ) + return "Pro"; + if (upper.includes("ENTERPRISE")) return "Enterprise"; + if (upper.includes("BUSINESS") || upper.includes("STANDARD")) return "Business"; + if (upper.includes("PLUS")) return "Plus"; + if (upper.includes("LITE") || upper.includes("LIGHT")) return "Lite"; + if (upper.includes("FREE") || upper.includes("INDIVIDUAL") || upper.includes("LEGACY")) + return "Free"; + return null; +} + + + +export function mapSubscriptionTierStringToPlanLabel(tierText: string): string | null { + const upper = tierText.toUpperCase(); + if (upper.includes("ULTRA")) return "Ultra"; + if (upper.includes("PRO") || upper.includes("PREMIUM") || upper.includes("GOOGLE ONE")) + return "Pro"; + if (upper.includes("ENTERPRISE")) return "Enterprise"; + if (upper.includes("STANDARD") || upper.includes("BUSINESS")) return "Business"; + if (upper.includes("PLUS")) return "Plus"; + if (upper.includes("LITE")) return "Lite"; + if (upper.includes("INDIVIDUAL") || upper.includes("FREE")) return "Free"; + // Strip a trailing "(RESTRICTED)" marker. Match the fixed literal anywhere then + // trim, instead of /\s*\(RESTRICTED\)\s*$/ whose overlapping \s* runs backtrack + // polynomially on whitespace-heavy upstream input (js/polynomial-redos). + const normalizedId = upper.replace(/\(RESTRICTED\)/i, "").trim(); + if (normalizedId) { + const mapped = mapCodeAssistTierIdToLabel(normalizedId); + if (mapped) return mapped; + } + return null; +} + + + +export function mapCodeAssistSubscriptionToPlanLabel(subscriptionInfo: unknown): string { + const subscription = toRecord(subscriptionInfo); + if (Object.keys(subscription).length === 0) return "Free"; + + const subscriptionTier = extractCodeAssistSubscriptionTier(subscriptionInfo); + if (subscriptionTier) { + const mapped = mapSubscriptionTierStringToPlanLabel(subscriptionTier); + if (mapped) return mapped; + if (subscriptionTier.toLowerCase() !== "free") { + return subscriptionTier.charAt(0).toUpperCase() + subscriptionTier.slice(1).toLowerCase(); + } + } + + const currentTier = toRecord(subscription.currentTier); + const tierName = String( + getFieldValue(currentTier, "name", "displayName") || + subscription.subscriptionType || + subscription.tier || + "" + ); + const mappedName = tierName ? mapSubscriptionTierStringToPlanLabel(tierName) : null; + if (mappedName) return mappedName; + + const tierId = extractCodeAssistTierId(subscription); + if (tierId) { + const mapped = mapCodeAssistTierIdToLabel(tierId); + if (mapped) return mapped; + } + if (currentTier.upgradeSubscriptionType) return "Free"; + if (tierName) return tierName.charAt(0).toUpperCase() + tierName.slice(1).toLowerCase(); + return "Free"; +} + + + +const KNOWN_ANTIGRAVITY_PLAN_LABELS = new Set([ + "Ultra", + "Pro", + "Enterprise", + "Business", + "Plus", + "Lite", +]); + + + +/** + * Map raw loadCodeAssist tier data to short display labels (Antigravity Manager parity). + */ +export function getAntigravityPlanLabel(subscriptionInfo: unknown, fallbackInfo?: unknown): string { + const livePlan = mapCodeAssistSubscriptionToPlanLabel(subscriptionInfo); + const fallbackPlan = mapCodeAssistSubscriptionToPlanLabel(fallbackInfo); + + if (KNOWN_ANTIGRAVITY_PLAN_LABELS.has(livePlan)) return livePlan; + if (KNOWN_ANTIGRAVITY_PLAN_LABELS.has(fallbackPlan)) return fallbackPlan; + if (livePlan !== "Free") return livePlan; + return fallbackPlan !== "Free" ? fallbackPlan : livePlan; +} + + + +/** + * Proactive credit balance probe for Antigravity. + * + * Fires a minimal streamGenerateContent request with GOOGLE_ONE_AI credits enabled + * and maxOutputTokens=1 to extract the `remainingCredits` field from the SSE stream. + * This uses ~1 credit but lets us show the balance on the dashboard without waiting + * for a real user request. + * + * Returns the credit balance, or null if the probe failed. + */ +async function probeAntigravityCreditBalance( + accessToken: string, + accountId: string, + projectId?: string | null, + options: AntigravityUsageOptions = {}, + providerSpecificData: JsonRecord = {} +): Promise { + if (!accessToken) return null; + + const cacheKey = buildAntigravityUsageCacheKey(accessToken, projectId || accountId); + const cached = _antigravityCreditProbeCache.get(cacheKey); + if ( + !options.forceRefresh && + cached && + Date.now() - cached.fetchedAt < ANTIGRAVITY_CREDIT_PROBE_TTL_MS + ) { + return cached.data; + } + + const inflight = _antigravityCreditProbeInflight.get(cacheKey); + if (inflight) return inflight; + + const promise = probeAntigravityCreditBalanceUncached( + accessToken, + accountId, + projectId, + providerSpecificData + ) + .then( + (data) => { + _antigravityCreditProbeCache.set(cacheKey, { data, fetchedAt: Date.now() }); + return data; + }, + (error) => { + _antigravityCreditProbeCache.set(cacheKey, { data: null, fetchedAt: Date.now() }); + throw error; + } + ) + .finally(() => { + _antigravityCreditProbeInflight.delete(cacheKey); + }); + + _antigravityCreditProbeInflight.set(cacheKey, promise); + return promise; +} + + + +async function probeAntigravityCreditBalanceUncached( + accessToken: string, + accountId: string, + projectId?: string | null, + providerSpecificData: JsonRecord = {} +): Promise { + try { + if (!projectId) return null; + + // Try all base URLs (some accounts only work with specific endpoints) + for (const baseUrl of ANTIGRAVITY_BASE_URLS) { + const url = `${baseUrl}/v1internal:streamGenerateContent?alt=sse`; + + const sessionId = getAntigravitySessionId({ connectionId: accountId, projectId }); + const body = { + project: projectId, + model: "gemini-2-flash", + userAgent: "antigravity", + requestType: "agent", + requestId: generateAntigravityRequestId(), + enabledCreditTypes: ["GOOGLE_ONE_AI"], + request: { + model: "gemini-2-flash", + contents: [{ role: "user", parts: [{ text: "hi" }] }], + generationConfig: { maxOutputTokens: 1 }, + sessionId, + }, + }; + + const headers: Record = { + "Content-Type": "application/json", + Authorization: `Bearer ${accessToken}`, + Accept: "text/event-stream", + }; + applyAntigravityClientProfileHeaders( + headers, + { connectionId: accountId, projectId, providerSpecificData }, + body + ); + + try { + const res = await fetch(url, { + method: "POST", + headers, + body: JSON.stringify(body), + signal: AbortSignal.timeout(10_000), + }); + + if (!res.ok) continue; + + // Read the full SSE response and scan for remainingCredits + const rawSSE = await res.text(); + const lines = rawSSE.split("\n"); + + for (const line of lines) { + const trimmed = line.trim(); + if (!trimmed.startsWith("data:")) continue; + const payload = trimmed.slice(5).trim(); + if (payload === "[DONE]") break; + try { + const parsed = JSON.parse(payload); + if (Array.isArray(parsed?.remainingCredits)) { + const googleCredit = parsed.remainingCredits.find( + (c: { creditType?: string }) => c?.creditType === "GOOGLE_ONE_AI" + ); + if (googleCredit) { + const balance = parseInt(googleCredit.creditAmount, 10); + if (!isNaN(balance)) { + updateAntigravityRemainingCredits(accountId, balance); + return balance; + } + } + } + } catch { + // Skip malformed SSE lines + } + } + } catch { + // Individual endpoint failure; try next + } + } + + return null; + } catch { + // Probe is best-effort — don't let it break the usage fetch + return null; + } +} + + + +/** + * Antigravity Usage - Fetch quota from Google Cloud Code API. + * fetchAvailableModels is catalog/eligibility data and may keep reporting full buckets + * after real usage. retrieveUserQuota is the consumption signal for Gemini-family + * buckets, so prefer it when present and fall back to fetchAvailableModels only for + * models that have no retrieveUserQuota entry (for example Claude/GPT OSS buckets). + */ +export async function getAntigravityUsage( + provider: "antigravity" | "agy", + accessToken?: string, + providerSpecificData?: JsonRecord, + connectionProjectId?: string, + connectionId?: string, + options: AntigravityUsageOptions = {} +) { + if (!accessToken) { + return { plan: "Free", message: "Antigravity access token not available." }; + } + + let subscriptionInfo: unknown = null; + try { + subscriptionInfo = await getAntigravitySubscriptionInfoCached( + accessToken, + providerSpecificData, + options + ); + const savedProjectId = + typeof providerSpecificData?.projectId === "string" && providerSpecificData.projectId.trim() + ? providerSpecificData.projectId.trim() + : null; + const subscriptionProject = toRecord(subscriptionInfo).cloudaicompanionProject; + const projectId = + savedProjectId || + connectionProjectId || + (typeof subscriptionProject === "string" + ? subscriptionProject + : typeof toRecord(subscriptionProject).id === "string" + ? (toRecord(subscriptionProject).id as string) + : null); + + // Derive accountId for credit balance cache. + // Must match executor key: credentials.connectionId + const accountId: string = connectionId || "unknown"; + + // Read cached credit balance (hydrated from DB on first access) + let creditBalance = getAntigravityRemainingCredits(accountId); + + // If no cached balance and credits mode is enabled, fire a minimal probe + const creditsMode = getCreditsMode(); + if ((options.forceRefresh || creditBalance === null) && creditsMode !== "off") { + creditBalance = await probeAntigravityCreditBalance( + accessToken, + accountId, + projectId, + options, + providerSpecificData || {} + ); + } + + const [data, userQuotaData] = await Promise.all([ + fetchAntigravityAvailableModelsCached(accessToken, projectId, options), + fetchAntigravityUserQuotaCached(accessToken, projectId, options), + ]); + const dataObj = toRecord(data); + if (dataObj.__antigravityForbidden === true) { + return { message: "Antigravity access forbidden. Check subscription." }; + } + const modelEntries = toRecord(dataObj.models); + const userQuotaEntries = new Map(); + const userQuotaObj = toRecord(userQuotaData); + if (Array.isArray(userQuotaObj.buckets)) { + for (const bucketValue of userQuotaObj.buckets) { + const bucket = toRecord(bucketValue); + const modelId = toClientAntigravityQuotaModelId(String(bucket.modelId || "").trim()); + if (!modelId) continue; + userQuotaEntries.set(modelId, bucket); + } + } + const quotas: Record = {}; + + // Parse per-model quota info from fetchAvailableModels response. + for (const [rawModelKey, infoValue] of Object.entries(modelEntries)) { + const info = toRecord(infoValue); + const quotaInfo = toRecord(info.quotaInfo); + const modelKey = toClientAntigravityQuotaModelId(rawModelKey); + + // Skip internal, excluded, and models without quota info + if ( + !modelKey || + info.isInternal === true || + !(provider === "agy" + ? isUserCallableAgyModelId(modelKey) + : isUserCallableAntigravityModelId(modelKey)) || + Object.keys(quotaInfo).length === 0 + ) { + continue; + } + + const liveQuota = userQuotaEntries.get(modelKey); + const quotaSource = liveQuota || quotaInfo; + const rawFraction = toNumber(quotaSource.remainingFraction, -1); + const resetAt = parseResetTime(quotaSource.resetTime); + // Distinguish "upstream did not report remainingFraction" from "remaining is 0%". + // fetchAvailableModels is a catalog view and can be stale/full; retrieveUserQuota is + // the source of truth for actual Gemini consumption when it includes the model. + const fractionReported = rawFraction >= 0; + if (!fractionReported) { + console.warn( + `[Antigravity] model ${modelKey} returned no remainingFraction — quota unknown` + ); + } + const remainingFraction = fractionReported ? Math.max(0, Math.min(1, rawFraction)) : 0; + // Models with no resetTime AND a reported full fraction are unlimited + // (e.g. tab-completion models). Unreported fraction is NEVER unlimited. + const isUnlimited = fractionReported && !resetAt && remainingFraction >= 1; + const remainingPercentage = remainingFraction * 100; + const QUOTA_NORMALIZED_BASE = 1000; + const total = QUOTA_NORMALIZED_BASE; + const remaining = Math.round(total * remainingFraction); + const used = isUnlimited ? 0 : Math.max(0, total - remaining); + + quotas[modelKey] = applyLocalUsageFallback( + { + used, + total: isUnlimited ? 0 : total, + resetAt, + remainingPercentage: isUnlimited ? 100 : remainingPercentage, + unlimited: isUnlimited, + fractionReported, + quotaSource: liveQuota ? "retrieveUserQuota" : "fetchAvailableModels", + }, + provider, + connectionId, + modelKey + ); + } + + // Include retrieveUserQuota buckets not listed in the static/public Antigravity catalog yet. + // This keeps Provider Limits honest when Google adds a new Gemini tier before our catalog is + // updated. Hidden/internal catalog entries above are still filtered by the public pass. + for (const [modelKey, bucket] of userQuotaEntries) { + if ( + quotas[modelKey] || + !(provider === "agy" + ? isUserCallableAgyModelId(modelKey) + : isUserCallableAntigravityModelId(modelKey)) + ) { + continue; + } + const rawFraction = toNumber(bucket.remainingFraction, -1); + if (rawFraction < 0) continue; + const remainingFraction = Math.max(0, Math.min(1, rawFraction)); + const resetAt = parseResetTime(bucket.resetTime); + const isUnlimited = !resetAt && remainingFraction >= 1; + const QUOTA_NORMALIZED_BASE = 1000; + const total = QUOTA_NORMALIZED_BASE; + const remaining = Math.round(total * remainingFraction); + quotas[modelKey] = { + used: isUnlimited ? 0 : Math.max(0, total - remaining), + total: isUnlimited ? 0 : total, + resetAt, + remainingPercentage: isUnlimited ? 100 : remainingFraction * 100, + unlimited: isUnlimited, + fractionReported: true, + quotaSource: "retrieveUserQuota", + }; + } + + return { + plan: getAntigravityPlanLabel(subscriptionInfo, providerSpecificData), + quotas: { + ...quotas, + ...(creditBalance !== null && { + credits: { + used: 0, + total: 0, + remaining: creditBalance, + unlimited: false, + resetAt: null, + }, + }), + }, + subscriptionInfo, + }; + } catch (error) { + return { + plan: getAntigravityPlanLabel(subscriptionInfo, providerSpecificData), + subscriptionInfo, + message: `Antigravity error: ${(error as Error).message}`, + }; + } +} + + + +/** + * Get Antigravity subscription info (cached, 5 min TTL) + * Prevents duplicate loadCodeAssist calls within the same quota cycle. + */ +async function getAntigravitySubscriptionInfoCached( + accessToken: string, + providerSpecificData?: JsonRecord, + options: AntigravityUsageOptions = {} +): Promise { + const profile = getAntigravityClientProfile({ providerSpecificData }); + const cacheKey = `${accessToken.substring(0, 16)}:${profile}`; + + if (options.forceRefresh) { + _antigravitySubCache.delete(cacheKey); + } else { + const cached = _antigravitySubCache.get(cacheKey); + if (cached && Date.now() - cached.fetchedAt < ANTIGRAVITY_CACHE_TTL_MS) { + return cached.data; + } + } + + const data = await getAntigravitySubscriptionInfo(accessToken, providerSpecificData); + if (data != null) { + _antigravitySubCache.set(cacheKey, { data, fetchedAt: Date.now() }); + } + return data; +} + + + +/** + * Get Antigravity subscription info using correct Antigravity headers. + * Must match the headers used in providers.js postExchange (not CLI headers). + */ +async function getAntigravitySubscriptionInfo( + accessToken: string, + providerSpecificData?: JsonRecord +): Promise { + try { + const profile = getAntigravityClientProfile({ providerSpecificData }); + const response = await fetch(ANTIGRAVITY_CONFIG.loadProjectApiUrl, { + method: "POST", + headers: + profile === "harness" + ? getAntigravityBootstrapHeaders(profile, accessToken) + : getAntigravityHeaders("loadCodeAssist", accessToken), + body: JSON.stringify({ metadata: getAntigravityLoadCodeAssistMetadata() }), + }); + + if (!response.ok) return null; + + return await response.json(); + } catch { + return null; + } +} + diff --git a/open-sse/services/usage/bailian.ts b/open-sse/services/usage/bailian.ts new file mode 100644 index 00000000000..6fcb38af588 --- /dev/null +++ b/open-sse/services/usage/bailian.ts @@ -0,0 +1,51 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { fetchBailianQuota, type BailianTripleWindowQuota } from "../bailianQuotaFetcher.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + + + + +/** + * Bailian (Alibaba Coding Plan) Usage + * Fetches triple-window quota (5h, weekly, monthly) and returns worst-case. + */ +export async function getBailianCodingPlanUsage( + connectionId: string, + apiKey: string, + providerSpecificData?: Record +) { + try { + const connection = { apiKey, providerSpecificData }; + const quota = await fetchBailianQuota(connectionId, connection); + + if (!quota) { + return { message: "Bailian Coding Plan connected. Unable to fetch quota." }; + } + + const bailianQuota = quota as BailianTripleWindowQuota; + const used = bailianQuota.used; + const total = bailianQuota.total; + const remaining = Math.max(0, total - used); + const remainingPercentage = Math.round(remaining); + + return { + plan: "Alibaba Coding Plan", + used, + total, + remaining, + remainingPercentage, + resetAt: bailianQuota.resetAt, + unlimited: false, + displayName: "Alibaba Coding Plan", + }; + } catch (error) { + return { message: `Bailian Coding Plan error: ${(error as Error).message}` }; + } +} + diff --git a/open-sse/services/usage/claude.ts b/open-sse/services/usage/claude.ts new file mode 100644 index 00000000000..388a9ad45ff --- /dev/null +++ b/open-sse/services/usage/claude.ts @@ -0,0 +1,201 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { safePercentage } from "@/shared/utils/formatting"; + +import { CLAUDE_CODE_VERSION, fetchClaudeBootstrap } from "../../executors/claudeIdentity.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { CLAUDE_CONFIG } from "./constants.ts"; +import { toRecord, parseResetTime } from "./utils.ts"; +import { UsageQuota, JsonRecord } from "./types.ts"; + + + +export function getClaudePlanLabel(...candidates: Array): string | null { + for (const candidate of candidates) { + if (typeof candidate !== "string") continue; + const trimmed = candidate.trim(); + if ( + !trimmed || + trimmed.toLowerCase() === "claude code" || + trimmed.toLowerCase() === "unknown" + ) { + continue; + } + return trimmed; + } + return null; +} + + + +/** + * Claude Usage - Try to fetch from Anthropic API + */ +export async function getClaudeUsage(accessToken?: string) { + if (!accessToken) { + return { message: "Claude connected. Access token not available.", bootstrap: null }; + } + + // Refresh bootstrap in parallel; best-effort, failure non-fatal. + const bootstrapPromise = fetchClaudeBootstrap(accessToken).catch(() => null); + try { + // Real CLI uses axios here, not Stainless — UA is `claude-code/` + // (not `claude-cli/...`) and the shape is simpler than /v1/messages. + const ctrl = new AbortController(); + const timer = setTimeout(() => ctrl.abort(), 10_000); + let oauthResponse; + try { + oauthResponse = await fetch(CLAUDE_CONFIG.oauthUsageUrl, { + method: "GET", + headers: { + Accept: "application/json, text/plain, */*", + "Accept-Encoding": "gzip, compress, deflate, br", + Authorization: `Bearer ${accessToken}`, + "Content-Type": "application/json", + "User-Agent": `claude-code/${CLAUDE_CODE_VERSION}`, + "anthropic-beta": "oauth-2025-04-20", + }, + signal: ctrl.signal, + }); + } finally { + clearTimeout(timer); + } + + if (oauthResponse.ok) { + const data = toRecord(await oauthResponse.json()); + const quotas: Record = {}; + + // utilization = percentage USED (e.g., 90 means 90% used, 10% remaining) + // Confirmed via user report #299: Claude.ai shows 87% used = OmniRoute must show 13% remaining. + const hasUtilization = (window: JsonRecord) => + window && typeof window === "object" && safePercentage(window.utilization) !== undefined; + + const createQuotaObject = (window: JsonRecord) => { + const used = safePercentage(window.utilization) as number; // utilization = % used + const remaining = Math.max(0, 100 - used); + return { + used, + total: 100, + remaining, + resetAt: parseResetTime(window.resets_at), + remainingPercentage: remaining, + unlimited: false, + }; + }; + + const fiveHour = toRecord(data.five_hour); + if (hasUtilization(fiveHour)) { + quotas["session (5h)"] = createQuotaObject(fiveHour); + } + + const sevenDay = toRecord(data.seven_day); + if (hasUtilization(sevenDay)) { + quotas["weekly (7d)"] = createQuotaObject(sevenDay); + } + + // Map Anthropic's internal codenames (e.g., omelette → Designer) for display. + const MODEL_DISPLAY_NAMES: Record = { + omelette: "designer", + }; + for (const [key, value] of Object.entries(data)) { + const valueRecord = toRecord(value); + if (key.startsWith("seven_day_") && key !== "seven_day" && hasUtilization(valueRecord)) { + const codename = key.replace("seven_day_", ""); + const modelName = MODEL_DISPLAY_NAMES[codename] || codename; + quotas[`weekly ${modelName} (7d)`] = createQuotaObject(valueRecord); + } + } + + const bootstrap = await bootstrapPromise; + const plan = + getClaudePlanLabel( + typeof data.tier === "string" ? data.tier : null, + typeof data.plan === "string" ? data.plan : null, + typeof data.subscription_type === "string" ? data.subscription_type : null, + bootstrap?.organization_rate_limit_tier + ) ?? undefined; + + return { + ...(plan ? { plan } : {}), + quotas, + extraUsage: data.extra_usage ?? null, + bootstrap, + }; + } + + // Fallback: OAuth endpoint returned non-OK, try legacy settings/org endpoint + console.warn( + `[Claude Usage] OAuth endpoint returned ${oauthResponse.status}, falling back to legacy` + ); + const legacy = await getClaudeUsageLegacy(accessToken); + return { ...legacy, bootstrap: await bootstrapPromise }; + } catch (error) { + return { + message: `Claude connected. Unable to fetch usage: ${(error as Error).message}`, + bootstrap: await bootstrapPromise, + }; + } +} + + + +/** + * Legacy Claude usage fetcher for API key / org admin users. + * Uses /v1/settings + /v1/organizations/{org_id}/usage endpoints. + */ +async function getClaudeUsageLegacy(accessToken?: string) { + try { + const settingsResponse = await fetch(CLAUDE_CONFIG.settingsUrl, { + method: "GET", + headers: { + Authorization: `Bearer ${accessToken}`, + "anthropic-version": CLAUDE_CONFIG.apiVersion, + }, + }); + + if (settingsResponse.ok) { + const settings = toRecord(await settingsResponse.json()); + + const organizationId = + typeof settings.organization_id === "string" ? settings.organization_id : ""; + if (organizationId) { + const usageResponse = await fetch( + CLAUDE_CONFIG.usageUrl.replace("{org_id}", organizationId), + { + method: "GET", + headers: { + Authorization: `Bearer ${accessToken}`, + "anthropic-version": CLAUDE_CONFIG.apiVersion, + }, + } + ); + + if (usageResponse.ok) { + const usage = await usageResponse.json(); + return { + plan: settings.plan || "Unknown", + organization: settings.organization_name, + quotas: usage, + }; + } + } + + return { + plan: settings.plan || "Unknown", + organization: settings.organization_name, + message: "Claude connected. Usage details require admin access.", + }; + } + + return { message: "Claude connected. Usage API requires admin permissions." }; + } catch (error) { + return { message: `Claude connected. Unable to fetch usage: ${(error as Error).message}` }; + } +} + diff --git a/open-sse/services/usage/codex.ts b/open-sse/services/usage/codex.ts new file mode 100644 index 00000000000..c320eef4e82 --- /dev/null +++ b/open-sse/services/usage/codex.ts @@ -0,0 +1,140 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { CODEX_CONFIG } from "./constants.ts"; +import { toRecord, getFieldValue, toNumber, parseResetTime } from "./utils.ts"; +import { UsageQuota } from "./types.ts"; + + + +/** + * Codex (OpenAI) Usage - Fetch from ChatGPT backend API + * IMPORTANT: Uses persisted workspaceId from OAuth to ensure correct workspace binding. + * No fallback to other workspaces - strict binding to user's selected workspace. + */ +export async function getCodexUsage( + accessToken?: string, + providerSpecificData: Record = {} +) { + try { + // Use persisted workspace ID from OAuth - NO FALLBACK + const accountId = + typeof providerSpecificData.workspaceId === "string" + ? providerSpecificData.workspaceId + : null; + + const headers: Record = { + Authorization: `Bearer ${accessToken}`, + "Content-Type": "application/json", + Accept: "application/json", + }; + if (accountId) { + headers["chatgpt-account-id"] = accountId; + } + + const response = await fetch(CODEX_CONFIG.usageUrl, { + method: "GET", + headers, + }); + + if (!response.ok) { + if (response.status === 401 || response.status === 403) { + return { + message: `Codex token expired or access denied. Please re-authenticate the connection.`, + }; + } + throw new Error(`Codex API error: ${response.status}`); + } + + const data = await response.json(); + + // Parse rate limit info (supports both snake_case and camelCase) + const rateLimit = toRecord(getFieldValue(data, "rate_limit", "rateLimit")); + const primaryWindow = toRecord(getFieldValue(rateLimit, "primary_window", "primaryWindow")); + const secondaryWindow = toRecord( + getFieldValue(rateLimit, "secondary_window", "secondaryWindow") + ); + + // Parse reset times (reset_at is Unix timestamp in seconds) + const parseWindowReset = (window: unknown) => { + const resetAt = toNumber(getFieldValue(window, "reset_at", "resetAt"), 0); + const resetAfterSeconds = toNumber( + getFieldValue(window, "reset_after_seconds", "resetAfterSeconds"), + 0 + ); + if (resetAt > 0) return parseResetTime(resetAt * 1000); + if (resetAfterSeconds > 0) return parseResetTime(Date.now() + resetAfterSeconds * 1000); + return null; + }; + + // Build quota windows + const quotas: Record = {}; + + // Primary window (5-hour) + if (Object.keys(primaryWindow).length > 0) { + const usedPercent = toNumber(getFieldValue(primaryWindow, "used_percent", "usedPercent"), 0); + quotas.session = { + used: usedPercent, + total: 100, + remaining: 100 - usedPercent, + resetAt: parseWindowReset(primaryWindow), + unlimited: false, + }; + } + + // Secondary window (weekly) + if (Object.keys(secondaryWindow).length > 0) { + const usedPercent = toNumber( + getFieldValue(secondaryWindow, "used_percent", "usedPercent"), + 0 + ); + quotas.weekly = { + used: usedPercent, + total: 100, + remaining: 100 - usedPercent, + resetAt: parseWindowReset(secondaryWindow), + unlimited: false, + }; + } + + // Code review rate limit (3rd window — differs per plan: Plus/Pro/Team) + const codeReviewRateLimit = toRecord( + getFieldValue(data, "code_review_rate_limit", "codeReviewRateLimit") + ); + const codeReviewWindow = toRecord( + getFieldValue(codeReviewRateLimit, "primary_window", "primaryWindow") + ); + + // Only include code review quota if the API returned data for it + const codeReviewUsedRaw = getFieldValue(codeReviewWindow, "used_percent", "usedPercent"); + const codeReviewRemainingRaw = getFieldValue( + codeReviewWindow, + "remaining_count", + "remainingCount" + ); + if (codeReviewUsedRaw !== null || codeReviewRemainingRaw !== null) { + const codeReviewUsedPercent = toNumber(codeReviewUsedRaw, 0); + quotas.code_review = { + used: codeReviewUsedPercent, + total: 100, + remaining: 100 - codeReviewUsedPercent, + resetAt: parseWindowReset(codeReviewWindow), + unlimited: false, + }; + } + + return { + plan: String(getFieldValue(data, "plan_type", "planType") || "unknown"), + limitReached: Boolean(getFieldValue(rateLimit, "limit_reached", "limitReached")), + quotas, + }; + } catch (error) { + return { message: `Failed to fetch Codex usage: ${(error as Error).message}` }; + } +} + diff --git a/open-sse/services/usage/constants.ts b/open-sse/services/usage/constants.ts new file mode 100644 index 00000000000..f42e6cd73ad --- /dev/null +++ b/open-sse/services/usage/constants.ts @@ -0,0 +1,151 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { + getAntigravityFetchAvailableModelsUrls, + ANTIGRAVITY_BASE_URLS, +} from "../../config/antigravityUpstream.ts"; + +import { + isUserCallableAntigravityModelId, + toClientAntigravityModelId, +} from "../../config/antigravityModelAliases.ts"; + +import { isUserCallableAgyModelId } from "../../config/agyModels.ts"; + +import { getGlmQuotaUrl } from "../../config/glmProvider.ts"; + +import { getGitHubCopilotInternalUserHeaders } from "../../config/providerHeaderProfiles.ts"; + +import { + antigravityUserAgent, + getAntigravityHeaders, + getAntigravityLoadCodeAssistMetadata, +} from "../antigravityHeaders.ts"; + +import { + getAntigravityRemainingCredits, + updateAntigravityRemainingCredits, +} from "../../executors/antigravity.ts"; + + + + +// Quota / usage upstream URLs (overridable for testing or relays). +export const CROF_USAGE_URL = process.env.OMNIROUTE_CROF_USAGE_URL ?? "https://crof.ai/usage_api/"; + + +export const GEMINI_CLI_USAGE_URL = + process.env.OMNIROUTE_GEMINI_CLI_USAGE_URL ?? + "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist"; + + +export const CODEWHISPERER_BASE_URL = + process.env.OMNIROUTE_CODEWHISPERER_BASE_URL ?? "https://codewhisperer.us-east-1.amazonaws.com"; + + + +// Antigravity API config (credentials from PROVIDERS via credential loader) +export const ANTIGRAVITY_CONFIG = { + quotaApiUrls: getAntigravityFetchAvailableModelsUrls(), + loadProjectApiUrl: "https://daily-cloudcode-pa.sandbox.googleapis.com/v1internal:loadCodeAssist", + tokenUrl: "https://oauth2.googleapis.com/token", + get clientId() { + return PROVIDERS.antigravity.clientId; + }, + get clientSecret() { + return PROVIDERS.antigravity.clientSecret; + }, + get userAgent() { + return antigravityUserAgent(); + }, +}; + + + +// Codex (OpenAI) API config +export const CODEX_CONFIG = { + usageUrl: "https://chatgpt.com/backend-api/wham/usage", +}; + + + +// Claude API config +export const CLAUDE_CONFIG = { + oauthUsageUrl: "https://api.anthropic.com/api/oauth/usage", + usageUrl: "https://api.anthropic.com/v1/organizations/{org_id}/usage", + settingsUrl: "https://api.anthropic.com/v1/settings", + apiVersion: "2023-06-01", +}; + + + +// Kimi Coding API config +export const KIMI_CONFIG = { + baseUrl: "https://api.kimi.com/coding/v1", + usageUrl: "https://api.kimi.com/coding/v1/usages", + apiVersion: "2023-06-01", +}; + + + +export const NANOGPT_CONFIG = { + usageUrl: "https://nano-gpt.com/api/subscription/v1/usage", +}; + + + +export const OPENCODE_GO_QUOTA_URL = + process.env.OMNIROUTE_OPENCODE_GO_QUOTA_URL ?? "https://api.z.ai/api/monitor/usage/quota/limit"; + + +export const OPENCODE_GO_QUOTA_TOTALS = { + session: 12, + weekly: 30, + mcp_monthly: 60, +} as const; + + +export const OPENCODE_GO_QUOTA_ORDER = ["session", "weekly", "mcp_monthly"] as const; + + +export type OpenCodeGoQuotaName = (typeof OPENCODE_GO_QUOTA_ORDER)[number]; + + + +// Cursor dashboard usage API config +// The endpoint that powers https://cursor.com/dashboard/spending. Validates the WorkOS +// session via the WorkosCursorSessionToken cookie (format: `${userId}::${jwt}`) and +// rejects requests without a matching Origin/Referer (Invalid origin for state-changing request). +export const CURSOR_USAGE_CONFIG = { + usageUrl: "https://cursor.com/api/dashboard/get-current-period-usage", + origin: "https://cursor.com", + referer: "https://cursor.com/dashboard/spending", + userAgent: + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36", +}; + + + +export const MINIMAX_USAGE_CONFIG = { + minimax: { + usageUrls: [ + "https://www.minimax.io/v1/token_plan/remains", + "https://api.minimax.io/v1/api/openplatform/coding_plan/remains", + ], + }, + "minimax-cn": { + usageUrls: [ + "https://www.minimaxi.com/v1/api/openplatform/coding_plan/remains", + "https://api.minimaxi.com/v1/api/openplatform/coding_plan/remains", + ], + }, +} as const; + + + +export const GLM_QUOTA_ORDER = ["5 Hours Quota", "Weekly Quota", "Monthly Tools", "Tokens", "Time Limit"]; + diff --git a/open-sse/services/usage/core.ts b/open-sse/services/usage/core.ts new file mode 100644 index 00000000000..a22c58c5225 --- /dev/null +++ b/open-sse/services/usage/core.ts @@ -0,0 +1,191 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { + getAntigravityFetchAvailableModelsUrls, + ANTIGRAVITY_BASE_URLS, +} from "../../config/antigravityUpstream.ts"; + +import { + isUserCallableAntigravityModelId, + toClientAntigravityModelId, +} from "../../config/antigravityModelAliases.ts"; + +import { isUserCallableAgyModelId } from "../../config/agyModels.ts"; + +import { getGlmQuotaUrl } from "../../config/glmProvider.ts"; + +import { getGitHubCopilotInternalUserHeaders } from "../../config/providerHeaderProfiles.ts"; + +import { fetchBailianQuota, type BailianTripleWindowQuota } from "../bailianQuotaFetcher.ts"; + +import { fetchDeepseekQuota, type DeepseekQuota } from "../deepseekQuotaFetcher.ts"; + +import { fetchOpencodeQuota, type OpencodeTripleWindowQuota } from "../opencodeQuotaFetcher.ts"; + +import { + applyAntigravityClientProfileHeaders, + getAntigravityBootstrapHeaders, + getAntigravityClientProfile, +} from "../antigravityClientProfile.ts"; + +import { + antigravityUserAgent, + getAntigravityHeaders, + getAntigravityLoadCodeAssistMetadata, +} from "../antigravityHeaders.ts"; + +import { + getAntigravityRemainingCredits, + updateAntigravityRemainingCredits, +} from "../../executors/antigravity.ts"; + +import { getCreditsMode } from "../antigravityCredits.ts"; + +import { CLAUDE_CODE_VERSION, fetchClaudeBootstrap } from "../../executors/claudeIdentity.ts"; + +import { generateAntigravityRequestId, getAntigravitySessionId } from "../antigravityIdentity.ts"; + +import { + extractCodeAssistOnboardTierId, + extractCodeAssistSubscriptionTier, +} from "../codeAssistSubscription.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { UsageProviderConnection } from "./types.ts"; +import { getGitHubUsage } from "./github.ts"; +import { getGeminiUsage } from "./gemini.ts"; +import { getAntigravityUsage } from "./antigravity.ts"; +import { getClaudeUsage } from "./claude.ts"; +import { getCodexUsage } from "./codex.ts"; +import { getCursorUsage } from "./cursor.ts"; +import { getKiroUsage } from "./kiro.ts"; +import { getKimiUsage } from "./kimi.ts"; +import { getQwenUsage } from "./qwen.ts"; +import { getQoderUsage } from "./qoder.ts"; +import { getGlmUsage } from "./glm.ts"; +import { getOpenCodeGoUsage } from "./opencodeGo.ts"; +import { getMiniMaxUsage } from "./minimax.ts"; +import { getCrofUsage } from "./crof.ts"; +import { getBailianCodingPlanUsage } from "./bailian.ts"; +import { getNanoGptUsage } from "./nanogpt.ts"; +import { getDeepseekUsage } from "./deepseek.ts"; +import { getOpencodeUsage } from "./opencode.ts"; +import { getXiaomiMimoUsage } from "./xiaomi.ts"; + + + +/** + * Single source of truth for which providers have a `getUsageForProvider` + * implementation. Consumers like `genericQuotaFetcher.ts` reference this so + * the registration list can't drift from the switch statement below. + * + * If you add a new provider to the switch, add it here too. + */ +export const USAGE_FETCHER_PROVIDERS = [ + "github", + "gemini-cli", + "antigravity", + "agy", + "claude", + "codex", + "cursor", + "kiro", + "amazon-q", + "kimi-coding", + "qwen", + "qoder", + "glm", + "glm-cn", + "zai", + "glmt", + "opencode-go", + "minimax", + "minimax-cn", + "crof", + "bailian-coding-plan", + "nanogpt", + "deepseek", + "opencode", + "opencode-zen", + "xiaomi-mimo", +] as const; + + + +/** + * Get usage data for a provider connection + * @param {Object} connection - Provider connection with accessToken + * @returns {Promise} Usage data with quotas + */ +export async function getUsageForProvider( + connection: UsageProviderConnection, + options: { forceRefresh?: boolean } = {} +) { + const { id, provider, accessToken, apiKey, providerSpecificData, projectId, email } = connection; + + switch (provider) { + case "github": + return await getGitHubUsage(accessToken, providerSpecificData); + case "gemini-cli": + return await getGeminiUsage(accessToken, providerSpecificData, projectId); + case "antigravity": + case "agy": + return await getAntigravityUsage( + provider, + accessToken, + providerSpecificData, + projectId, + id, + options + ); + case "claude": + return await getClaudeUsage(accessToken); + case "codex": + return await getCodexUsage(accessToken, providerSpecificData); + case "cursor": + return await getCursorUsage(accessToken || "", providerSpecificData); + case "kiro": + case "amazon-q": + return await getKiroUsage(accessToken, providerSpecificData); + case "kimi-coding": + return await getKimiUsage(accessToken); + case "qwen": + return await getQwenUsage(accessToken, providerSpecificData); + case "qoder": + return await getQoderUsage(accessToken); + case "glm": + case "glm-cn": + case "zai": + case "glmt": + return await getGlmUsage(apiKey || "", { + ...(providerSpecificData || {}), + ...(provider === "glm-cn" ? { apiRegion: "china" } : {}), + }); + case "opencode-go": + return await getOpenCodeGoUsage(apiKey || ""); + case "minimax": + case "minimax-cn": + return await getMiniMaxUsage(apiKey || "", provider); + case "crof": + return await getCrofUsage(apiKey || ""); + case "bailian-coding-plan": + return await getBailianCodingPlanUsage(id || "", apiKey || "", providerSpecificData); + case "nanogpt": + return await getNanoGptUsage(apiKey || ""); + case "deepseek": + return await getDeepseekUsage(id || "", apiKey || ""); + case "opencode": + case "opencode-zen": + return await getOpencodeUsage(id || "", apiKey || ""); + case "xiaomi-mimo": + return await getXiaomiMimoUsage(id || ""); + default: + return { message: `Usage API not implemented for ${provider}` }; + } +} + diff --git a/open-sse/services/usage/crof.ts b/open-sse/services/usage/crof.ts new file mode 100644 index 00000000000..67fdfe7eb1e --- /dev/null +++ b/open-sse/services/usage/crof.ts @@ -0,0 +1,119 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { CROF_USAGE_URL } from "./constants.ts"; +import { JsonRecord, UsageQuota } from "./types.ts"; +import { toRecord, toNumber } from "./utils.ts"; + + + +// CrofAI surfaces a tiny endpoint with two signals: +// GET https://crof.ai/usage_api/ → { usable_requests: number|null, credits: number } +// `usable_requests` is the daily request bucket on a subscription plan; `null` +// for pay-as-you-go. `credits` is the USD credit balance. We surface both as +// quotas so the Limits & Quotas page can render whichever the account uses. +export async function getCrofUsage(apiKey: string) { + if (!apiKey) { + return { message: "CrofAI API key not available. Add a key to view usage." }; + } + + let response: Response; + try { + response = await fetch(CROF_USAGE_URL, { + method: "GET", + headers: { + Authorization: `Bearer ${apiKey}`, + Accept: "application/json", + }, + }); + } catch (error) { + return { message: `CrofAI connected. Unable to fetch usage: ${(error as Error).message}` }; + } + + const rawText = await response.text(); + + if (response.status === 401 || response.status === 403) { + return { message: "CrofAI connected. The API key was rejected by /usage_api/." }; + } + + if (!response.ok) { + return { message: `CrofAI connected. /usage_api/ returned HTTP ${response.status}.` }; + } + + let payload: JsonRecord = {}; + if (rawText) { + try { + payload = toRecord(JSON.parse(rawText)); + } catch { + return { message: "CrofAI connected. Unable to parse /usage_api/ response." }; + } + } + + const usableRequestsRaw = payload["usable_requests"]; + const usableRequests = + usableRequestsRaw === null || usableRequestsRaw === undefined + ? null + : toNumber(usableRequestsRaw, 0); + const credits = toNumber(payload["credits"], 0); + + const quotas: Record = {}; + + if (usableRequests !== null) { + // CrofAI's /usage_api/ returns only the remaining count; the daily + // allotment is not exposed. CrofAI Pro plan = 1,000 requests/day per + // their pricing page, so use that as the baseline total. If the user + // is on a plan with a higher cap we widen the total to whatever they + // currently report so we never compute a negative `used`. + // Without this, total=0 makes the dashboard's percentage formula read + // 0% (interpreted as "depleted" → red) even on a fresh bucket. + const CROF_DAILY_BASELINE = 1000; + const remaining = Math.max(0, usableRequests); + const total = Math.max(CROF_DAILY_BASELINE, remaining); + const used = Math.max(0, total - remaining); + + // CrofAI also does not return a reset timestamp and the docs only say + // "requests left today". The Crof.ai dashboard shows the daily bucket + // resetting at ~05:00 UTC (verified against the live countdown on + // 2026-04-25), so synthesize the next 05:00 UTC instant to match. + // Swap for a real field if Crof ever exposes one. + const now = new Date(); + const RESET_HOUR_UTC = 5; + const todayResetMs = Date.UTC( + now.getUTCFullYear(), + now.getUTCMonth(), + now.getUTCDate(), + RESET_HOUR_UTC + ); + const nextResetMs = + todayResetMs > now.getTime() ? todayResetMs : todayResetMs + 24 * 60 * 60 * 1000; + const nextResetIso = new Date(nextResetMs).toISOString(); + + quotas["Requests Today"] = { + used, + total, + remaining, + resetAt: nextResetIso, + unlimited: false, + displayName: `Requests Today: ${remaining} left`, + }; + } + + // Credits are an open balance — render as unlimited so the UI shows the + // dollar value rather than a misleading 0/0 bar. + quotas["Credits"] = { + used: 0, + total: 0, + remaining: 0, + resetAt: null, + unlimited: true, + displayName: `Credits: $${credits.toFixed(4)}`, + }; + + return { quotas }; +} + diff --git a/open-sse/services/usage/cursor.ts b/open-sse/services/usage/cursor.ts new file mode 100644 index 00000000000..5c956656ea5 --- /dev/null +++ b/open-sse/services/usage/cursor.ts @@ -0,0 +1,153 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { toRecord, toNumber, clampPercentage, parseResetTime } from "./utils.ts"; +import { CURSOR_USAGE_CONFIG } from "./constants.ts"; +import { UsageQuota } from "./types.ts"; + + + +/** + * Decode the `sub` claim of a Cursor JWT (the WorkOS user id). + * Returns null if the token is not a parseable JWT. + */ +function decodeCursorJwtSub(token: string): string | null { + if (!token || typeof token !== "string") return null; + const parts = token.split("."); + if (parts.length !== 3) return null; + try { + let payload = parts[1].replace(/-/g, "+").replace(/_/g, "/"); + while (payload.length % 4 !== 0) payload += "="; + const decoded = JSON.parse(Buffer.from(payload, "base64").toString("utf8")); + const sub = decoded?.sub; + return typeof sub === "string" && sub.length > 0 ? sub : null; + } catch { + return null; + } +} + + + +/** + * Cursor Pro Plan Usage + * Fetches current-billing-cycle spend from the cursor.com dashboard API and exposes three + * windows that mirror the cursor.com/dashboard/spending UI: Total / Auto + Composer / API. + */ +export async function getCursorUsage(accessToken: string, providerSpecificData?: unknown) { + if (!accessToken) { + return { message: "Cursor access token missing. Re-import the connection from Cursor IDE." }; + } + + const storedUserId = (() => { + const raw = toRecord(providerSpecificData).userId; + return typeof raw === "string" && raw.length > 0 ? raw : null; + })(); + const userId = storedUserId || decodeCursorJwtSub(accessToken); + + if (!userId) { + return { + message: "Cursor token missing user id. Re-import the connection from Cursor IDE.", + }; + } + + try { + const response = await fetch(CURSOR_USAGE_CONFIG.usageUrl, { + method: "POST", + redirect: "manual", + headers: { + Cookie: `WorkosCursorSessionToken=${userId}::${accessToken}`, + Origin: CURSOR_USAGE_CONFIG.origin, + Referer: CURSOR_USAGE_CONFIG.referer, + "Content-Type": "application/json", + Accept: "application/json", + "User-Agent": CURSOR_USAGE_CONFIG.userAgent, + }, + body: "{}", + }); + + // 3xx redirect to WorkOS authkit means the session cookie was rejected. + if (response.status >= 300 && response.status < 400) { + return { + plan: "Cursor", + message: "Cursor session expired. Re-import the token from Cursor IDE.", + }; + } + + if (!response.ok) { + const errorText = (await response.text()).slice(0, 200); + if (response.status === 401 || response.status === 403) { + return { + plan: "Cursor", + message: "Cursor session unauthorized. Re-import the token from Cursor IDE.", + }; + } + return { + plan: "Cursor", + message: `Cursor usage endpoint error (${response.status}): ${errorText}`, + }; + } + + const data = toRecord(await response.json()); + const planUsage = toRecord(data.planUsage); + + if (Object.keys(planUsage).length === 0) { + return { + plan: "Cursor", + message: "Cursor connected. No active plan usage returned.", + }; + } + + const limitCents = Math.max(0, toNumber(planUsage.limit, 0)); + const totalSpendCents = Math.max(0, toNumber(planUsage.totalSpend, 0)); + const autoPercentUsed = clampPercentage(toNumber(planUsage.autoPercentUsed, 0)); + const apiPercentUsed = clampPercentage(toNumber(planUsage.apiPercentUsed, 0)); + const totalPercentUsed = clampPercentage(toNumber(planUsage.totalPercentUsed, 0)); + + // billingCycleEnd is a numeric-string in ms; coerce so parseResetTime sees a number. + const billingCycleEndMs = toNumber(data.billingCycleEnd, 0); + const resetAt = billingCycleEndMs > 0 ? parseResetTime(billingCycleEndMs) : null; + + // Convert cents → dollars rounded to 2 decimal places. + const toDollars = (cents: number) => Math.round(cents) / 100; + + const limitDollars = toDollars(limitCents); + const buildWindow = (percentUsed: number, usedCentsOverride?: number): UsageQuota => { + const usedCents = + typeof usedCentsOverride === "number" + ? usedCentsOverride + : Math.round((limitCents * percentUsed) / 100); + const used = toDollars(Math.min(usedCents, limitCents)); + const remaining = toDollars(Math.max(limitCents - Math.min(usedCents, limitCents), 0)); + return { + used, + total: limitDollars, + remaining, + remainingPercentage: clampPercentage(100 - percentUsed), + resetAt, + unlimited: false, + }; + }; + + const quotas: Record = { + Total: buildWindow(totalPercentUsed, totalSpendCents), + "Auto + Composer": buildWindow(autoPercentUsed), + API: buildWindow(apiPercentUsed), + }; + + return { + plan: "Cursor Pro", + quotas, + }; + } catch (error) { + return { + plan: "Cursor", + message: `Cursor connected. Unable to fetch usage: ${(error as Error).message}`, + }; + } +} + diff --git a/open-sse/services/usage/deepseek.ts b/open-sse/services/usage/deepseek.ts new file mode 100644 index 00000000000..e85c08c5c91 --- /dev/null +++ b/open-sse/services/usage/deepseek.ts @@ -0,0 +1,63 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { fetchDeepseekQuota, type DeepseekQuota } from "../deepseekQuotaFetcher.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { UsageQuota } from "./types.ts"; + + + +/** + * DeepSeek Usage + * Fetches balance from the DeepSeek balance API. + * Returns all balances (USD and CNY) as "credits" for credits-style UI display. + */ +export async function getDeepseekUsage(connectionId: string, apiKey: string) { + try { + const connection = { apiKey }; + const quota = await fetchDeepseekQuota(connectionId, connection); + + if (!quota) { + return { message: "DeepSeek API key not available. Add a key to view usage." }; + } + + const deepseekQuota = quota as DeepseekQuota; + const { balances, isAvailable, limitReached } = deepseekQuota; + + const quotas: Record = {}; + + // Show all balances as credits-style entries (e.g., credits_usd, credits_cny) + // The UI will display them as "🪙 Balance (USD) $50.00" + for (const balanceInfo of balances) { + const key = `credits_${balanceInfo.currency.toLowerCase()}`; + quotas[key] = { + used: 0, + total: 0, + remaining: balanceInfo.balance, + remainingPercentage: 100, + resetAt: null, + unlimited: true, + currency: balanceInfo.currency, + grantedBalance: balanceInfo.grantedBalance, + toppedUpBalance: balanceInfo.toppedUpBalance, + }; + } + + const plan = isAvailable ? "DeepSeek" : "DeepSeek (Insufficient Balance)"; + + return { + plan, + quotas, + isAvailable, + limitReached, + }; + } catch (error) { + return { message: `DeepSeek error: ${(error as Error).message}` }; + } +} + diff --git a/open-sse/services/usage/gemini.ts b/open-sse/services/usage/gemini.ts new file mode 100644 index 00000000000..25ee8ae6ad8 --- /dev/null +++ b/open-sse/services/usage/gemini.ts @@ -0,0 +1,161 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { SubscriptionCacheEntry, JsonRecord, UsageQuota } from "./types.ts"; +import { getAntigravityUsage, mapCodeAssistSubscriptionToPlanLabel } from "./antigravity.ts"; +import { toRecord, toNumber, parseResetTime } from "./utils.ts"; +import { GEMINI_CLI_USAGE_URL } from "./constants.ts"; + + + +// ── Gemini CLI subscription info cache ────────────────────────────────────── +// Prevents duplicate loadCodeAssist calls within the same quota cycle. +// Key: accessToken → { data, fetchedAt } +export const _geminiCliSubCache = new Map(); + + +export const GEMINI_CLI_CACHE_TTL_MS = 5 * 60 * 1000; + + // 5 minutes + +/** + * Gemini CLI Usage — fetch per-model quota from Cloud Code Assist API. + * Gemini CLI and Antigravity share the same upstream (cloudcode-pa.googleapis.com), + * so this follows the same pattern as getAntigravityUsage(). + */ +export async function getGeminiUsage( + accessToken?: string, + providerSpecificData?: JsonRecord, + connectionProjectId?: string +) { + if (!accessToken) { + return { plan: "Free", message: "Gemini CLI access token not available." }; + } + + try { + const subscriptionInfo = await getGeminiCliSubscriptionInfoCached(accessToken); + const projectId = + connectionProjectId || + providerSpecificData?.projectId || + toRecord(subscriptionInfo).cloudaicompanionProject || + null; + + const plan = getGeminiCliPlanLabel(subscriptionInfo); + + if (!projectId) { + return { plan, message: "Gemini CLI project ID not available." }; + } + + // Use retrieveUserQuota (same endpoint as Gemini CLI /stats command). + // Returns per-model buckets with remainingFraction and resetTime. + const response = await fetch( + "https://cloudcode-pa.googleapis.com/v1internal:retrieveUserQuota", + { + method: "POST", + headers: { + Authorization: `Bearer ${accessToken}`, + "Content-Type": "application/json", + }, + body: JSON.stringify({ project: projectId }), + signal: AbortSignal.timeout(10000), + } + ); + + if (!response.ok) { + return { plan, message: `Gemini CLI quota error (${response.status}).` }; + } + + const data = await response.json(); + const quotas: Record = {}; + + const dataRecord = toRecord(data); + if (Array.isArray(dataRecord.buckets)) { + for (const bucketValue of dataRecord.buckets) { + const bucket = toRecord(bucketValue); + if (!bucket.modelId || bucket.remainingFraction == null) continue; + + const remainingFraction = toNumber(bucket.remainingFraction, 0); + const remainingPercentage = remainingFraction * 100; + const QUOTA_NORMALIZED_BASE = 1000; + const total = QUOTA_NORMALIZED_BASE; + const remaining = Math.round(total * remainingFraction); + const used = Math.max(0, total - remaining); + + quotas[String(bucket.modelId)] = { + used, + total, + resetAt: parseResetTime(bucket.resetTime), + remainingPercentage, + unlimited: false, + }; + } + } + + return { plan, quotas }; + } catch (error) { + return { message: `Gemini CLI error: ${(error as Error).message}` }; + } +} + + + +/** + * Get Gemini CLI subscription info (cached, 5 min TTL) + */ +async function getGeminiCliSubscriptionInfoCached(accessToken: string): Promise { + const cacheKey = accessToken; + const cached = _geminiCliSubCache.get(cacheKey); + + if (cached && Date.now() - cached.fetchedAt < GEMINI_CLI_CACHE_TTL_MS) { + return cached.data; + } + + const data = await getGeminiCliSubscriptionInfo(accessToken); + _geminiCliSubCache.set(cacheKey, { data, fetchedAt: Date.now() }); + return data; +} + + + +/** + * Get Gemini CLI subscription info using correct headers. + */ +async function getGeminiCliSubscriptionInfo(accessToken: string): Promise { + try { + const response = await fetch(GEMINI_CLI_USAGE_URL, { + method: "POST", + headers: { + Authorization: `Bearer ${accessToken}`, + "Content-Type": "application/json", + }, + body: JSON.stringify({ + metadata: { + ideType: "IDE_UNSPECIFIED", + platform: "PLATFORM_UNSPECIFIED", + pluginType: "GEMINI", + }, + }), + }); + + if (!response.ok) return null; + + return await response.json(); + } catch { + return null; + } +} + + + +/** + * Map Gemini CLI subscription tier to display label (same tiers as Antigravity). + */ +export function getGeminiCliPlanLabel(subscriptionInfo: unknown): string { + return mapCodeAssistSubscriptionToPlanLabel(subscriptionInfo); +} + diff --git a/open-sse/services/usage/github.ts b/open-sse/services/usage/github.ts new file mode 100644 index 00000000000..39458dde1a8 --- /dev/null +++ b/open-sse/services/usage/github.ts @@ -0,0 +1,272 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { + getAntigravityFetchAvailableModelsUrls, + ANTIGRAVITY_BASE_URLS, +} from "../../config/antigravityUpstream.ts"; + +import { + isUserCallableAntigravityModelId, + toClientAntigravityModelId, +} from "../../config/antigravityModelAliases.ts"; + +import { isUserCallableAgyModelId } from "../../config/agyModels.ts"; + +import { getGlmQuotaUrl } from "../../config/glmProvider.ts"; + +import { getGitHubCopilotInternalUserHeaders } from "../../config/providerHeaderProfiles.ts"; + +import { fetchBailianQuota, type BailianTripleWindowQuota } from "../bailianQuotaFetcher.ts"; + +import { fetchDeepseekQuota, type DeepseekQuota } from "../deepseekQuotaFetcher.ts"; + +import { fetchOpencodeQuota, type OpencodeTripleWindowQuota } from "../opencodeQuotaFetcher.ts"; + +import { + applyAntigravityClientProfileHeaders, + getAntigravityBootstrapHeaders, + getAntigravityClientProfile, +} from "../antigravityClientProfile.ts"; + +import { + antigravityUserAgent, + getAntigravityHeaders, + getAntigravityLoadCodeAssistMetadata, +} from "../antigravityHeaders.ts"; + +import { + getAntigravityRemainingCredits, + updateAntigravityRemainingCredits, +} from "../../executors/antigravity.ts"; + +import { getCreditsMode } from "../antigravityCredits.ts"; + +import { CLAUDE_CODE_VERSION, fetchClaudeBootstrap } from "../../executors/claudeIdentity.ts"; + +import { generateAntigravityRequestId, getAntigravitySessionId } from "../antigravityIdentity.ts"; + +import { + extractCodeAssistOnboardTierId, + extractCodeAssistSubscriptionTier, +} from "../codeAssistSubscription.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { JsonRecord, UsageQuota } from "./types.ts"; +import { toRecord, parseResetTime, getFieldValue, shouldDisplayGitHubQuota, toNumber, clampPercentage, toDisplayLabel } from "./utils.ts"; + + + +/** + * GitHub Copilot Usage + * Uses GitHub accessToken (not copilotToken) to call copilot_internal/user API + */ +export async function getGitHubUsage(accessToken?: string, providerSpecificData?: JsonRecord) { + try { + if (!accessToken) { + throw new Error("No GitHub access token available. Please re-authorize the connection."); + } + + // copilot_internal/user API requires GitHub OAuth token, not copilotToken + const response = await fetch("https://api.github.com/copilot_internal/user", { + headers: getGitHubCopilotInternalUserHeaders(`token ${accessToken}`), + }); + + if (!response.ok) { + const error = await response.text(); + if (response.status === 401 || response.status === 403) { + return { + message: `GitHub token expired or permission denied. Please re-authenticate the connection.`, + }; + } + throw new Error(`GitHub API error: ${error}`); + } + + const data = await response.json(); + const dataRecord = toRecord(data); + + // Handle different response formats (paid vs free) + if (dataRecord.quota_snapshots) { + // Paid plan format + const snapshots = toRecord(dataRecord.quota_snapshots); + const resetAt = parseResetTime( + getFieldValue(dataRecord, "quota_reset_date", "quotaResetDate") + ); + const premiumQuota = formatGitHubQuotaSnapshot(snapshots.premium_interactions, resetAt); + const chatQuota = formatGitHubQuotaSnapshot(snapshots.chat, resetAt); + const completionsQuota = formatGitHubQuotaSnapshot(snapshots.completions, resetAt); + const quotas: Record = {}; + + if (shouldDisplayGitHubQuota(premiumQuota)) { + quotas.premium_interactions = premiumQuota; + } + if (shouldDisplayGitHubQuota(chatQuota)) { + quotas.chat = chatQuota; + } + if (shouldDisplayGitHubQuota(completionsQuota)) { + quotas.completions = completionsQuota; + } + + return { + plan: inferGitHubPlanName(dataRecord, premiumQuota), + resetDate: getFieldValue(dataRecord, "quota_reset_date", "quotaResetDate"), + quotas, + }; + } else if (dataRecord.monthly_quotas || dataRecord.limited_user_quotas) { + // Free/limited plan format. NOTE (#2876): the upstream field + // `limited_user_quotas[name]` is the *remaining* count for the month + // (it counts down toward 0 and resets on `limited_user_reset_date`), + // NOT the used count. The pre-3.8.6 implementation inverted this and + // showed "0% when not used / 100% when fully used" on the dashboard. + // Confirmed against three independent upstream parsers: + // - robinebers/openusage docs/providers/copilot.md (Free Tier table) + // - raycast/extensions agent-usage/src/copilot/fetcher.ts (inline comment) + // - looplj/axonhub frontend/src/components/quota-badges.tsx + const monthlyQuotas = toRecord(dataRecord.monthly_quotas); + const remainingQuotas = toRecord(dataRecord.limited_user_quotas); + const resetDate = getFieldValue( + dataRecord, + "limited_user_reset_date", + "limitedUserResetDate" + ); + const resetAt = parseResetTime(resetDate); + const quotas: Record = {}; + + const addLimitedQuota = (name: string) => { + const total = toNumber(getFieldValue(monthlyQuotas, name, name), 0); + if (total <= 0) return null; + const remainingRaw = Math.max(0, toNumber(getFieldValue(remainingQuotas, name, name), 0)); + const remaining = Math.min(remainingRaw, total); + const used = Math.max(total - remaining, 0); + quotas[name] = { + used, + total, + remaining, + remainingPercentage: clampPercentage((remaining / total) * 100), + unlimited: false, + resetAt, + }; + return quotas[name]; + }; + + const premiumQuota = addLimitedQuota("premium_interactions"); + addLimitedQuota("chat"); + addLimitedQuota("completions"); + + return { + plan: inferGitHubPlanName(dataRecord, premiumQuota), + resetDate, + quotas, + }; + } + + return { message: "GitHub Copilot connected. Unable to parse quota data." }; + } catch (error) { + throw new Error(`Failed to fetch GitHub usage: ${error.message}`); + } +} + + + +export function formatGitHubQuotaSnapshot( + quota: unknown, + resetAt: string | null = null +): UsageQuota | null { + const source = toRecord(quota); + if (Object.keys(source).length === 0) return null; + + const unlimited = source.unlimited === true; + const entitlement = toNumber(source.entitlement, Number.NaN); + const totalValue = toNumber(source.total, Number.NaN); + const remainingValue = toNumber(source.remaining, Number.NaN); + const usedValue = toNumber(source.used, Number.NaN); + const percentRemainingValue = toNumber( + getFieldValue(source, "percent_remaining", "percentRemaining"), + Number.NaN + ); + + let total = Number.isFinite(totalValue) + ? Math.max(0, totalValue) + : Number.isFinite(entitlement) + ? Math.max(0, entitlement) + : 0; + let remaining = Number.isFinite(remainingValue) ? Math.max(0, remainingValue) : undefined; + let used = Number.isFinite(usedValue) ? Math.max(0, usedValue) : undefined; + let remainingPercentage = Number.isFinite(percentRemainingValue) + ? clampPercentage(percentRemainingValue) + : undefined; + + if (used === undefined && total > 0 && remaining !== undefined) { + used = Math.max(total - remaining, 0); + } + + if (remaining === undefined && total > 0 && used !== undefined) { + remaining = Math.max(total - used, 0); + } + + if (remainingPercentage === undefined && total > 0 && remaining !== undefined) { + remainingPercentage = clampPercentage((remaining / total) * 100); + } + + if (total <= 0 && remainingPercentage !== undefined) { + total = 100; + used = 100 - remainingPercentage; + remaining = remainingPercentage; + } + + return { + used: Math.max(0, used ?? 0), + total, + remaining, + remainingPercentage, + resetAt, + unlimited, + }; +} + + + +export function inferGitHubPlanName(data: JsonRecord, premiumQuota: UsageQuota | null): string { + const rawPlan = getFieldValue(data, "copilot_plan", "copilotPlan"); + const rawSku = getFieldValue(data, "access_type_sku", "accessTypeSku"); + const planText = typeof rawPlan === "string" ? rawPlan.trim() : ""; + const skuText = typeof rawSku === "string" ? rawSku.trim() : ""; + const combined = `${skuText} ${planText}`.trim().toUpperCase(); + const monthlyQuotas = toRecord(getFieldValue(data, "monthly_quotas", "monthlyQuotas")); + const premiumTotal = + premiumQuota?.total || + toNumber(getFieldValue(monthlyQuotas, "premium_interactions", "premiumInteractions"), 0); + const chatTotal = toNumber(getFieldValue(monthlyQuotas, "chat", "chat"), 0); + + if (combined.includes("PRO+") || combined.includes("PRO_PLUS") || combined.includes("PROPLUS")) { + return "Copilot Pro+"; + } + if (combined.includes("ENTERPRISE")) return "Copilot Enterprise"; + if (combined.includes("BUSINESS")) return "Copilot Business"; + if (combined.includes("STUDENT")) return "Copilot Student"; + if (combined.includes("FREE")) return "Copilot Free"; + if (combined.includes("PRO")) return "Copilot Pro"; + + if (premiumTotal >= 1400) return "Copilot Pro+"; + if (premiumTotal >= 900) return "Copilot Enterprise"; + if (premiumTotal >= 250) { + if (combined.includes("INDIVIDUAL")) return "Copilot Pro"; + return "Copilot Business"; + } + if (premiumTotal > 0 || chatTotal === 50) return "Copilot Free"; + + if (skuText) { + const label = toDisplayLabel(skuText); + return label ? `Copilot ${label}` : "GitHub Copilot"; + } + if (planText) { + const label = toDisplayLabel(planText); + return label ? `Copilot ${label}` : "GitHub Copilot"; + } + return "GitHub Copilot"; +} + diff --git a/open-sse/services/usage/glm.ts b/open-sse/services/usage/glm.ts new file mode 100644 index 00000000000..77342768f74 --- /dev/null +++ b/open-sse/services/usage/glm.ts @@ -0,0 +1,190 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { getGlmQuotaUrl } from "../../config/glmProvider.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { JsonRecord, UsageQuota } from "./types.ts"; +import { toNumber, toRecord, toPercentage, toTitleCase } from "./utils.ts"; +import { GLM_QUOTA_ORDER } from "./constants.ts"; + + + +function getGlmTokenQuotaName( + limit: JsonRecord, + existingQuotas: Record +): string { + const unit = toNumber(limit.unit, 0); + const number = toNumber(limit.number, 0); + + if (unit === 3 && number === 5) return "session"; + if ((unit === 4 && number === 7) || (unit === 3 && number >= 24 * 7)) return "weekly"; + + return existingQuotas.session ? "weekly" : "session"; +} + + + +function getGlmQuotaDisplayName(quotaName: string): string { + if (quotaName === "session") return "5 Hours Quota"; + if (quotaName === "weekly") return "Weekly Quota"; + return quotaName; +} + + + +function getGlmQuotaLabel(type: unknown, unit: unknown): string | null { + const normalized = typeof type === "string" ? type.trim().toUpperCase() : ""; + const unitValue = toNumber(unit, -1); + + switch (normalized) { + case "TOKENS_LIMIT": + case "TOKEN_LIMIT": + if (unitValue === 3) return "5 Hours Quota"; + if (unitValue === 6) return "Weekly Quota"; + return "Tokens"; + case "TIME_LIMIT": + case "TIME_USAGE_LIMIT": + if (unitValue === 5) return "Monthly Tools"; + return "Time Limit"; + default: + return null; + } +} + + + +function orderGlmQuotas(quotas: Record): Record { + const ordered: Record = {}; + + for (const key of GLM_QUOTA_ORDER) { + if (quotas[key]) ordered[key] = quotas[key]; + } + + for (const [key, quota] of Object.entries(quotas)) { + if (!ordered[key]) ordered[key] = quota; + } + + return ordered; +} + + + +/** + * Remaining-percentage for a GLM/z.ai TIME_LIMIT ("Monthly") quota. With an absolute + * monthly cap (`total > 0`) it is `remaining / total`. Coding plans that have no + * monthly cap (only 5-hour windows) report `total = 0`; in that case fall back to the + * percentage-derived remaining so "no monthly cap" renders as full/100% instead of a + * misleading 0% (#3580). + */ +export function glmMonthlyRemainingPercentage(total: number, remaining: number): number { + if (total > 0) { + return Math.max(0, Math.min(100, Math.round((remaining / total) * 100))); + } + return Math.max(0, Math.min(100, Math.round(remaining))); +} + + + +export async function getGlmUsage(apiKey: string, providerSpecificData?: Record) { + if (!apiKey) { + return { message: "API key not available. Add a coding plan API key to view usage." }; + } + + const quotaUrl = getGlmQuotaUrl(providerSpecificData); + + const res = await fetch(quotaUrl, { + headers: { + Authorization: `Bearer ${apiKey}`, + Accept: "application/json", + }, + }); + + if (!res.ok) { + if (res.status === 401) throw new Error("Invalid API key"); + throw new Error(`GLM quota API error (${res.status})`); + } + + const json = await res.json(); + if (toNumber(json.code, 200) === 401 || json.success === false) { + throw new Error("Invalid API key"); + } + + const data = toRecord(json.data); + const limits: unknown[] = Array.isArray(data.limits) ? data.limits : []; + const quotas: Record = {}; + + for (const limit of limits) { + const src = toRecord(limit); + const type = String(src.type || "").toUpperCase(); + const resetMs = toNumber(src.nextResetTime, 0); + const resetAt = resetMs > 0 ? new Date(resetMs).toISOString() : null; + + if (type === "TOKENS_LIMIT") { + const quotaName = getGlmTokenQuotaName(src, quotas); + const usedPercent = toPercentage(src.percentage); + const remaining = Math.max(0, 100 - usedPercent); + + quotas[quotaName] = { + used: usedPercent, + total: 100, + remaining, + remainingPercentage: remaining, + resetAt, + displayName: getGlmQuotaDisplayName(quotaName), + details: Array.isArray(src.models) + ? (src.models as unknown[]).map((m) => { + const modelInfo = toRecord(m); + return { + name: String(modelInfo.model || ""), + used: toNumber(modelInfo.percentage, 0), + }; + }) + : [], + unlimited: false, + }; + continue; + } + + if (type === "TIME_LIMIT") { + const total = toNumber(src.usage, toNumber(src.total, 0)); + const remaining = toNumber(src.remaining, Math.max(0, 100 - toPercentage(src.percentage))); + const used = toNumber(src.currentValue, Math.max(0, total - remaining)); + const remainingPercentage = glmMonthlyRemainingPercentage(total, remaining); + + quotas["mcp_monthly"] = { + used, + total, + remaining, + remainingPercentage, + resetAt, + unlimited: false, + displayName: "Monthly", + details: Array.isArray(src.usageDetails) + ? src.usageDetails.map((item) => { + const detail = toRecord(item); + return { + name: String(detail.modelCode || detail.name || "usage"), + used: toNumber(detail.usage, 0), + }; + }) + : undefined, + }; + } + } + + const levelRaw = + typeof data.planName === "string" + ? data.planName + : typeof data.level === "string" + ? data.level + : ""; + const plan = levelRaw ? toTitleCase(levelRaw.replace(/\s*plan$/i, "")) : null; + + return { plan, quotas: orderGlmQuotas(quotas) }; +} + diff --git a/open-sse/services/usage/index.ts b/open-sse/services/usage/index.ts new file mode 100644 index 00000000000..9796cf88594 --- /dev/null +++ b/open-sse/services/usage/index.ts @@ -0,0 +1,24 @@ +// Re-export everything to maintain backward compatibility +export * from "./constants.ts"; +export * from "./types.ts"; +export * from "./utils.ts"; +export * from "./minimax.ts"; +export * from "./crof.ts"; +export * from "./glm.ts"; +export * from "./opencodeGo.ts"; +export * from "./bailian.ts"; +export * from "./deepseek.ts"; +export * from "./xiaomi.ts"; +export * from "./opencode.ts"; +export * from "./nanogpt.ts"; +export * from "./cursor.ts"; +export * from "./github.ts"; +export * from "./gemini.ts"; +export * from "./antigravity.ts"; +export * from "./claude.ts"; +export * from "./codex.ts"; +export * from "./kiro.ts"; +export * from "./kimi.ts"; +export * from "./qwen.ts"; +export * from "./qoder.ts"; +export * from "./core.ts"; diff --git a/open-sse/services/usage/kimi.ts b/open-sse/services/usage/kimi.ts new file mode 100644 index 00000000000..1a7d2abb11b --- /dev/null +++ b/open-sse/services/usage/kimi.ts @@ -0,0 +1,193 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { safePercentage } from "@/shared/utils/formatting"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { KIMI_CONFIG } from "./constants.ts"; +import { UsageQuota, JsonRecord } from "./types.ts"; +import { toRecord, toNumber, parseResetTime } from "./utils.ts"; + + + +/** + * Map Kimi membership level to display name + * LEVEL_BASIC = Moderato, LEVEL_INTERMEDIATE = Allegretto, + * LEVEL_ADVANCED = Allegro, LEVEL_STANDARD = Vivace + */ +function getKimiPlanName(level: unknown): string { + if (!level) return ""; + const normalizedLevel = String(level); + + const levelMap = { + LEVEL_BASIC: "Moderato", + LEVEL_INTERMEDIATE: "Allegretto", + LEVEL_ADVANCED: "Allegro", + LEVEL_STANDARD: "Vivace", + }; + + return ( + levelMap[normalizedLevel as keyof typeof levelMap] || + normalizedLevel.replace("LEVEL_", "").toLowerCase() + ); +} + + + +/** + * Kimi Coding Usage - Fetch quota from Kimi API + * Uses the official /v1/usages endpoint with custom X-Msh-* headers + */ +export async function getKimiUsage(accessToken?: string) { + // Generate device info for headers (same as OAuth flow) + const deviceId = "kimi-usage-" + Date.now(); + const platform = "omniroute"; + const version = "2.1.2"; + const deviceModel = + typeof process !== "undefined" ? `${process.platform} ${process.arch}` : "unknown"; + + try { + const response = await fetch(KIMI_CONFIG.usageUrl, { + method: "GET", + headers: { + Authorization: `Bearer ${accessToken}`, + "Content-Type": "application/json", + "X-Msh-Platform": platform, + "X-Msh-Version": version, + "X-Msh-Device-Model": deviceModel, + "X-Msh-Device-Id": deviceId, + }, + }); + + const responseText = await response.text(); + + if (!response.ok) { + return { + plan: "Kimi Coding", + message: `Kimi Coding connected. API Error ${response.status}: ${responseText.slice(0, 100)}`, + }; + } + + let data; + try { + data = JSON.parse(responseText); + } catch { + return { + plan: "Kimi Coding", + message: "Kimi Coding connected. Invalid JSON response from API.", + }; + } + + const quotas: Record = {}; + const dataObj = toRecord(data); + + // Parse Kimi usage response format + // Format: { user: {...}, usage: { limit: "100", used: "92", remaining: "8", resetTime: "..." }, limits: [...] } + const usageObj = toRecord(dataObj.usage); + + // Check for Kimi's actual usage fields (strings, not numbers) + const usageLimit = toNumber(usageObj.limit || usageObj.Limit, 0); + const usageUsed = toNumber(usageObj.used || usageObj.Used, 0); + const usageRemaining = toNumber(usageObj.remaining || usageObj.Remaining, 0); + const usageResetTime = + usageObj.resetTime || usageObj.ResetTime || usageObj.reset_at || usageObj.resetAt; + + if (usageLimit > 0) { + const percentRemaining = usageLimit > 0 ? (usageRemaining / usageLimit) * 100 : 0; + + quotas["Weekly"] = { + used: usageUsed, + total: usageLimit, + remaining: usageRemaining, + remainingPercentage: percentRemaining, + resetAt: parseResetTime(usageResetTime), + unlimited: false, + }; + } + + // Also parse limits array for rate limits + const limitsArray = Array.isArray(dataObj.limits) ? dataObj.limits : []; + for (let i = 0; i < limitsArray.length; i++) { + const limitItem = toRecord(limitsArray[i]); + const window = toRecord(limitItem.window); + const detail = toRecord(limitItem.detail); + + const limit = toNumber(detail.limit || detail.Limit, 0); + const remaining = toNumber(detail.remaining || detail.Remaining, 0); + const resetTime = detail.resetTime || detail.reset_at || detail.resetAt; + + if (limit > 0) { + quotas["Ratelimit"] = { + used: limit - remaining, + total: limit, + remaining, + remainingPercentage: limit > 0 ? (remaining / limit) * 100 : 0, + resetAt: parseResetTime(resetTime), + unlimited: false, + }; + } + } + + // Check for quota windows (Claude-like format with utilization) as fallback + const hasUtilization = (window: JsonRecord) => + window && typeof window === "object" && safePercentage(window.utilization) !== undefined; + + const createQuotaObject = (window: JsonRecord) => { + const remaining = safePercentage(window.utilization) as number; + const used = 100 - remaining; + return { + used, + total: 100, + remaining, + resetAt: parseResetTime(window.resets_at), + remainingPercentage: remaining, + unlimited: false, + }; + }; + + if (hasUtilization(toRecord(dataObj.five_hour))) { + quotas["session (5h)"] = createQuotaObject(toRecord(dataObj.five_hour)); + } + + if (hasUtilization(toRecord(dataObj.seven_day))) { + quotas["weekly (7d)"] = createQuotaObject(toRecord(dataObj.seven_day)); + } + + // Check for model-specific quotas + for (const [key, value] of Object.entries(dataObj)) { + const valueRecord = toRecord(value); + if (key.startsWith("seven_day_") && key !== "seven_day" && hasUtilization(valueRecord)) { + const modelName = key.replace("seven_day_", ""); + quotas[`weekly ${modelName} (7d)`] = createQuotaObject(valueRecord); + } + } + + if (Object.keys(quotas).length > 0) { + const userRecord = toRecord(dataObj.user); + const membershipLevel = toRecord(userRecord.membership).level; + const planName = getKimiPlanName(membershipLevel); + return { + plan: planName || "Kimi Coding", + quotas, + }; + } + + // No quota data in response + const userRecord = toRecord(dataObj.user); + const membershipLevel = toRecord(userRecord.membership).level; + const planName = getKimiPlanName(membershipLevel); + return { + plan: planName || "Kimi Coding", + message: "Kimi Coding connected. Usage tracked per request.", + }; + } catch (error) { + return { + message: `Kimi Coding connected. Unable to fetch usage: ${(error as Error).message}`, + }; + } +} + diff --git a/open-sse/services/usage/kiro.ts b/open-sse/services/usage/kiro.ts new file mode 100644 index 00000000000..6f85d37ae97 --- /dev/null +++ b/open-sse/services/usage/kiro.ts @@ -0,0 +1,105 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { JsonRecord, UsageQuota } from "./types.ts"; +import { parseResetTime, toRecord, toNumber } from "./utils.ts"; +import { CODEWHISPERER_BASE_URL } from "./constants.ts"; + + + +/** + * Build the Kiro usage result from a GetUsageLimits response. When the account returns no + * usage breakdown (some AWS IAM / Builder ID accounts don't expose per-resource quota via + * GetUsageLimits), return an informative message instead of empty `quotas:{}` — otherwise the + * dashboard renders a blank quota card with no explanation (#3506). Exported for testing. + */ +export function buildKiroUsageResult( + data: JsonRecord +): { plan: string; quotas: Record } | { message: string } { + const usageList = Array.isArray(data.usageBreakdownList) ? data.usageBreakdownList : []; + const quotaInfo: Record = {}; + const resetAt = parseResetTime(data.nextDateReset || data.resetDate); + + usageList.forEach((breakdownValue: unknown) => { + const breakdown = toRecord(breakdownValue); + const resourceType = + typeof breakdown.resourceType === "string" ? breakdown.resourceType.toLowerCase() : "unknown"; + const used = toNumber(breakdown.currentUsageWithPrecision, 0); + const total = toNumber(breakdown.usageLimitWithPrecision, 0); + + quotaInfo[resourceType] = { used, total, remaining: total - used, resetAt, unlimited: false }; + + const freeTrialInfo = toRecord(breakdown.freeTrialInfo); + if (Object.keys(freeTrialInfo).length > 0) { + const freeUsed = toNumber(freeTrialInfo.currentUsageWithPrecision, 0); + const freeTotal = toNumber(freeTrialInfo.usageLimitWithPrecision, 0); + quotaInfo[`${resourceType}_freetrial`] = { + used: freeUsed, + total: freeTotal, + remaining: freeTotal - freeUsed, + resetAt, + unlimited: false, + }; + } + }); + + if (Object.keys(quotaInfo).length === 0) { + return { + message: + "Kiro connected, but the account returned no usage breakdown. Some AWS IAM / Builder ID accounts don't expose per-resource quota via GetUsageLimits.", + }; + } + + return { + plan: String(toRecord(data.subscriptionInfo).subscriptionTitle || "").trim() || "Kiro", + quotas: quotaInfo, + }; +} + + + +/** + * Kiro (AWS CodeWhisperer) Usage + */ +export async function getKiroUsage(accessToken?: string, providerSpecificData?: JsonRecord) { + try { + const profileArn = providerSpecificData?.profileArn; + if (!profileArn) { + return { message: "Kiro connected. Profile ARN not available for quota tracking." }; + } + + // Kiro uses AWS CodeWhisperer GetUsageLimits API + const payload = { + origin: "AI_EDITOR", + profileArn: profileArn, + resourceType: "AGENTIC_REQUEST", + }; + + const response = await fetch(CODEWHISPERER_BASE_URL, { + method: "POST", + headers: { + Authorization: `Bearer ${accessToken}`, + "Content-Type": "application/x-amz-json-1.0", + "x-amz-target": "AmazonCodeWhispererService.GetUsageLimits", + Accept: "application/json", + }, + body: JSON.stringify(payload), + }); + + if (!response.ok) { + const errorText = await response.text(); + throw new Error(`Kiro API error (${response.status}): ${errorText}`); + } + + const data = toRecord(await response.json()); + return buildKiroUsageResult(data); + } catch (error) { + throw new Error(`Failed to fetch Kiro usage: ${error.message}`); + } +} + diff --git a/open-sse/services/usage/minimax.ts b/open-sse/services/usage/minimax.ts new file mode 100644 index 00000000000..55d63e605c9 --- /dev/null +++ b/open-sse/services/usage/minimax.ts @@ -0,0 +1,356 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { JsonRecord, UsageQuota } from "./types.ts"; +import { pickFirstNonEmptyString, getFieldValue, toNumber, parseResetTime, createQuotaFromUsage, toRecord } from "./utils.ts"; +import { MINIMAX_USAGE_CONFIG } from "./constants.ts"; + + + +export function inferMiniMaxPlanLabelFromTotals(models: JsonRecord[]): string | null { + const maxSessionTotal = models.reduce( + (maxTotal, model) => Math.max(maxTotal, getMiniMaxSessionTotal(model)), + 0 + ); + + if (maxSessionTotal >= 15_000) return "Max"; + if (maxSessionTotal >= 4_500) return "Plus"; + if (maxSessionTotal >= 1_500) return "Starter"; + return null; +} + + + +export function getMiniMaxPlanLabel(payload: JsonRecord, models: JsonRecord[] = []): string { + const raw = pickFirstNonEmptyString( + getFieldValue(payload, "current_subscribe_title", "currentSubscribeTitle"), + getFieldValue(payload, "plan_name", "planName"), + getFieldValue(payload, "plan", "plan"), + getFieldValue(payload, "current_plan_title", "currentPlanTitle"), + getFieldValue(payload, "combo_title", "comboTitle") + ); + + if (!raw) return inferMiniMaxPlanLabelFromTotals(models) || "Coding Plan"; + + const cleaned = raw + .replace(/^minimax\s+/i, "") + .replace(/\bcoding\s+plan\b/gi, "") + .replace(/\s{2,}/g, " ") + .trim(); + + return cleaned || inferMiniMaxPlanLabelFromTotals(models) || "Coding Plan"; +} + + + +export function getMiniMaxQuotaResetAt( + model: JsonRecord, + capturedAtMs: number, + remainsTimeSnakeKey: string, + remainsTimeCamelKey: string, + endTimeSnakeKey: string, + endTimeCamelKey: string +): string | null { + const remainsMs = toNumber(getFieldValue(model, remainsTimeSnakeKey, remainsTimeCamelKey), 0); + if (remainsMs > 0) { + return new Date(capturedAtMs + remainsMs).toISOString(); + } + + return parseResetTime(getFieldValue(model, endTimeSnakeKey, endTimeCamelKey)); +} + + + +export function isMiniMaxTextQuotaModel(modelName: string): boolean { + const normalized = modelName.trim().toLowerCase(); + return ( + normalized.startsWith("minimax-m") || + normalized.startsWith("coding-plan") || + // MiniMax Coding Plan surfaces the text/coding quota under model "general" + // (media buckets like "video"/"image"/"music" are excluded). + normalized === "general" + ); +} + + + +export function getMiniMaxSessionTotal(model: JsonRecord): number { + return Math.max( + 0, + toNumber(getFieldValue(model, "current_interval_total_count", "currentIntervalTotalCount"), 0) + ); +} + + + +export function getMiniMaxWeeklyTotal(model: JsonRecord): number { + return Math.max( + 0, + toNumber(getFieldValue(model, "current_weekly_total_count", "currentWeeklyTotalCount"), 0) + ); +} + + + +function pickMiniMaxRepresentativeModel( + models: JsonRecord[], + getTotal: (model: JsonRecord) => number +): JsonRecord | null { + const withQuota = models.filter((model) => getTotal(model) > 0); + const pool = withQuota.length > 0 ? withQuota : models; + if (pool.length === 0) return null; + + return pool.reduce((best, current) => (getTotal(current) > getTotal(best) ? current : best)); +} + + + +export function createMiniMaxQuotaFromCount( + total: number, + count: number, + resetAt: string | null, + countMeansRemaining: boolean +): UsageQuota { + const used = countMeansRemaining ? Math.max(total - count, 0) : count; + return createQuotaFromUsage(used, total, resetAt); +} + + + +/** + * MiniMax Coding Plan exposes per-window remaining as a 0–100 percent + * (`current_interval_remaining_percent` / `current_weekly_remaining_percent`) + * with zero request counts. Read it defensively (string-encoded numbers ok). + */ +export function getMiniMaxRemainingPercent( + model: JsonRecord, + snakeKey: string, + camelKey: string +): number | null { + const raw = getFieldValue(model, snakeKey, camelKey); + if (raw === null || raw === undefined || raw === "") return null; + const parsed = toNumber(raw, NaN); + return Number.isFinite(parsed) ? Math.max(0, Math.min(100, parsed)) : null; +} + + + +/** Build a 0–100 percent-based window quota (used = 100 − remaining). */ +export function createMiniMaxQuotaFromPercent( + remainingPercent: number, + resetAt: string | null +): UsageQuota { + const clamped = Math.max(0, Math.min(100, remainingPercent)); + return createQuotaFromUsage(100 - clamped, 100, resetAt); +} + + + +/** + * Build one MiniMax usage window (session or weekly) from the representative + * model. Token Plan keys report request counts (`*_total_count`); Coding Plan + * keys report zero counts and a `*_remaining_percent` instead — fall back to + * that so the Coding Plan still surfaces a quota. The percent signal is keyed + * off "counts == 0 + percent present", NOT the endpoint URL, because the + * `token_plan/remains` and `coding_plan/remains` endpoints return identical + * Coding-Plan payloads for a Coding Plan key. + */ +function buildMiniMaxWindow( + models: JsonRecord[], + getTotal: (model: JsonRecord) => number, + usageCountKeys: [string, string], + percentKeys: [string, string], + resetKeys: [string, string, string, string], + capturedAtMs: number, + countMeansRemaining: boolean +): UsageQuota | null { + const model = pickMiniMaxRepresentativeModel(models, getTotal); + if (!model) return null; + + const resetAt = getMiniMaxQuotaResetAt(model, capturedAtMs, ...resetKeys); + const total = getTotal(model); + + if (total > 0) { + const count = Math.max(0, toNumber(getFieldValue(model, ...usageCountKeys), 0)); + return createMiniMaxQuotaFromCount(total, count, resetAt, countMeansRemaining); + } + + const remainingPercent = getMiniMaxRemainingPercent(model, ...percentKeys); + return remainingPercent !== null + ? createMiniMaxQuotaFromPercent(remainingPercent, resetAt) + : null; +} + + + +export function getMiniMaxAuthErrorMessage(message: string): string { + const normalized = message.toLowerCase(); + if ( + normalized.includes("token plan") || + normalized.includes("coding plan") || + normalized.includes("active period") || + normalized.includes("invalid api key") || + normalized.includes("invalid key") || + normalized.includes("subscription") + ) { + return "MiniMax Token Plan API key invalid or inactive. Use an active Token Plan key."; + } + + return "MiniMax access denied. Confirm the key is an active Token Plan API key."; +} + + + +export function getMiniMaxErrorSummary(status: number, message: string): string { + const compact = message.replace(/\s+/g, " ").trim(); + if (!compact) { + return `MiniMax usage endpoint error (${status}).`; + } + if (compact.length <= 160) { + return `MiniMax usage endpoint error (${status}): ${compact}`; + } + return `MiniMax usage endpoint error (${status}): ${compact.slice(0, 157)}...`; +} + + + +export async function getMiniMaxUsage(apiKey: string, provider: "minimax" | "minimax-cn") { + if (!apiKey) { + return { message: "MiniMax API key not available. Add a Token Plan API key." }; + } + + const usageUrls = MINIMAX_USAGE_CONFIG[provider].usageUrls; + let lastErrorMessage = ""; + + for (let index = 0; index < usageUrls.length; index += 1) { + const usageUrl = usageUrls[index]; + const canFallback = index < usageUrls.length - 1; + + try { + const response = await fetch(usageUrl, { + method: "GET", + headers: { + Authorization: `Bearer ${apiKey}`, + Accept: "application/json", + "Content-Type": "application/json", + }, + }); + + const rawText = await response.text(); + let payload: JsonRecord = {}; + if (rawText) { + try { + payload = toRecord(JSON.parse(rawText)); + } catch { + payload = {}; + } + } + + const baseResp = toRecord(getFieldValue(payload, "base_resp", "baseResp")); + const apiStatusCode = toNumber(getFieldValue(baseResp, "status_code", "statusCode"), 0); + const apiStatusMessage = String( + getFieldValue(baseResp, "status_msg", "statusMsg") ?? "" + ).trim(); + const combinedMessage = `${apiStatusMessage} ${rawText}`.trim(); + const authLikeStatusMessage = + /token plan|coding plan|invalid api key|invalid key|unauthorized|inactive/i; + + if ( + response.status === 401 || + response.status === 403 || + apiStatusCode === 1004 || + authLikeStatusMessage.test(apiStatusMessage) + ) { + return { message: getMiniMaxAuthErrorMessage(apiStatusMessage || combinedMessage) }; + } + + if (!response.ok) { + lastErrorMessage = getMiniMaxErrorSummary(response.status, combinedMessage); + if ( + (response.status === 404 || response.status === 405 || response.status >= 500) && + canFallback + ) { + continue; + } + return { message: `MiniMax connected. ${lastErrorMessage}` }; + } + + if (rawText && Object.keys(payload).length === 0) { + return { message: "MiniMax connected. Unable to parse usage response." }; + } + + if (apiStatusCode !== 0) { + if (apiStatusMessage) { + return { message: `MiniMax connected. ${apiStatusMessage}` }; + } + return { message: "MiniMax connected. Upstream quota API returned an error." }; + } + + const capturedAtMs = Date.now(); + const modelRemains = getFieldValue(payload, "model_remains", "modelRemains"); + const allModels = Array.isArray(modelRemains) + ? modelRemains.map((item) => toRecord(item)) + : []; + const textModels = allModels.filter((model) => { + const modelName = String(getFieldValue(model, "model_name", "modelName") ?? ""); + return isMiniMaxTextQuotaModel(modelName); + }); + + if (textModels.length === 0) { + return { message: "MiniMax connected. No text quota data was returned." }; + } + + const countMeansRemaining = usageUrl.includes("/coding_plan/remains"); + const quotas: Record = {}; + + const sessionQuota = buildMiniMaxWindow( + textModels, + getMiniMaxSessionTotal, + ["current_interval_usage_count", "currentIntervalUsageCount"], + ["current_interval_remaining_percent", "currentIntervalRemainingPercent"], + ["remains_time", "remainsTime", "end_time", "endTime"], + capturedAtMs, + countMeansRemaining + ); + if (sessionQuota) { + quotas["session (5h)"] = sessionQuota; + } + + const weeklyQuota = buildMiniMaxWindow( + textModels, + getMiniMaxWeeklyTotal, + ["current_weekly_usage_count", "currentWeeklyUsageCount"], + ["current_weekly_remaining_percent", "currentWeeklyRemainingPercent"], + ["weekly_remains_time", "weeklyRemainsTime", "weekly_end_time", "weeklyEndTime"], + capturedAtMs, + countMeansRemaining + ); + if (weeklyQuota) { + quotas["weekly (7d)"] = weeklyQuota; + } + + if (Object.keys(quotas).length === 0) { + return { message: "MiniMax connected. Unable to extract text quota usage." }; + } + + return { plan: getMiniMaxPlanLabel(payload, textModels), quotas }; + } catch (error) { + lastErrorMessage = (error as Error).message; + if (!canFallback) { + break; + } + } + } + + return { + message: lastErrorMessage + ? `MiniMax connected. Unable to fetch usage: ${lastErrorMessage}` + : "MiniMax connected. Unable to fetch usage.", + }; +} + diff --git a/open-sse/services/usage/nanogpt.ts b/open-sse/services/usage/nanogpt.ts new file mode 100644 index 00000000000..ad6a83e9e18 --- /dev/null +++ b/open-sse/services/usage/nanogpt.ts @@ -0,0 +1,94 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { NANOGPT_CONFIG } from "./constants.ts"; +import { toRecord, toNumber, clampPercentage, parseResetTime } from "./utils.ts"; +import { UsageQuota } from "./types.ts"; + + + +/** + * NanoGPT Usage + * Fetches subscription-level quota from the NanoGPT API. + * Returns daily/weekly token limits and daily image limits for PRO accounts. + */ +export async function getNanoGptUsage(apiKey: string) { + if (!apiKey) { + return { message: "NanoGPT API key not available. Add a key to view usage." }; + } + + try { + const res = await fetch(NANOGPT_CONFIG.usageUrl, { + headers: { Authorization: `Bearer ${apiKey}` }, + }); + + if (!res.ok) { + if (res.status === 401) return { message: "Invalid NanoGPT API key." }; + return { message: `NanoGPT quota API error (${res.status})` }; + } + + const data = toRecord(await res.json()); + const quotas: Record = {}; + + // active -> PRO, otherwise FREE + const plan = data.active ? "PRO" : "FREE"; + + if (data.active) { + // 1. Tokens limit + // dailyInputTokens if exists, else weeklyInputTokens + let tokenQuota = toRecord(data.dailyInputTokens); + let tokenLabel = "Daily Tokens"; + if (!tokenQuota.resetAt) { + const weeklyQuota = toRecord(data.weeklyInputTokens); + if (weeklyQuota.remaining !== undefined) { + tokenQuota = weeklyQuota; + tokenLabel = "Weekly Tokens"; + } + } + + if (tokenQuota.remaining !== undefined) { + const used = toNumber(tokenQuota.used, 0); + const remaining = toNumber(tokenQuota.remaining, 0); + const total = used + remaining; + quotas[tokenLabel] = { + used, + total, + remaining, + remainingPercentage: clampPercentage(100 - toNumber(tokenQuota.percentUsed, 0) * 100), + resetAt: parseResetTime(tokenQuota.resetAt), + unlimited: false, + }; + } + + // 2. Images limit + const imageQuota = toRecord(data.dailyImages); + if (imageQuota.remaining !== undefined) { + const used = toNumber(imageQuota.used, 0); + const remaining = toNumber(imageQuota.remaining, 0); + const total = used + remaining; + quotas["Daily Images"] = { + used, + total, + remaining, + remainingPercentage: clampPercentage(100 - toNumber(imageQuota.percentUsed, 0) * 100), + resetAt: parseResetTime(imageQuota.resetAt), + unlimited: false, + }; + } + + if (Object.keys(quotas).length === 0) { + return { plan, message: "NanoGPT connected, but no active limits found." }; + } + } + + return { plan, quotas }; + } catch (error) { + return { message: `NanoGPT connected. Unable to fetch usage: ${(error as Error).message}` }; + } +} + diff --git a/open-sse/services/usage/opencode.ts b/open-sse/services/usage/opencode.ts new file mode 100644 index 00000000000..fafff107d3c --- /dev/null +++ b/open-sse/services/usage/opencode.ts @@ -0,0 +1,85 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { fetchOpencodeQuota, type OpencodeTripleWindowQuota } from "../opencodeQuotaFetcher.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { UsageQuota } from "./types.ts"; + + + +/** + * OpenCode Go / OpenCode / OpenCode Zen Usage + * Delegates to the dedicated opencodeQuotaFetcher and shapes the result into + * the standard `{ plan, quotas }` usage response expected by the limits page. + * + * Three rolling windows are surfaced: $12/5h, $30/wk, $60/mo. + */ +export async function getOpencodeUsage(connectionId: string, apiKey: string) { + if (!apiKey) { + return { message: "OpenCode API key not available. Add a key to view usage." }; + } + + try { + const quota = (await fetchOpencodeQuota(connectionId, { + apiKey, + })) as OpencodeTripleWindowQuota | null; + + if (!quota) { + return { message: "OpenCode connected. Unable to fetch quota data." }; + } + + const { window5h, windowWeekly, windowMonthly, limitReached } = quota; + + const quotas: Record = {}; + + // $12 / 5-hour rolling window + quotas["window_5h"] = { + used: window5h.percentUsed * 12, + total: 12, + remaining: (1 - window5h.percentUsed) * 12, + remainingPercentage: (1 - window5h.percentUsed) * 100, + resetAt: window5h.resetAt, + unlimited: false, + displayName: "$12 / 5-hour", + currency: "USD", + }; + + // $30 / weekly window + quotas["window_weekly"] = { + used: windowWeekly.percentUsed * 30, + total: 30, + remaining: (1 - windowWeekly.percentUsed) * 30, + remainingPercentage: (1 - windowWeekly.percentUsed) * 100, + resetAt: windowWeekly.resetAt, + unlimited: false, + displayName: "$30 / week", + currency: "USD", + }; + + // $60 / monthly window + quotas["window_monthly"] = { + used: windowMonthly.percentUsed * 60, + total: 60, + remaining: (1 - windowMonthly.percentUsed) * 60, + remainingPercentage: (1 - windowMonthly.percentUsed) * 100, + resetAt: windowMonthly.resetAt, + unlimited: false, + displayName: "$60 / month", + currency: "USD", + }; + + return { + plan: "OpenCode Go", + quotas, + limitReached, + }; + } catch (error) { + return { message: `OpenCode error: ${sanitizeErrorMessage(error)}` }; + } +} + diff --git a/open-sse/services/usage/opencodeGo.ts b/open-sse/services/usage/opencodeGo.ts new file mode 100644 index 00000000000..34b153e2dad --- /dev/null +++ b/open-sse/services/usage/opencodeGo.ts @@ -0,0 +1,198 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { JsonRecord, UsageQuota } from "./types.ts"; +import { toNumber, toPercentage, roundCurrency, clampPercentage, toRecord, parseResetTime, toTitleCase } from "./utils.ts"; +import { OpenCodeGoQuotaName, OPENCODE_GO_QUOTA_TOTALS, OPENCODE_GO_QUOTA_ORDER, OPENCODE_GO_QUOTA_URL } from "./constants.ts"; + + + +function getOpenCodeGoTokenQuotaName( + limit: JsonRecord, + existingQuotas: Record +): "session" | "weekly" { + const unit = toNumber(limit.unit, 0); + const number = toNumber(limit.number, 0); + + if (unit === 3 && number === 5) return "session"; + if (unit === 6 && number === 1) return "weekly"; + if ((unit === 4 && number === 7) || (unit === 3 && number >= 24 * 7)) return "weekly"; + + return existingQuotas.session ? "weekly" : "session"; +} + + + +function getOpenCodeGoQuotaDisplayName(quotaName: OpenCodeGoQuotaName): string { + if (quotaName === "session") return "5-hour rolling"; + if (quotaName === "weekly") return "Weekly"; + return "Monthly"; +} + + + +function normalizeOpenCodeGoQuotaToken(apiKey: string): string { + return apiKey.trim().replace(/^Bearer\s+/i, ""); +} + + + +function buildOpenCodeGoDollarQuota( + quotaName: OpenCodeGoQuotaName, + percentage: unknown, + resetAt: string | null, + usedOverride?: unknown, + details?: UsageQuota["details"] +): UsageQuota { + const total = OPENCODE_GO_QUOTA_TOTALS[quotaName]; + const percentUsed = toPercentage(percentage); + const rawUsed = toNumber(usedOverride, Number.NaN); + const used = roundCurrency( + Number.isFinite(rawUsed) ? Math.max(0, Math.min(total, rawUsed)) : (total * percentUsed) / 100 + ); + const remaining = roundCurrency(Math.max(0, total - used)); + const remainingPercentage = + total > 0 + ? clampPercentage(Math.round((remaining / total) * 100)) + : clampPercentage(100 - percentUsed); + + return { + used, + total, + remaining, + remainingPercentage, + resetAt, + unlimited: false, + displayName: getOpenCodeGoQuotaDisplayName(quotaName), + currency: "USD", + details, + }; +} + + + +function orderOpenCodeGoQuotas(quotas: Record): Record { + const ordered: Record = {}; + + for (const key of OPENCODE_GO_QUOTA_ORDER) { + if (quotas[key]) ordered[key] = quotas[key]; + } + + for (const [key, quota] of Object.entries(quotas)) { + if (!ordered[key]) ordered[key] = quota; + } + + return ordered; +} + + + +export async function getOpenCodeGoUsage(apiKey: string) { + const token = normalizeOpenCodeGoQuotaToken(apiKey); + + if (!token) { + return { message: "API key not available. Add an OpenCode Go API key to view usage." }; + } + + const res = await fetch(OPENCODE_GO_QUOTA_URL, { + headers: { + Authorization: token, + "Accept-Language": "en-US,en", + "Content-Type": "application/json", + Accept: "application/json", + }, + }); + + if (!res.ok) { + if (res.status === 401 || res.status === 403) { + return { + message: "OpenCode Go quota endpoint rejected this API key. Chat requests still work.", + }; + } + return { message: `OpenCode Go quota API error (${res.status})` }; + } + + let json: unknown; + try { + json = await res.json(); + } catch { + return { message: "OpenCode Go quota response parsing failed." }; + } + + const code = toNumber((json as Record).code, 200); + if (code === 401 || code === 403 || (json as Record).success === false) { + return { + message: "OpenCode Go quota endpoint rejected this API key. Chat requests still work.", + }; + } + + const data = toRecord((json as Record).data); + const limits: unknown[] = Array.isArray(data.limits) ? data.limits : []; + const quotas: Record = {}; + + for (const limit of limits) { + const src = toRecord(limit); + const type = String(src.type || "").toUpperCase(); + const resetAt = parseResetTime(src.nextResetTime); + + if (type === "TOKENS_LIMIT" || type === "TOKEN_LIMIT") { + const quotaName = getOpenCodeGoTokenQuotaName(src, quotas); + + quotas[quotaName] = buildOpenCodeGoDollarQuota( + quotaName, + src.percentage, + resetAt, + undefined, + Array.isArray(src.models) + ? (src.models as unknown[]).map((model) => { + const modelInfo = toRecord(model); + return { + name: String(modelInfo.model || modelInfo.modelCode || "usage"), + used: toNumber(modelInfo.percentage, 0), + }; + }) + : undefined + ); + continue; + } + + if (type === "TIME_LIMIT" || type === "TIME_USAGE_LIMIT") { + quotas.mcp_monthly = buildOpenCodeGoDollarQuota( + "mcp_monthly", + src.percentage, + resetAt, + src.currentValue, + Array.isArray(src.usageDetails) + ? src.usageDetails.map((item) => { + const detail = toRecord(item); + return { + name: String(detail.modelCode || detail.name || "usage"), + used: toNumber(detail.usage, 0), + }; + }) + : undefined + ); + } + } + + const levelRaw = + typeof data.planName === "string" + ? data.planName + : typeof data.level === "string" + ? data.level + : ""; + const planLabel = toTitleCase(levelRaw.replace(/\s*plan$/i, "")); + const plan = planLabel + ? /^opencode\s+go\b/i.test(planLabel) + ? planLabel + : `OpenCode Go ${planLabel}` + : null; + + return { plan, quotas: orderOpenCodeGoQuotas(quotas) }; +} + diff --git a/open-sse/services/usage/qoder.ts b/open-sse/services/usage/qoder.ts new file mode 100644 index 00000000000..ff2af0fb529 --- /dev/null +++ b/open-sse/services/usage/qoder.ts @@ -0,0 +1,24 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + + + + +/** + * Qoder Usage + */ +export async function getQoderUsage(accessToken?: string) { + void accessToken; + try { + // Qoder may have usage endpoint + return { message: "Qoder connected. Usage tracked per request." }; + } catch (error) { + return { message: "Unable to fetch Qoder usage." }; + } +} + diff --git a/open-sse/services/usage/qwen.ts b/open-sse/services/usage/qwen.ts new file mode 100644 index 00000000000..607a0cb1dff --- /dev/null +++ b/open-sse/services/usage/qwen.ts @@ -0,0 +1,30 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { JsonRecord } from "./types.ts"; + + + +/** + * Qwen Usage + */ +export async function getQwenUsage(accessToken?: string, providerSpecificData?: JsonRecord) { + void accessToken; + try { + const resourceUrl = providerSpecificData?.resourceUrl; + if (!resourceUrl) { + return { message: "Qwen connected. No resource URL available." }; + } + + // Qwen may have usage endpoint at resource URL + return { message: "Qwen connected. Usage tracked per request." }; + } catch (error) { + return { message: "Unable to fetch Qwen usage." }; + } +} + diff --git a/open-sse/services/usage/types.ts b/open-sse/services/usage/types.ts new file mode 100644 index 00000000000..9548dca1aba --- /dev/null +++ b/open-sse/services/usage/types.ts @@ -0,0 +1,64 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { USAGE_FETCHER_PROVIDERS } from "./core.ts"; + + + +export type JsonRecord = Record; + + +export type UsageQuota = { + used: number; + total: number; + remaining?: number; + remainingPercentage?: number; + resetAt: string | null; + unlimited: boolean; + /** + * True when the upstream provider reported the remaining fraction. False + * means the API didn't include the field and the 0 value here is a sentinel, + * NOT a confirmed-exhausted state. Antigravity-specific. + */ + fractionReported?: boolean; + quotaSource?: "retrieveUserQuota" | "fetchAvailableModels" | "localUsageHistory"; + displayName?: string; + details?: Array<{ + name: string; + used: number; + }>; + currency?: string; + grantedBalance?: number; + toppedUpBalance?: number; +}; + + +export type UsageProviderConnection = JsonRecord & { + id?: string; + provider?: string; + accessToken?: string; + apiKey?: string; + providerSpecificData?: JsonRecord; + projectId?: string; + email?: string; +}; + + +export type SubscriptionCacheEntry = { + data: unknown; + fetchedAt: number; +}; + + + +export type UsageFetcherProvider = (typeof USAGE_FETCHER_PROVIDERS)[number]; + + // Don't prevent process exit + +export interface AntigravityUsageOptions { + forceRefresh?: boolean; +} + diff --git a/open-sse/services/usage/utils.ts b/open-sse/services/usage/utils.ts new file mode 100644 index 00000000000..8bdd5966eb0 --- /dev/null +++ b/open-sse/services/usage/utils.ts @@ -0,0 +1,190 @@ + +import { + extractCodeAssistOnboardTierId, + extractCodeAssistSubscriptionTier, +} from "../codeAssistSubscription.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { JsonRecord, UsageQuota } from "./types.ts"; +import { formatGitHubQuotaSnapshot, inferGitHubPlanName } from "./github.ts"; +import { getGeminiCliPlanLabel } from "./gemini.ts"; +import { getAntigravityPlanLabel, mapCodeAssistSubscriptionToPlanLabel, mapCodeAssistTierIdToLabel, mapSubscriptionTierStringToPlanLabel } from "./antigravity.ts"; +import { getMiniMaxPlanLabel, inferMiniMaxPlanLabelFromTotals, getMiniMaxQuotaResetAt, isMiniMaxTextQuotaModel, getMiniMaxSessionTotal, getMiniMaxWeeklyTotal, createMiniMaxQuotaFromCount, createMiniMaxQuotaFromPercent, getMiniMaxRemainingPercent, getMiniMaxUsage, getMiniMaxAuthErrorMessage, getMiniMaxErrorSummary } from "./minimax.ts"; +import { getOpencodeUsage } from "./opencode.ts"; +import { getClaudePlanLabel } from "./claude.ts"; +import { getXiaomiMimoUsage } from "./xiaomi.ts"; + + + +export function toRecord(value: unknown): JsonRecord { + return value && typeof value === "object" && !Array.isArray(value) ? (value as JsonRecord) : {}; +} + + + +export function toNumber(value: unknown, fallback = 0): number { + const parsed = + typeof value === "number" + ? value + : typeof value === "string" && value.trim().length > 0 + ? Number(value) + : Number.NaN; + return Number.isFinite(parsed) ? parsed : fallback; +} + + + +export function toPercentage(value: unknown): number { + return Math.max(0, Math.min(100, toNumber(value, 0))); +} + + + +export function toTitleCase(value: string): string { + return value + .trim() + .split(/[\s_-]+/) + .filter(Boolean) + .map((part) => part.charAt(0).toUpperCase() + part.slice(1).toLowerCase()) + .join(" "); +} + + + +export function getFieldValue(source: unknown, snakeKey: string, camelKey: string): unknown { + const obj = toRecord(source); + return obj[snakeKey] ?? obj[camelKey] ?? null; +} + + + +export function clampPercentage(value: number): number { + return Math.max(0, Math.min(100, value)); +} + + + +export function roundCurrency(value: number): number { + return Math.round(value * 100) / 100; +} + + + +export function toDisplayLabel(value: string): string { + return value + .replace(/^copilot[_\s-]*/i, "") + .split(/[\s_-]+/) + .filter(Boolean) + .map((part) => { + if (/^pro\+$/i.test(part)) return "Pro+"; + if (/^[a-z]{2,}$/.test(part)) + return part.charAt(0).toUpperCase() + part.slice(1).toLowerCase(); + return part; + }) + .join(" ") + .trim(); +} + + + +export function shouldDisplayGitHubQuota(quota: UsageQuota | null): quota is UsageQuota { + if (!quota) return false; + if (quota.unlimited && quota.total <= 0) return false; + return quota.total > 0 || quota.remainingPercentage !== undefined; +} + + + +export function pickFirstNonEmptyString(...values: unknown[]): string | undefined { + for (const value of values) { + if (typeof value !== "string") continue; + const trimmed = value.trim(); + if (trimmed) return trimmed; + } + return undefined; +} + + + +export function createQuotaFromUsage( + usedValue: unknown, + totalValue: unknown, + resetValue: unknown +): UsageQuota { + const total = Math.max(0, toNumber(totalValue, 0)); + const used = total > 0 ? Math.min(Math.max(0, toNumber(usedValue, 0)), total) : 0; + const remaining = total > 0 ? Math.max(total - used, 0) : 0; + + return { + used, + total, + remaining, + remainingPercentage: total > 0 ? clampPercentage((remaining / total) * 100) : 0, + resetAt: parseResetTime(resetValue), + unlimited: false, + }; +} + + + +/** + * Parse reset date/time to ISO string + * Handles multiple formats: Unix timestamp (ms), ISO date string, etc. + */ +export function parseResetTime(resetValue: unknown): string | null { + if (!resetValue) return null; + + try { + let date: Date; + if (resetValue instanceof Date) { + date = resetValue; + } else if (typeof resetValue === "number") { + date = new Date(resetValue < 1e12 ? resetValue * 1000 : resetValue); + } else if (typeof resetValue === "string") { + date = new Date(resetValue); + } else { + return null; + } + + // Epoch-zero (1970-01-01) means no scheduled reset — treat as null + if (date.getTime() <= 0) return null; + + return date.toISOString(); + } catch (error) { + return null; + } +} + + + +export const __testing = { + parseResetTime, + formatGitHubQuotaSnapshot, + inferGitHubPlanName, + getGeminiCliPlanLabel, + getAntigravityPlanLabel, + extractCodeAssistSubscriptionTier, + extractCodeAssistOnboardTierId, + getMiniMaxPlanLabel, + inferMiniMaxPlanLabelFromTotals, + getOpencodeUsage, + getClaudePlanLabel, + createQuotaFromUsage, + getMiniMaxQuotaResetAt, + isMiniMaxTextQuotaModel, + getMiniMaxSessionTotal, + getMiniMaxWeeklyTotal, + createMiniMaxQuotaFromCount, + createMiniMaxQuotaFromPercent, + getMiniMaxRemainingPercent, + getMiniMaxUsage, + getXiaomiMimoUsage, + getMiniMaxAuthErrorMessage, + getMiniMaxErrorSummary, + mapCodeAssistSubscriptionToPlanLabel, + mapCodeAssistTierIdToLabel, + mapSubscriptionTierStringToPlanLabel, + toDisplayLabel, +}; + diff --git a/open-sse/services/usage/xiaomi.ts b/open-sse/services/usage/xiaomi.ts new file mode 100644 index 00000000000..11096355c92 --- /dev/null +++ b/open-sse/services/usage/xiaomi.ts @@ -0,0 +1,103 @@ +/** + * Usage Fetcher - Get usage data from provider APIs + */ + +import { PROVIDERS } from "../../config/constants.ts"; + +import { + getAntigravityFetchAvailableModelsUrls, + ANTIGRAVITY_BASE_URLS, +} from "../../config/antigravityUpstream.ts"; + +import { + isUserCallableAntigravityModelId, + toClientAntigravityModelId, +} from "../../config/antigravityModelAliases.ts"; + +import { isUserCallableAgyModelId } from "../../config/agyModels.ts"; + +import { getGlmQuotaUrl } from "../../config/glmProvider.ts"; + +import { getGitHubCopilotInternalUserHeaders } from "../../config/providerHeaderProfiles.ts"; + +import { getDbInstance } from "@/lib/db/core"; + +import { fetchBailianQuota, type BailianTripleWindowQuota } from "../bailianQuotaFetcher.ts"; + +import { fetchDeepseekQuota, type DeepseekQuota } from "../deepseekQuotaFetcher.ts"; + +import { fetchOpencodeQuota, type OpencodeTripleWindowQuota } from "../opencodeQuotaFetcher.ts"; + +import { + applyAntigravityClientProfileHeaders, + getAntigravityBootstrapHeaders, + getAntigravityClientProfile, +} from "../antigravityClientProfile.ts"; + +import { + antigravityUserAgent, + getAntigravityHeaders, + getAntigravityLoadCodeAssistMetadata, +} from "../antigravityHeaders.ts"; + +import { + getAntigravityRemainingCredits, + updateAntigravityRemainingCredits, +} from "../../executors/antigravity.ts"; + +import { getCreditsMode } from "../antigravityCredits.ts"; + +import { CLAUDE_CODE_VERSION, fetchClaudeBootstrap } from "../../executors/claudeIdentity.ts"; + +import { generateAntigravityRequestId, getAntigravitySessionId } from "../antigravityIdentity.ts"; + +import { + extractCodeAssistOnboardTierId, + extractCodeAssistSubscriptionTier, +} from "../codeAssistSubscription.ts"; + +import { sanitizeErrorMessage } from "../../utils/error.ts"; + +import { createQuotaFromUsage } from "./utils.ts"; + + + +// Xiaomi MiMo Token Plan monthly limit (tokens). Keep in sync with the +// "xiaomi-mimo" preset in src/lib/quota/planRegistry.ts. +const XIAOMI_MIMO_MONTHLY_TOKEN_LIMIT = 4_100_000_000; + + + +/** + * Xiaomi MiMo — SELF-TRACKED monthly quota. + * + * Xiaomi exposes plan usage only behind the console session cookie (the API key + * cannot reach the `tokenPlan/usage` endpoint), so there is no upstream usage + * API to call. Instead we count the tokens OmniRoute itself routed to this + * connection in the current UTC month (from `usage_history`) and compare them + * to the known Token Plan monthly limit. This reflects only traffic that went + * through OmniRoute, not the provider's own dashboard figure. + */ +export async function getXiaomiMimoUsage(connectionId: string) { + if (!connectionId) { + return { message: "Xiaomi MiMo: connection id unavailable for self-tracked quota." }; + } + try { + const { getMonthlyProviderTokensForConnection } = await import("@/lib/usage/usageStats"); + const used = getMonthlyProviderTokensForConnection("xiaomi-mimo", connectionId); + const total = XIAOMI_MIMO_MONTHLY_TOKEN_LIMIT; + const now = new Date(); + const resetAt = new Date( + Date.UTC(now.getUTCFullYear(), now.getUTCMonth() + 1, 1) + ).toISOString(); + return { + plan: "Xiaomi MiMo Token Plan (OmniRoute-tracked)", + quotas: { + monthly: createQuotaFromUsage(used, total, resetAt), + }, + }; + } catch (error) { + return { message: `Xiaomi MiMo self-tracked usage error: ${(error as Error).message}` }; + } +} + diff --git a/open-sse/translator/request/openai-to-claude.ts b/open-sse/translator/request/openai-to-claude.ts index 2671d7a5e88..7db6062331f 100644 --- a/open-sse/translator/request/openai-to-claude.ts +++ b/open-sse/translator/request/openai-to-claude.ts @@ -212,7 +212,7 @@ export function openaiToClaudeRequest(model, body, stream) { if (body.temperature !== undefined) { result.temperature = body.temperature; } - if (body.top_p !== undefined) { + if (body.temperature === undefined && body.top_p !== undefined) { result.top_p = body.top_p; } if (body.stop !== undefined) { diff --git a/open-sse/translator/request/openai-to-gemini.ts b/open-sse/translator/request/openai-to-gemini.ts index 8cf99546021..c1453aafef2 100644 --- a/open-sse/translator/request/openai-to-gemini.ts +++ b/open-sse/translator/request/openai-to-gemini.ts @@ -816,7 +816,7 @@ register( FORMATS.GEMINI, (model, body, stream = false, credentials = null) => openaiToGeminiRequest(model, body, stream, credentials, { - signaturelessToolCallMode: "native", + signaturelessToolCallMode: "context", }), null ); diff --git a/open-sse/translator/response/gemini-to-openai.ts b/open-sse/translator/response/gemini-to-openai.ts index a07fa15ef93..a99c3d671b4 100644 --- a/open-sse/translator/response/gemini-to-openai.ts +++ b/open-sse/translator/response/gemini-to-openai.ts @@ -387,16 +387,9 @@ export function geminiToOpenAIResponse(chunk, state) { } if (hasFunctionCall) { - // Flush any still-open textual reasoning wrapper as reasoning_content BEFORE - // the tool call. A signed native functionCall arriving while a `` - // (etc.) tag opened in an earlier chunk is still buffered must not silently - // drop that buffered reasoning — flushOpenTextualReasoning emits it and clears - // the active-tag/content buffers. (LEDGER-4 / #3821-review) - flushOpenTextualReasoning(state, results); - // Also drop any partial open-tag fragment buffered at a chunk boundary - // (flushOpenTextualReasoning early-returns when only this is set), matching the - // pre-fix branch which cleared all three buffers. (#3821-review convergence) + state.activeTextualReasoningTag = undefined; state.textualReasoningTagBuffer = undefined; + state.textualReasoningContentBuffer = undefined; emitFunctionCallPart(part, state, results); } continue; diff --git a/open-sse/tsconfig.json b/open-sse/tsconfig.json index 6a35a3d3e16..f64585be7e0 100644 --- a/open-sse/tsconfig.json +++ b/open-sse/tsconfig.json @@ -7,6 +7,7 @@ "checkJs": true, "noEmit": true, "allowImportingTsExtensions": true, + "resolveJsonModule": true, "skipLibCheck": true, "esModuleInterop": true, "strict": false, diff --git a/open-sse/utils/proxyDispatcher.ts b/open-sse/utils/proxyDispatcher.ts index acd399315d8..4b49eaf7706 100644 --- a/open-sse/utils/proxyDispatcher.ts +++ b/open-sse/utils/proxyDispatcher.ts @@ -138,8 +138,15 @@ function buildProxyUrlString(parsed: URL, port: string): string { return `${parsed.protocol}//${auth}${parsed.hostname}:${port}`; } +/** + * SOCKS5 proxy support defaults ON (opt-OUT). A fresh deploy with no env set + * should honour SOCKS5 proxies out of the box — they were silently rejected + * before (default OFF), making accounts fall back to the host IP. Only an + * explicit falsey value (false/0/no/off) disables it. + */ export function isSocks5ProxyEnabled(): boolean { - return process.env.ENABLE_SOCKS5_PROXY === "true"; + const raw = (process.env.ENABLE_SOCKS5_PROXY ?? "").trim().toLowerCase(); + return !["false", "0", "no", "off"].includes(raw); } export function proxyUrlForLogs(proxyUrl: string): string { @@ -173,7 +180,7 @@ export function normalizeProxyUrl( } if (parsed.protocol === "socks5:" && !allowSocks5) { throw new Error( - "[ProxyDispatcher] SOCKS5 proxy is disabled (set ENABLE_SOCKS5_PROXY=true to enable)" + "[ProxyDispatcher] SOCKS5 proxy is disabled (remove ENABLE_SOCKS5_PROXY=false to enable — it is ON by default)" ); } if (!parsed.hostname) { @@ -233,7 +240,7 @@ export function proxyConfigToUrl( } if (protocol === "socks5:" && !allowSocks5) { throw new Error( - "[ProxyDispatcher] SOCKS5 proxy is disabled (set ENABLE_SOCKS5_PROXY=true to enable)" + "[ProxyDispatcher] SOCKS5 proxy is disabled (remove ENABLE_SOCKS5_PROXY=false to enable — it is ON by default)" ); } diff --git a/open-sse/utils/stream.ts b/open-sse/utils/stream.ts index 4091d5cc383..48633fdc7cd 100644 --- a/open-sse/utils/stream.ts +++ b/open-sse/utils/stream.ts @@ -109,7 +109,11 @@ function normalizeResponsesSseIds(payload: JsonRecord): boolean { } } - if (payload.response && typeof payload.response === "object" && !Array.isArray(payload.response)) { + if ( + payload.response && + typeof payload.response === "object" && + !Array.isArray(payload.response) + ) { const response = payload.response as JsonRecord; let responseChanged = false; const normalizedResponse = { ...response }; @@ -1016,6 +1020,32 @@ export function createSSEStream(options: StreamOptions = {}) { } }; + const emitClaudeEmptyStreamErrorAndAbort = ( + controller: TransformStreamDefaultController, + decrementPendingRequest = true + ) => { + clearIdleTimer(); + const msg = "Claude returned an empty response (no content block)"; + console.warn( + `[STREAM] Empty Claude stream at flush - emitting error (${provider || "provider"}:${model || "unknown"})` + ); + const errorBody = buildErrorBody(502, msg); + const errorEvent: Record = { type: "error", error: errorBody.error }; + const errOutput = formatSSE(errorEvent, FORMATS.CLAUDE); + reqLogger?.appendConvertedChunk?.(errOutput); + clientPayloadCollector.push(errorEvent); + controller.enqueue(encoder.encode(errOutput)); + if (onFailure) { + try { + void onFailure({ status: 502, message: msg, code: "empty_response" }); + } catch {} + } + if (decrementPendingRequest) { + trackPendingRequest(model, provider, connectionId, false); + } + controller.error(markPendingRequestCleared(new Error(msg))); + }; + const emitTranslatedClientItem = ( controller: TransformStreamDefaultController, item: Record @@ -1059,13 +1089,8 @@ export function createSSEStream(options: StreamOptions = {}) { sourceFormat === FORMATS.CLAUDE && shouldInjectClaudeEmptyResponseBeforeCurrentEvent(claudeEmptyResponseLifecycle, itemSanitized) ) { - const eventType = getClaudeEventType(itemSanitized); - emitSyntheticClaudeEmptyResponse(controller, { - includeContentBlock: true, - includeMessageDelta: - eventType === "message_stop" && !claudeEmptyResponseLifecycle.hasMessageDelta, - includeMessageStop: false, - }); + emitClaudeEmptyStreamErrorAndAbort(controller); + return; } if (sourceFormat === FORMATS.CLAUDE && isClaudeEventPayload(itemSanitized)) { @@ -1300,12 +1325,8 @@ export function createSSEStream(options: StreamOptions = {}) { type: eventType, }) ) { - emitSyntheticClaudeEmptyResponse(controller, { - includeContentBlock: true, - includeMessageDelta: - eventType === "message_stop" && !claudeEmptyResponseLifecycle.hasMessageDelta, - includeMessageStop: false, - }); + emitClaudeEmptyStreamErrorAndAbort(controller); + return; } pendingPassthroughEventLine = line; @@ -1337,7 +1358,8 @@ export function createSSEStream(options: StreamOptions = {}) { // clients like OpenCode, so drop it only for Responses-native consumers. const hasActiveDeltaValue = (value: unknown): boolean => { if (typeof value === "string") return value.length > 0; - if (Array.isArray(value)) return value.some((entry) => hasActiveDeltaValue(entry)); + if (Array.isArray(value)) + return value.some((entry) => hasActiveDeltaValue(entry)); if (value && typeof value === "object") { return Object.values(value).some((entry) => hasActiveDeltaValue(entry)); } @@ -1605,7 +1627,12 @@ export function createSSEStream(options: StreamOptions = {}) { parsed, passthroughResponsesOutputItems ); - if (stripped || backfilled || textualToolCallBackfilled || responsesIdsNormalized) { + if ( + stripped || + backfilled || + textualToolCallBackfilled || + responsesIdsNormalized + ) { output = `data: ${JSON.stringify(parsed)}\n`; injectedUsage = true; } @@ -1632,13 +1659,8 @@ export function createSSEStream(options: StreamOptions = {}) { parsed ) ) { - emitSyntheticClaudeEmptyResponse(controller, { - includeContentBlock: true, - includeMessageDelta: - parsed.type === "message_stop" && - !claudeEmptyResponseLifecycle.hasMessageDelta, - includeMessageStop: false, - }); + emitClaudeEmptyStreamErrorAndAbort(controller); + return; } updateClaudeEmptyResponseLifecycle(claudeEmptyResponseLifecycle, parsed); const restoredToolName = restoreClaudePassthroughToolUseName(parsed, toolNameMap); @@ -1708,14 +1730,16 @@ export function createSSEStream(options: StreamOptions = {}) { !parsed.choices[0].delta.reasoning_content ); const hadNonStringToolCallId = Array.isArray(parsed.choices) - ? parsed.choices.some((choice) => - Array.isArray(choice?.delta?.tool_calls) && - choice.delta.tool_calls.some( - (tc) => tc?.id != null && typeof tc.id !== "string" - ) + ? parsed.choices.some( + (choice) => + Array.isArray(choice?.delta?.tool_calls) && + choice.delta.tool_calls.some( + (tc) => tc?.id != null && typeof tc.id !== "string" + ) ) : false; - const hadNonStringTopLevelId = parsed?.id != null && typeof parsed.id !== "string"; + const hadNonStringTopLevelId = + parsed?.id != null && typeof parsed.id !== "string"; parsed = sanitizeStreamingChunk(parsed); if ( @@ -2148,13 +2172,8 @@ export function createSSEStream(options: StreamOptions = {}) { bufferedPayload ) ) { - const eventType = getClaudeEventType(bufferedPayload); - emitSyntheticClaudeEmptyResponse(controller, { - includeContentBlock: true, - includeMessageDelta: - eventType === "message_stop" && !claudeEmptyResponseLifecycle.hasMessageDelta, - includeMessageStop: false, - }); + emitClaudeEmptyStreamErrorAndAbort(controller, false); + return; } if (isClaudeEventPayload(bufferedPayload)) { updateClaudeEmptyResponseLifecycle(claudeEmptyResponseLifecycle, bufferedPayload); @@ -2164,7 +2183,8 @@ export function createSSEStream(options: StreamOptions = {}) { // Normalize numeric IDs for final buffered data: chunk (same as transform path) if (typeof bufferedPayload === "object" && !Array.isArray(bufferedPayload)) { const flushedParsed = bufferedPayload as JsonRecord; - const flushedType = typeof flushedParsed.type === "string" ? flushedParsed.type : ""; + const flushedType = + typeof flushedParsed.type === "string" ? flushedParsed.type : ""; const isResponses = flushedType.startsWith("response."); const isClaude = isClaudeEventPayload(flushedParsed); if (isResponses) { @@ -2181,7 +2201,9 @@ export function createSSEStream(options: StreamOptions = {}) { } if (Array.isArray(flushedParsed.choices)) { for (const choice of flushedParsed.choices as JsonRecord[]) { - const tcs = (choice as JsonRecord | undefined)?.delta as JsonRecord | undefined; + const tcs = (choice as JsonRecord | undefined)?.delta as + | JsonRecord + | undefined; if (Array.isArray(tcs?.tool_calls)) { for (const tc of tcs.tool_calls as JsonRecord[]) { if (tc?.id != null && typeof tc.id !== "string") { @@ -2208,11 +2230,8 @@ export function createSSEStream(options: StreamOptions = {}) { } if (shouldInjectClaudeEmptyResponseOnFlush(claudeEmptyResponseLifecycle)) { - emitSyntheticClaudeEmptyResponse(controller, { - includeContentBlock: true, - includeMessageDelta: !claudeEmptyResponseLifecycle.hasMessageDelta, - includeMessageStop: !claudeEmptyResponseLifecycle.hasMessageStop, - }); + emitClaudeEmptyStreamErrorAndAbort(controller, false); + return; } else if (shouldInjectClaudeMissingFinalizersOnFlush(claudeEmptyResponseLifecycle)) { emitSyntheticClaudeEmptyResponse(controller, { includeContentBlock: false, @@ -2489,11 +2508,8 @@ export function createSSEStream(options: StreamOptions = {}) { if (sourceFormat === FORMATS.CLAUDE) { if (shouldInjectClaudeEmptyResponseOnFlush(claudeEmptyResponseLifecycle)) { - emitSyntheticClaudeEmptyResponse(controller, { - includeContentBlock: true, - includeMessageDelta: !claudeEmptyResponseLifecycle.hasMessageDelta, - includeMessageStop: !claudeEmptyResponseLifecycle.hasMessageStop, - }); + emitClaudeEmptyStreamErrorAndAbort(controller, false); + return; } else if (shouldInjectClaudeMissingFinalizersOnFlush(claudeEmptyResponseLifecycle)) { emitSyntheticClaudeEmptyResponse(controller, { includeContentBlock: false, @@ -2631,7 +2647,7 @@ export function createSSEStream(options: StreamOptions = {}) { ); } -export default createSSEStream +export default createSSEStream; // Convenience functions for backward compatibility export function createSSETransformStreamWithLogger( diff --git a/package-lock.json b/package-lock.json index 78cbbecb02f..fa09e1edb86 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "omniroute", - "version": "3.8.22", + "version": "3.8.21", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "omniroute", - "version": "3.8.22", + "version": "3.8.21", "hasInstallScript": true, "license": "MIT", "workspaces": [ @@ -21587,7 +21587,7 @@ }, "open-sse": { "name": "@omniroute/open-sse", - "version": "3.8.22" + "version": "3.8.21" } } } diff --git a/package.json b/package.json index 07985c58001..735ba1543f8 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "omniroute", - "version": "3.8.22", + "version": "3.8.21", "description": "Unified AI router with 160+ providers, RTK+Caveman compression, auto fallback, MCP/A2A, desktop, PWA, and OpenAI-compatible APIs.", "type": "module", "bin": { @@ -31,14 +31,7 @@ "scripts/build/native-binary-compat.mjs", "scripts/build/build-next-isolated.mjs", "README.md", - "LICENSE", - "!**/__tests__/**", - "!**/*.test.ts", - "!**/*.test.tsx", - "!**/*.test.js", - "!**/*.test.mjs", - "!**/*.spec.ts", - "!**/*.spec.tsx" + "LICENSE" ], "workspaces": [ "open-sse" diff --git a/public/audio/ui-notify.mp3 b/public/audio/ui-notify.mp3 new file mode 100644 index 00000000000..326d3fa5337 Binary files /dev/null and b/public/audio/ui-notify.mp3 differ diff --git a/scripts/build/prepublish.ts b/scripts/build/prepublish.ts index b999c5c6f51..b5c2e14dfba 100644 --- a/scripts/build/prepublish.ts +++ b/scripts/build/prepublish.ts @@ -129,7 +129,10 @@ if (!existsSync(standaloneServerJs)) { stdio: "inherit", }); if (!existsSync(standaloneServerJs)) { - console.error("\n ❌ Standalone build not found after `npm run build` at:", standaloneServerJs); + console.error( + "\n ❌ Standalone build not found after `npm run build` at:", + standaloneServerJs + ); console.error(" Make sure next.config.mjs has: output: 'standalone'"); process.exit(1); } @@ -263,6 +266,53 @@ if (existsSync(cliSrcFile)) { } } +// ── Step 8.8: Build @omniroute/opencode-plugin ────────────── +// The plugin ships bundled inside the omniroute npm package (see root +// package.json "files": ["@omniroute/", ...]). Its built `dist/` MUST be +// present in the publish tarball so `omniroute setup opencode` can copy it +// into the user's OpenCode plugin dir. If the build fails we surface the +// error — shipping without the plugin's dist breaks the documented install +// flow for every downstream user. +const opencodePluginSrc = join(ROOT, "@omniroute", "opencode-plugin"); +const opencodePluginDist = join(opencodePluginSrc, "dist", "index.js"); +const opencodePluginCjs = join(opencodePluginSrc, "dist", "index.cjs"); +if (existsSync(opencodePluginSrc) && existsSync(join(opencodePluginSrc, "package.json"))) { + const pluginAlreadyBuilt = existsSync(opencodePluginDist) && existsSync(opencodePluginCjs); + if (!pluginAlreadyBuilt) { + console.log("\n 🔨 Building @omniroute/opencode-plugin (tsup)..."); + try { + // The plugin is a standalone package (not an npm workspace), so the root + // install never populates its node_modules — and tsup with `dts: true` + // needs the plugin's own devDependencies (typescript, @opencode-ai/plugin + // types). Without this install a fresh CI publish fails at this step. + if (!existsSync(join(opencodePluginSrc, "node_modules"))) { + const NPM_BIN = process.platform === "win32" ? "npm.cmd" : "npm"; + execFileSync(NPM_BIN, ["install", "--no-audit", "--no-fund"], { + cwd: opencodePluginSrc, + stdio: "inherit", + }); + } + execFileSync(NPX_BIN, ["tsup"], { + cwd: opencodePluginSrc, + stdio: "inherit", + env: { ...process.env, NODE_ENV: "production" }, + }); + console.log(" ✅ @omniroute/opencode-plugin bundled to @omniroute/opencode-plugin/dist/"); + } catch (err: any) { + console.error(" ❌ Failed to build @omniroute/opencode-plugin:", err.message); + console.error(" The published package would be missing the plugin dist."); + console.error( + " Run `cd @omniroute/opencode-plugin && npm install && npm run build` to debug." + ); + process.exit(1); + } + } else { + console.log(" ✅ @omniroute/opencode-plugin dist/ already present (skipping rebuild)"); + } +} else { + console.log(" ⏭️ @omniroute/opencode-plugin not found in workspace (skipping build)"); +} + // ── Step 9: Copy shared utilities needed at runtime ──────── const sharedApiKey = join(ROOT, "src", "shared", "utils", "apiKey.js"); const sharedApiKeyDest = join(DIST_DIR, "src", "shared", "utils"); @@ -379,7 +429,9 @@ const remainingUnexpectedFiles = findUnexpectedArtifactPaths(walkFiles(DIST_DIR) if (remainingUnexpectedFiles.length > 0) { console.error("\n ❌ Staged dist/ still contains unexpected publish artifacts:"); - remainingUnexpectedFiles.forEach((violation: string) => console.error(` - dist/${violation}`)); + remainingUnexpectedFiles.forEach((violation: string) => + console.error(` - dist/${violation}`) + ); process.exit(1); } diff --git a/scripts/check/check-fetch-targets.mjs b/scripts/check/check-fetch-targets.mjs index b51517e2084..6c504376acc 100644 --- a/scripts/check/check-fetch-targets.mjs +++ b/scripts/check/check-fetch-targets.mjs @@ -24,7 +24,7 @@ const IGNORE = [ // inventada. CADA UM precisa de triagem: criar a rota, corrigir o path, ou remover a // chamada morta. NÃO adicione novos aqui sem justificativa — esse é o ponto do gate. const KNOWN_MISSING = new Set([ - // All previously known-missing routes have been resolved. + "/api/settings/obsidian/webdav", // ObsidianSourceCard.tsx — só existe /api/settings/obsidian ]); function walk(dir, acc = []) { diff --git a/src/app/(dashboard)/dashboard/providers/[id]/ProviderDetailPageClient.tsx b/src/app/(dashboard)/dashboard/providers/[id]/ProviderDetailPageClient.tsx index 47aa3ffe05b..b605b066778 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/ProviderDetailPageClient.tsx +++ b/src/app/(dashboard)/dashboard/providers/[id]/ProviderDetailPageClient.tsx @@ -1,250 +1,55 @@ "use client"; -import { useState, useEffect, useCallback, useRef, useMemo } from "react"; -// Phase 1f extractions — Issue #3501 -import { useProviderConnections } from "./hooks/useProviderConnections"; -import { useProviderSettings } from "./hooks/useProviderSettings"; -import { useProviderModels } from "./hooks/useProviderModels"; -import { LlmChatCard } from "@/app/(dashboard)/dashboard/media-providers/components/LlmChatCard"; -import { ServiceKindTabs } from "@/app/(dashboard)/dashboard/media-providers/components/ServiceKindTabs"; -import { EmbeddingExampleCard } from "@/app/(dashboard)/dashboard/media-providers/components/EmbeddingExampleCard"; -import { ImageExampleCard } from "@/app/(dashboard)/dashboard/media-providers/components/ImageExampleCard"; -import { TtsExampleCard } from "@/app/(dashboard)/dashboard/media-providers/components/TtsExampleCard"; -import { SttExampleCard } from "@/app/(dashboard)/dashboard/media-providers/components/SttExampleCard"; -import { WebSearchExampleCard } from "@/app/(dashboard)/dashboard/media-providers/components/WebSearchExampleCard"; -import { WebFetchExampleCard } from "@/app/(dashboard)/dashboard/media-providers/components/WebFetchExampleCard"; -import { VideoExampleCard } from "@/app/(dashboard)/dashboard/media-providers/components/VideoExampleCard"; -import { MusicExampleCard } from "@/app/(dashboard)/dashboard/media-providers/components/MusicExampleCard"; -import type { ServiceKind } from "@/shared/constants/providers"; -import { useNotificationStore } from "@/store/notificationStore"; -import { useParams, useRouter } from "next/navigation"; +// Issue #3501 strangler-fig decomposition — Phase 1t (final push) +import { useState, useEffect, useCallback, useMemo } from "react"; +import { useParams } from "next/navigation"; import Link from "next/link"; import { useTranslations } from "next-intl"; +import { Card, Button, CardSkeleton, NoAuthProviderCard, NoAuthAccountCard } from "@/shared/components"; import { - Card, - Button, - Badge, - Modal, - ConfirmModal, - CardSkeleton, - OAuthModal, - KiroOAuthWrapper, - CursorAuthModal, - TraeAuthModal, - Toggle, - Select, - ProxyConfigModal, - NoAuthProviderCard, - NoAuthAccountCard, -} from "@/shared/components"; -import { - LOCAL_PROVIDERS, NOAUTH_PROVIDERS, - AI_PROVIDERS, getProviderAlias, isOpenAICompatibleProvider, isAnthropicCompatibleProvider, isClaudeCodeCompatibleProvider, - isSelfHostedChatProvider, supportsApiKeyOnFreeProvider, - // providerAllowsOptionalApiKey + supportsBulkApiKey used by extracted AddApiKeyModal } from "@/shared/constants/providers"; -// antigravityClientProfile + parseBulkApiKeys used by extracted modals (AddApiKeyModal, EditConnectionModal) import { getModelsByProviderId } from "@/shared/constants/models"; -import { - compatibleProviderSupportsModelImport, - getCompatibleFallbackModels, -} from "@/lib/providers/managedAvailableModels"; -import { - matchesModelCatalogQuery, - normalizeModelCatalogSource, -} from "@/shared/utils/modelCatalogSearch"; +import { compatibleProviderSupportsModelImport, getCompatibleFallbackModels } from "@/lib/providers/managedAvailableModels"; +import { normalizeModelCatalogSource } from "@/shared/utils/modelCatalogSearch"; import { useCopyToClipboard } from "@/shared/hooks/useCopyToClipboard"; -import { pickDisplayValue } from "@/shared/utils/maskEmail"; import useEmailPrivacyStore from "@/store/emailPrivacyStore"; -import EmailPrivacyToggle from "@/shared/components/EmailPrivacyToggle"; -import ProviderIcon from "@/shared/components/ProviderIcon"; -import { type CodexServiceTier } from "@/lib/providers/requestDefaults"; -import { type CodexGlobalServiceMode } from "@/lib/providers/codexFastTier"; -// parseExtraApiKeys used by extracted EditConnectionModal -import { compareTr } from "@/shared/utils/turkishText"; -import RiskNoticeModal from "../components/RiskNoticeModal"; -import CodexCliGuideModal from "../components/CodexCliGuideModal"; -import { isRiskAcknowledged, useRiskAcknowledged } from "../hooks/useRiskAcknowledged"; +import { useNotificationStore } from "@/store/notificationStore"; import { resolveDashboardProviderInfo } from "../providerPageUtils"; -// webSessionCredentials used by extracted modals (AddApiKeyModal, EditConnectionModal) -import { - ImportCodexAuthModal, - ApplyCodexAuthModal, -} from "./components/modals/ImportCodexAuthModal"; -import { - ImportClaudeAuthModal, - ApplyClaudeAuthModal, -} from "./components/modals/ImportClaudeAuthModal"; -import { - ImportGeminiAuthModal, - ApplyGeminiAuthModal, -} from "./components/modals/ImportGeminiAuthModal"; - -import EditCompatibleNodeModal from "./components/modals/EditCompatibleNodeModal"; -import AddApiKeyModal from "./components/modals/AddApiKeyModal"; -import EditConnectionModal from "./components/modals/EditConnectionModal"; -import WebSessionCredentialGuide from "./components/WebSessionCredentialGuide"; -// Phase 1d extractions — Issue #3501 -import ConnectionRow, { - type ConnectionRowConnection, -} from "./components/ConnectionRow"; -import ModelCompatPopover from "./components/ModelCompatPopover"; -import SiliconFlowEndpointModal from "./components/SiliconFlowEndpointModal"; -import { CC_COMPATIBLE_DEFAULT_CHAT_PATH } from "./providerDetailConstants"; -import { - // CONFIGURABLE_BASE_URL_PROVIDERS, DEFAULT_PROVIDER_BASE_URLS, getLocalProviderMetadata, - // isBaseUrlConfigurableProvider, getProviderBaseUrlDefault, getProviderBaseUrlHint, - // getProviderBaseUrlPlaceholder, isGlmProvider, parseRoutingTagsInput, parseExcludedModelsInput, - // formatRoutingTagsInput, formatExcludedModelsInput, getWebSessionCredentialLabel, - // getWebSessionCredentialHint, getWebSessionCredentialCheckLabel, getAddCredentialModalTitle, - // CODEX_REASONING_STRENGTH_OPTIONS, CODEX_ACCOUNT_SERVICE_TIER_VALUES, getCodexRequestDefaults, - // getClaudeCodeCompatibleRequestDefaults, extractCommandCodeCredentialInput, - // normalizeAndValidateHttpBaseUrl, formatTimeAgo - // — all moved to extracted modals (AddApiKeyModal, EditConnectionModal, WebSessionCredentialGuide) - providerText, - providerCountText, - readBooleanToggle, - formatProviderModelsErrorResponse, - type ProviderMessageTranslator, - type LocalProviderMetadata, - type CommandCodeAuthFlowState, - type CompatByProtocolMap, - type CompatModelRow, - type CompatModelMap, -} from "./providerPageHelpers"; -// CODEX_GLOBAL_SERVICE_MODE_VALUES, getCodexServiceTierLabel, normalizeCodexLimitPolicy -// moved to hooks/useProviderSettings.ts + hooks/useProviderConnections.ts (Phase 1f) -// Phase 1e extractions — Issue #3501 +import { type ConnectionRowConnection } from "./components/ConnectionRow"; +import { useProviderConnections } from "./hooks/useProviderConnections"; +import { useProviderSettings } from "./hooks/useProviderSettings"; +import { useProviderModels } from "./hooks/useProviderModels"; +import { useCommandCodeAuth } from "./hooks/useCommandCodeAuth"; +import { useExternalLinkFlow } from "./hooks/useExternalLinkFlow"; +import { useAuthFileHandlers } from "./hooks/useAuthFileHandlers"; +import { useModelImportHandlers } from "./hooks/useModelImportHandlers"; +import { useApiKeySave } from "./hooks/useApiKeySave"; +import { useModelVisibilityHandlers } from "./hooks/useModelVisibilityHandlers"; import { useModelCompatState } from "./hooks/useModelCompatState"; -import ModelRow, { ModelVisibilityToolbar } from "./components/ModelRow"; -import PassthroughModelsSection from "./components/PassthroughModelsSection"; +import { useConnectionGate } from "./hooks/useConnectionGate"; +import { useProviderNodeActions } from "./hooks/useProviderNodeActions"; +import ProviderPlaygroundPanel from "./components/ProviderPlaygroundPanel"; +import ProviderModelsSection from "./components/ProviderModelsSection"; import CustomModelsSection from "./components/CustomModelsSection"; -import CompatibleModelsSection from "./components/CompatibleModelsSection"; -// recordToHeaderRows moved to components/ModelCompatPopover.tsx (Phase 1d) -// buildCompatMap, isModelHidden*, effectiveNormalize/Preserve*, anyNormalize/NoPreserveCompatBadge -// moved to providerPageHelpers.ts + hook useModelCompatState (Phase 1e) -// formatProviderModelsErrorResponse moved to providerPageHelpers.ts (Phase 1e) - -/** PATCH fields for provider model compat (matches API + `ModelCompatPerProtocol` shape). */ -type ModelCompatSavePatch = { - normalizeToolCallId?: boolean; - preserveOpenAIDeveloperRole?: boolean; - upstreamHeaders?: Record; - compatByProtocol?: CompatByProtocolMap; - isHidden?: boolean; -}; - -// MAX_BULK_IDS moved to hooks/useProviderConnections.ts (Phase 1f) -// ModelRowProps, PassthroughModelRowProps → components/ModelRow.tsx, PassthroughModelRow.tsx (Phase 1e) -// PassthroughModelsSectionProps → components/PassthroughModelsSection.tsx (Phase 1e) -// CustomModelsSectionProps → components/CustomModelsSection.tsx (Phase 1e) -// CompatibleModelsSectionProps → components/CompatibleModelsSection.tsx (Phase 1e) -// CooldownTimerProps moved to components/ConnectionRow.tsx (Phase 1d) - -// getModelSourceBadgeClass + ModelSourceBadge → components/ModelRow.tsx (Phase 1e) -// ConnectionRowConnection, ConnectionRowProps moved to components/ConnectionRow.tsx (Phase 1d) - -// ModelCompatPopover extracted to components/ModelCompatPopover.tsx (Phase 1d) - -// ──── ProviderPlaygroundPanel ──────────────────────────────────────────────── -// Renders a playground section on the individual provider page. -// Shows ServiceKindTabs if the provider declares multiple kinds; falls back to -// a single-kind panel or the LlmChatCard for standard LLM providers. - -const MEDIA_SERVICE_KINDS: ServiceKind[] = [ - "embedding", - "image", - "tts", - "stt", - "webSearch", - "webFetch", - "video", - "music", -]; - -function renderKindPanel(kind: ServiceKind, providerId: string): JSX.Element | null { - switch (kind) { - case "llm": - return ; - case "embedding": - return ; - case "image": - return ; - case "tts": - return ; - case "stt": - return ; - case "webSearch": - return ; - case "webFetch": - return ; - case "video": - return ; - case "music": - return ; - default: - return null; - } -} - -function ProviderPlaygroundPanel({ providerId }: { providerId: string }) { - // Resolve serviceKinds from AI_PROVIDERS. - // For providers without explicit serviceKinds (most LLM providers), we infer - // "llm" as the default. - const providerEntry = AI_PROVIDERS[providerId as keyof typeof AI_PROVIDERS] as - | (Record & { serviceKinds?: string[] }) - | undefined; - - const rawKinds: string[] = providerEntry?.serviceKinds ?? []; - - const ALL_VALID_KINDS = [ - "llm", - "embedding", - "image", - "imageToText", - "tts", - "stt", - "webSearch", - "webFetch", - "video", - "music", - ] as const; - - const kinds: ServiceKind[] = - rawKinds.length > 0 - ? rawKinds.filter((k): k is ServiceKind => (ALL_VALID_KINDS as readonly string[]).includes(k)) - : ["llm"]; - - // Filter out kinds that have no playground implementation yet - const playgroundableKinds = kinds.filter((k) => k !== "imageToText"); - - // useState must be called unconditionally (Rules of Hooks) - const [activeKind, setActiveKind] = useState(playgroundableKinds[0] ?? "llm"); - - if (playgroundableKinds.length === 0) return null; - - return ( -
-

Playground

- - {renderKindPanel(activeKind, providerId)} -
- ); -} +import ConnectionsListPanel from "./components/ConnectionsListPanel"; +import ConnectionsHeaderToolbar from "./components/ConnectionsHeaderToolbar"; +import ZedImportCard from "./components/ZedImportCard"; +import ProviderPageHeader from "./components/ProviderPageHeader"; +import CompatibleNodeCard from "./components/CompatibleNodeCard"; +import ProviderModalsPanel from "./components/ProviderModalsPanel"; +import EmptyConnectionsPlaceholder from "./components/EmptyConnectionsPlaceholder"; +import UpstreamProxyCard from "./components/UpstreamProxyCard"; +import SearchProviderCard from "./components/SearchProviderCard"; +// providerText used by UpstreamProxyCard (Phase 1t.7) export default function ProviderDetailPageClient() { const params = useParams(); - const router = useRouter(); const providerId = params.id as string; // ── UI-only modal state (not owned by hooks) ───────────────────────────── @@ -253,79 +58,15 @@ export default function ProviderDetailPageClient() { const [showAddApiKeyModal, setShowAddApiKeyModal] = useState(false); const [showSiliconFlowEndpointModal, setShowSiliconFlowEndpointModal] = useState(false); const [siliconFlowInitialBaseUrl, setSiliconFlowInitialBaseUrl] = useState(); - const [showRiskNoticeModal, setShowRiskNoticeModal] = useState(false); - const [commandCodeAuthState, setCommandCodeAuthState] = useState({ - phase: "idle", - state: "", - authUrl: "", - callbackUrl: "", - expiresAt: null, - message: "", - }); const [showEditModal, setShowEditModal] = useState(false); const [showEditNodeModal, setShowEditNodeModal] = useState(false); const [showTutorialModal, setShowTutorialModal] = useState(false); const [selectedConnection, setSelectedConnection] = useState(null); const [proxyTarget, setProxyTarget] = useState(null); - const [importingModels, setImportingModels] = useState(false); - const [importingZed, setImportingZed] = useState(false); - const [showZedManual, setShowZedManual] = useState(false); - const [zedManualProvider, setZedManualProvider] = useState("openai"); - const [zedManualToken, setZedManualToken] = useState(""); - const [importingZedManual, setImportingZedManual] = useState(false); - const [showImportModal, setShowImportModal] = useState(false); - const [importProgress, setImportProgress] = useState({ - current: 0, - total: 0, - phase: "idle" as "idle" | "fetching" | "importing" | "done" | "error", - status: "", - logs: [] as string[], - error: "", - importedCount: 0, - }); - const [compatSavingModelId, setCompatSavingModelId] = useState(null); - const [modelFilter, setModelFilter] = useState(""); - const [togglingModelId, setTogglingModelId] = useState(null); - const [testingModelId, setTestingModelId] = useState(null); - const [modelTestStatus, setModelTestStatus] = useState>({}); - const [testingAll, setTestingAll] = useState(false); - const [testProgress, setTestProgress] = useState<{ done: number; total: number } | null>(null); - const [autoHideFailed, setAutoHideFailed] = useState(true); - const [visibilityFilter, setVisibilityFilter] = useState<"all" | "visible" | "hidden">("all"); - const [bulkVisibilityAction, setBulkVisibilityAction] = useState<"select" | "deselect" | null>( - null - ); - const [applyingCodexAuthId, setApplyingCodexAuthId] = useState(null); - const [applyCodexModalConnectionId, setApplyCodexModalConnectionId] = useState( - null - ); - const [exportingCodexAuthId, setExportingCodexAuthId] = useState(null); const [importCodexModalOpen, setImportCodexModalOpen] = useState(false); const [codexCliGuideOpen, setCodexCliGuideOpen] = useState(false); - // "Adicionar Externo": public shareable device-flow link state. - const [externalLinkModalOpen, setExternalLinkModalOpen] = useState(false); - const [externalLinkUrl, setExternalLinkUrl] = useState(""); - const [externalLinkToken, setExternalLinkToken] = useState(null); - const [externalLinkLoading, setExternalLinkLoading] = useState(false); - const [externalLinkError, setExternalLinkError] = useState(null); - const { copied: externalLinkCopied, copy: externalLinkCopy } = useCopyToClipboard(); - const [applyingClaudeAuthId, setApplyingClaudeAuthId] = useState(null); - const [applyClaudeModalConnectionId, setApplyClaudeModalConnectionId] = useState( - null - ); - const [exportingClaudeAuthId, setExportingClaudeAuthId] = useState(null); const [importClaudeModalOpen, setImportClaudeModalOpen] = useState(false); - const [applyingGeminiAuthId, setApplyingGeminiAuthId] = useState(null); - const [applyGeminiModalConnectionId, setApplyGeminiModalConnectionId] = useState( - null - ); - const [exportingGeminiAuthId, setExportingGeminiAuthId] = useState(null); const [importGeminiModalOpen, setImportGeminiModalOpen] = useState(false); - const commandCodeAuthWindowRef = useRef(null); - const commandCodeAuthTimerRef = useRef(null); - const pendingRiskActionRef = useRef<(() => void) | null>(null); - const { acknowledged: riskAcknowledged, acknowledge: acknowledgeRisk } = - useRiskAcknowledged(providerId); const isOpenAICompatible = isOpenAICompatibleProvider(providerId); const isCcCompatible = isClaudeCodeCompatibleProvider(providerId); const isCommandCode = providerId === "command-code"; @@ -360,7 +101,6 @@ export default function ProviderDetailPageClient() { setSelectedIds, setBatchDeleteConfirmOpen, setBatchTestResults, - setConnections, setProviderNode, fetchConnections, fetchProxyConfig, @@ -420,6 +160,18 @@ export default function ProviderDetailPageClient() { const emailsVisible = useEmailPrivacyStore((s) => s.emailsVisible); const notify = useNotificationStore(); + // Phase 1i: external link flow — placed after notify/fetchConnections are defined + const { + externalLinkModalOpen, + setExternalLinkModalOpen, + externalLinkUrl, + externalLinkLoading, + externalLinkError, + externalLinkCopied, + externalLinkCopy, + openExternalLinkFlow, + } = useExternalLinkFlow({ providerId, notify, fetchConnections }); + const setShowOAuthModal = (show: boolean, connectionRow?: ConnectionRowConnection) => { _setShowOAuthModal(show); setReauthConnection(show && connectionRow ? connectionRow : null); @@ -436,6 +188,11 @@ export default function ProviderDetailPageClient() { const providerSupportsOAuth = providerInfo?.toggleAuthType === "oauth" || providerInfo?.toggleAuthType === "free"; const subscriptionRisk = providerInfo?.subscriptionRisk === true; + + // ── Phase 1t.3: connection gate + risk-notice modal state ─────────────── + const { showRiskNoticeModal, gateConnectionFlow, handleConfirmRiskNotice, handleCancelRiskNotice } = + useConnectionGate({ providerId, subscriptionRisk }); + const providerSupportsPat = supportsApiKeyOnFreeProvider(providerId); const isOAuth = providerSupportsOAuth && !providerSupportsPat; const isFreeNoAuth = NOAUTH_PROVIDERS[providerId]?.noAuth === true; @@ -485,60 +242,34 @@ export default function ProviderDetailPageClient() { const providerStorageAlias = isCompatible ? providerId : providerAlias; const providerDisplayAlias = isCompatible ? providerNode?.prefix || providerId : providerAlias; - const getApiLabel = () => { - if (isAnthropicProtocolCompatible) return t("messagesApi"); - const type = providerNode?.apiType; - switch (type) { - case "responses": - return t("responsesApi"); - case "embeddings": - return t("embeddings"); - case "audio-transcriptions": - return t("audioTranscriptions"); - case "audio-speech": - return t("audioSpeech"); - case "images-generations": - return t("imagesGenerations"); - default: - return t("chatCompletions"); - } - }; - - const getApiDefaultPath = () => { - if (isCcCompatible) return CC_COMPATIBLE_DEFAULT_CHAT_PATH; - if (isAnthropicCompatible) return "/messages"; - const type = providerNode?.apiType; - switch (type) { - case "responses": - return "/responses"; - case "embeddings": - return "/embeddings"; - case "audio-transcriptions": - return "/audio/transcriptions"; - case "audio-speech": - return "/audio/speech"; - case "images-generations": - return "/images/generations"; - default: - return "/chat/completions"; - } - }; - - const getApiPath = () => { - const defaultPath = getApiDefaultPath(); - return (providerNode?.chatPath || defaultPath).replace(/^\//, ""); - }; - - // fetchAliases, handleSetAlias, handleDeleteAlias → hooks/useProviderModels.ts (Phase 1f) - // fetchProviderModelMeta, fetchProxyConfig, fetchConnections → hooks/useProviderConnections.ts + useProviderModels.ts (Phase 1f) - // loadCodexSettings, loadClaudeRoutingSettings → hooks/useProviderSettings.ts (Phase 1f) - // loadConnProxies, handleRetestConnection, handleBatchTestAll, handleBatchRetest → hooks/useProviderConnections.ts (Phase 1f) - // handleDelete, handleBatchDeleteConfirm, handleBatchSetActive → hooks/useProviderConnections.ts (Phase 1f) - // handleUpdateConnectionStatus, handleToggleProxyEnabled, handleTogglePerKeyProxyEnabled → hooks/useProviderConnections.ts (Phase 1f) - // handleDistributeProxies, handleToggleRateLimit, handleToggleClaudeExtraUsage → hooks/useProviderConnections.ts (Phase 1f) - // handleToggleCliproxyapiMode, handleToggleCodexLimit, handleSwapPriority → hooks/useProviderConnections.ts (Phase 1f) - // handleToggleClaudeRoutingPreference, handleChangeCodexGlobalServiceMode → hooks/useProviderSettings.ts (Phase 1f) - // handleRefreshToken → hooks/useProviderConnections.ts (Phase 1f) + // ── Phase 1k: model import handlers ───────────────────────────────────── + const { + importingModels, + showImportModal, + importProgress, + togglingAutoSync, + canImportModels, + isAutoSyncEnabled, + setShowImportModal, + setImportProgress, + handleImportModels, + handleCompatibleImportWithProgress, + handleToggleAutoSync, + } = useModelImportHandlers({ + providerId, + models, + modelMeta, + modelAliases, + connections, + isFreeNoAuth, + handleSetAlias, + fetchAliases, + fetchProviderModelMeta, + fetchConnections, + notify, + t, + providerStorageAlias, + }); // ── model-related effects (loading gate) ──────────────────────────────── useEffect(() => { @@ -547,196 +278,6 @@ export default function ProviderDetailPageClient() { fetchAliases(); }, [loading, isSearchProvider, fetchProviderModelMeta, fetchAliases]); - const handleUpdateNode = async (formData: any) => { - try { - const res = await fetch(`/api/provider-nodes/${providerId}`, { - method: "PUT", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify(formData), - }); - const data = await res.json(); - if (res.ok) { - setProviderNode(data.node); - await fetchConnections(); - setShowEditNodeModal(false); - } - } catch (error) { - console.log("Error updating provider node:", error); - } - }; - - const handleZedImport = useCallback(async () => { - if (importingZed) return; - setImportingZed(true); - try { - const res = await fetch("/api/providers/zed/import", { method: "POST" }); - const data = await res.json(); - if (!res.ok || !data.success) { - if (data.zedDockerEnvironment) { - setShowZedManual(true); - } - notify.error(data.error || "Zed import failed"); - } else if (!data.count) { - const found = data.credentials?.length ?? 0; - if (found === 0) { - notify.info("No Zed credentials found in keychain"); - } else { - notify.info( - `Found ${found} keychain credential(s), but none matched supported providers` - ); - } - } else { - notify.success( - `Imported ${data.count} credential(s) from Zed for ${data.providers?.length ?? 0} provider(s)` - ); - await fetchConnections(); - } - } catch (e: any) { - notify.error(e?.message || "Zed import failed"); - } finally { - setImportingZed(false); - } - }, [importingZed, notify, fetchConnections]); - - const handleZedManualImport = useCallback(async () => { - if (importingZedManual || !zedManualToken.trim()) return; - setImportingZedManual(true); - try { - const res = await fetch("/api/providers/zed/manual-import", { - method: "POST", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ provider: zedManualProvider, token: zedManualToken.trim() }), - }); - const data = await res.json(); - if (!res.ok || !data.success) { - notify.error(data.error?.message ?? data.error ?? "Manual import failed"); - } else { - notify.success(`Imported ${zedManualProvider} token from Zed`); - setZedManualToken(""); - await fetchConnections(); - } - } catch (e: any) { - notify.error(e?.message || "Manual import failed"); - } finally { - setImportingZedManual(false); - } - }, [importingZedManual, zedManualProvider, zedManualToken, notify, fetchConnections]); - - // loadCodexSettings, loadClaudeRoutingSettings → hooks/useProviderSettings.ts (Phase 1f) - // loadConnProxies → hooks/useProviderConnections.ts (Phase 1f) - - const onTestModel = async (modelId: string, fullModel: string) => { - setTestingModelId(modelId); - setModelTestStatus((prev) => ({ ...prev, [modelId]: undefined })); - try { - const res = await fetch("/api/models/test", { - method: "POST", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ - providerId: selectedConnection?.provider || providerNode?.id || providerId, - modelId: fullModel, - connectionId: selectedConnection?.id, - }), - }); - const data = await res.json(); - if (res.ok && data.status === "ok") { - notify.success( - providerText( - t, - "testModelSuccess", - `Model ${modelId} is working. Latency: ${data.latencyMs}ms`, - { modelId, latencyMs: data.latencyMs } - ) - ); - setModelTestStatus((prev) => ({ ...prev, [modelId]: "ok" })); - } else { - notify.error(data.error || "Model test failed"); - setModelTestStatus((prev) => ({ ...prev, [modelId]: "error" })); - if (handleToggleModelHidden) { - await handleToggleModelHidden(providerStorageAlias, modelId, true); - } - } - } catch (err) { - notify.error("Network error testing model"); - setModelTestStatus((prev) => ({ ...prev, [modelId]: "error" })); - if (handleToggleModelHidden) { - await handleToggleModelHidden(providerStorageAlias, modelId, true); - } - } finally { - setTestingModelId(null); - } - }; - - const handleTestAll = async ( - targets: Array<{ modelId: string; fullModel: string }> - ): Promise => { - if (testingAll) return; - if (targets.length === 0) { - notify.error(providerText(t, "noModelsToTest", "No models to test")); - return; - } - setTestingAll(true); - setTestProgress({ done: 0, total: targets.length }); - - let ok = 0; - let error = 0; - let hiddenCount = 0; - - const CHUNK_SIZE = 3; - for (let i = 0; i < targets.length; i += CHUNK_SIZE) { - const chunk = targets.slice(i, i + CHUNK_SIZE); - await Promise.all( - chunk.map(async ({ modelId, fullModel }) => { - try { - const result: { - results?: Record< - string, - { - status?: "ok" | "error"; - rateLimited?: boolean; - isTimeout?: boolean; - error?: string; - } - >; - } = await fetch("/api/models/test-all", { - method: "POST", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ - providerId: providerId, - connectionId: selectedConnection?.id, - modelIds: [fullModel], - }), - }).then((r) => r.json()); - - const entry = result.results?.[fullModel]; - if (entry?.status === "ok") { - ok++; - } else { - error++; - if (autoHideFailed && !entry?.rateLimited && !entry?.isTimeout) { - await handleToggleModelHidden(providerStorageAlias, modelId, true); - hiddenCount++; - } - } - } catch (e) { - error++; - } - setTestProgress((prev) => (prev ? { done: prev.done + 1, total: prev.total } : null)); - }) - ); - } - - notify.info(providerText(t, "testAllResults", "{ok} ok, {error} error", { ok, error })); - if (hiddenCount > 0) { - notify.info(providerText(t, "testAllFailedHidden", "{count} hidden", { count: hiddenCount })); - } - setTestingAll(false); - setTestProgress(null); - }; - - // handleToggleSelectOne/All, handleBatchDeleteOpenModal/Confirm, handleDelete, - // handleBatchSetActive → hooks/useProviderConnections.ts (Phase 1f) - const handleOAuthSuccess = useCallback(() => { fetchConnections(); setShowOAuthModal(false); @@ -758,1047 +299,72 @@ export default function ProviderDetailPageClient() { openApiKeyAddFlow(); }, [isOAuth, openApiKeyAddFlow]); - // "Adicionar Externo": generate a single-use public link so a third party can - // complete the Codex device flow in their own browser. - const openExternalLinkFlow = useCallback(async () => { - setExternalLinkModalOpen(true); - setExternalLinkUrl(""); - setExternalLinkToken(null); - setExternalLinkError(null); - setExternalLinkLoading(true); - try { - const res = await fetch(`/api/oauth/${providerId}/public-link`, { - method: "POST", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify({}), - }); - const data = await res.json().catch(() => ({})); - if (res.ok && data?.url) { - setExternalLinkUrl(data.url); - setExternalLinkToken(data.token || null); - } else { - setExternalLinkError(data?.error || "Falha ao gerar o link."); - } - } catch { - setExternalLinkError("Não foi possível contatar o servidor."); - } finally { - setExternalLinkLoading(false); - } - }, [providerId]); - - // While the share popup is open, poll the ticket status so the dashboard can - // notify + refresh the connections the moment the external visitor finishes. - useEffect(() => { - if (!externalLinkModalOpen || !externalLinkToken) return; - let active = true; - const interval = setInterval(async () => { - if (!active) return; - try { - const res = await fetch( - `/api/oauth/${providerId}/public-link-status?token=${encodeURIComponent(externalLinkToken)}` - ); - const data = await res.json().catch(() => ({})); - if (!active) return; - if (data?.status === "completed") { - active = false; - clearInterval(interval); - notify.success("Conta Codex conectada pelo link externo."); - fetchConnections(); - setExternalLinkModalOpen(false); - setExternalLinkToken(null); - } else if (data?.status === "expired") { - active = false; - clearInterval(interval); - setExternalLinkError("O link expirou sem ser concluído."); - } - } catch { - /* transient network error — keep polling */ - } - }, 3000); - return () => { - active = false; - clearInterval(interval); - }; - }, [externalLinkModalOpen, externalLinkToken, providerId, notify, fetchConnections]); - - const gateConnectionFlow = useCallback( - (callback: () => void) => { - if (subscriptionRisk && !riskAcknowledged && !isRiskAcknowledged(providerId)) { - pendingRiskActionRef.current = callback; - setShowRiskNoticeModal(true); - return; - } - callback(); - }, - [providerId, riskAcknowledged, subscriptionRisk] - ); - - const handleConfirmRiskNotice = useCallback(() => { - acknowledgeRisk(); - setShowRiskNoticeModal(false); - const pendingAction = pendingRiskActionRef.current; - pendingRiskActionRef.current = null; - pendingAction?.(); - }, [acknowledgeRisk]); - - const handleCancelRiskNotice = useCallback(() => { - pendingRiskActionRef.current = null; - setShowRiskNoticeModal(false); - }, []); - - const clearCommandCodeAuthTimer = useCallback(() => { - if (commandCodeAuthTimerRef.current !== null) { - window.clearTimeout(commandCodeAuthTimerRef.current); - commandCodeAuthTimerRef.current = null; - } - }, []); - - useEffect(() => { - return () => { - clearCommandCodeAuthTimer(); - commandCodeAuthWindowRef.current?.close?.(); - }; - }, [clearCommandCodeAuthTimer]); - - const handleCloseAddApiKeyModal = useCallback(() => { - clearCommandCodeAuthTimer(); - setSiliconFlowInitialBaseUrl(undefined); - commandCodeAuthWindowRef.current?.close?.(); - commandCodeAuthWindowRef.current = null; - setCommandCodeAuthState({ - phase: "idle", - state: "", - authUrl: "", - callbackUrl: "", - expiresAt: null, - message: "", - }); - setShowAddApiKeyModal(false); - }, [clearCommandCodeAuthTimer]); - - const handleCommandCodeAuthApply = useCallback( - async (state: string, connectionId?: string, name?: string, setDefault?: boolean) => { - setCommandCodeAuthState((current) => ({ - ...current, - phase: "applying", - message: "Applying browser-approved key…", - })); - - try { - const res = await fetch("/api/providers/command-code/auth/apply", { - method: "POST", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ state, connectionId, name, setDefault }), - }); - const data = await res.json().catch(() => ({})); - - if (!res.ok) { - const errorMessage = data.error || "Failed to apply Command Code auth"; - setCommandCodeAuthState((current) => ({ - ...current, - phase: "error", - message: errorMessage, - })); - notify.error(errorMessage); - return false; - } - - setCommandCodeAuthState((current) => ({ - ...current, - phase: "applied", - message: "Command Code connected", - })); - commandCodeAuthWindowRef.current?.close?.(); - commandCodeAuthWindowRef.current = null; - await fetchConnections(); - handleCloseAddApiKeyModal(); - notify.success("Command Code connection added"); - return true; - } catch (error) { - console.error("Error applying Command Code auth:", error); - setCommandCodeAuthState((current) => ({ - ...current, - phase: "error", - message: "Failed to apply Command Code auth", - })); - notify.error("Failed to apply Command Code auth"); - return false; - } - }, - [fetchConnections, handleCloseAddApiKeyModal, notify] - ); - - const handleStartCommandCodeAuth = useCallback(async () => { - if (commandCodeAuthState.phase === "starting" || commandCodeAuthState.phase === "polling") { - return; - } - - clearCommandCodeAuthTimer(); - commandCodeAuthWindowRef.current?.close?.(); - - const popup = window.open("about:blank", "_blank"); - setCommandCodeAuthState({ - phase: "starting", - state: "", - authUrl: "", - callbackUrl: "", - expiresAt: null, - message: "Opening Command Code Studio…", - }); - - try { - const res = await fetch("/api/providers/command-code/auth/start", { - method: "POST", - headers: { "Content-Type": "application/json" }, - }); - const data = await res.json().catch(() => ({})); - - if (!res.ok || !data.state || !data.authUrl) { - const errorMessage = data.error || "Failed to start Command Code auth"; - setCommandCodeAuthState((current) => ({ - ...current, - phase: "error", - message: errorMessage, - })); - notify.error(errorMessage); - popup?.close?.(); - return; - } - - setCommandCodeAuthState({ - phase: "polling", - state: data.state, - authUrl: data.authUrl, - callbackUrl: data.callbackUrl || "", - expiresAt: data.expiresAt || null, - message: "Open the auth URL, approve access, then paste the returned key/JSON/URL below…", - }); - - if (popup) { - try { - popup.opener = null; - } catch { - // Ignore opener cleanup failures. - } - popup.location.href = data.authUrl; - commandCodeAuthWindowRef.current = popup; - } else { - const fallbackPopup = window.open(data.authUrl, "_blank", "noopener,noreferrer"); - if (!fallbackPopup) { - setCommandCodeAuthState((current) => ({ - ...current, - phase: "error", - message: "Popup blocked. Please allow popups and try Command Code Connect again.", - })); - notify.error("Popup blocked. Please allow popups and try Command Code Connect again."); - return; - } - commandCodeAuthWindowRef.current = fallbackPopup; - } - - const deadline = data.expiresAt ? new Date(data.expiresAt).getTime() : Date.now() + 180000; - const poll = async () => { - if (Date.now() >= deadline) { - setCommandCodeAuthState((current) => ({ - ...current, - phase: "expired", - message: "Command Code link expired", - })); - commandCodeAuthWindowRef.current?.close?.(); - commandCodeAuthWindowRef.current = null; - notify.error("Command Code auth expired"); - clearCommandCodeAuthTimer(); - return; - } - - try { - const statusRes = await fetch( - `/api/providers/command-code/auth/status?state=${encodeURIComponent(data.state)}`, - { method: "GET", cache: "no-store" } - ); - const statusData = await statusRes.json().catch(() => ({})); - const status = String(statusData.status || statusData.state || statusData.phase || "") - .toLowerCase() - .trim(); - - if (status === "expired") { - setCommandCodeAuthState((current) => ({ - ...current, - phase: "expired", - message: "Command Code link expired", - })); - commandCodeAuthWindowRef.current?.close?.(); - commandCodeAuthWindowRef.current = null; - notify.error("Command Code auth expired"); - clearCommandCodeAuthTimer(); - return; - } - - if (status === "applied") { - setCommandCodeAuthState((current) => ({ - ...current, - phase: "applied", - message: "Command Code connected", - })); - commandCodeAuthWindowRef.current?.close?.(); - commandCodeAuthWindowRef.current = null; - await fetchConnections(); - handleCloseAddApiKeyModal(); - notify.success("Command Code connection added"); - clearCommandCodeAuthTimer(); - return; - } - - if (status === "received") { - setCommandCodeAuthState((current) => ({ - ...current, - phase: "received", - message: "Browser approved, applying…", - })); - clearCommandCodeAuthTimer(); - await handleCommandCodeAuthApply( - data.state, - statusData.connectionId, - statusData.name, - statusData.setDefault - ); - return; - } - } catch { - // Keep polling until the contract reports a terminal state or timeout. - } - - commandCodeAuthTimerRef.current = window.setTimeout(poll, 2000); - }; - - commandCodeAuthTimerRef.current = window.setTimeout(poll, 1000); - } catch (error) { - console.error("Error starting Command Code auth:", error); - setCommandCodeAuthState((current) => ({ - ...current, - phase: "error", - message: "Failed to start Command Code auth", - })); - notify.error("Failed to start Command Code auth"); - popup?.close?.(); - commandCodeAuthWindowRef.current = null; - clearCommandCodeAuthTimer(); - } - }, [ - clearCommandCodeAuthTimer, + // ── Phase 1h: commandCode auth flow ───────────────────────────────────── + const { + commandCodeAuthState, handleCloseAddApiKeyModal, - commandCodeAuthState.phase, + handleStartCommandCodeAuth, + handleOpenCommandCodeConnect, + } = useCommandCodeAuth({ + providerId, fetchConnections, - handleCommandCodeAuthApply, + setSiliconFlowInitialBaseUrl, + setShowAddApiKeyModal, notify, - ]); - - const handleOpenCommandCodeConnect = useCallback(() => { - setShowAddApiKeyModal(true); - void handleStartCommandCodeAuth(); - }, [handleStartCommandCodeAuth]); - - const handleSaveApiKey = async (formData) => { - try { - const res = await fetch("/api/providers", { - method: "POST", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ provider: providerId, ...formData }), - }); - if (res.ok) { - const connectionData = await res.json(); - const newConnection = connectionData?.connection; - await fetchConnections(); - setShowAddApiKeyModal(false); - setSiliconFlowInitialBaseUrl(undefined); - - // Universal: sync models from the provider endpoint on every new connection - // (was previously Gemini-only). Do NOT re-introduce a providerId guard here. - if (newConnection?.id) { - setShowImportModal(true); - setImportProgress({ - current: 0, - total: 0, - phase: "fetching", - status: t("fetchingModels"), - logs: [], - error: "", - importedCount: 0, - }); - - try { - const syncRes = await fetch(`/api/providers/${newConnection.id}/sync-models`, { - method: "POST", - signal: AbortSignal.timeout(30_000), // 30s timeout — model sync shouldn't hang - }); - const syncData = await syncRes.json(); - - if (!syncRes.ok || syncData.error) { - setImportProgress((prev) => ({ - ...prev, - phase: "error", - status: t("failedFetchModels"), - error: syncData.error?.message || syncData.error || t("failedImportModels"), - })); - return null; - } - - const syncedCount = syncData.syncedModels || 0; - const availableCount = - typeof syncData.availableModelsCount === "number" - ? syncData.availableModelsCount - : Array.isArray(syncData.models) - ? syncData.models.length - : syncedCount; - const syncedModelList: Array<{ id: string; name?: string }> = syncData.models || []; - const logs: string[] = []; - if (syncedModelList.length > 0) { - logs.push(`✓ ${availableCount} models available`); - logs.push(""); - for (const m of syncedModelList) { - logs.push(` ${m.name || m.id}`); - } - } - - setImportProgress((prev) => ({ - ...prev, - phase: "done", - status: t("modelsImported", { count: availableCount }), - total: availableCount, - current: availableCount, - importedCount: availableCount, - logs, - })); - - await fetchProviderModelMeta(); - } catch (syncError) { - setImportProgress((prev) => ({ - ...prev, - phase: "error", - status: t("failedFetchModels"), - error: String(syncError), - })); - } - } - return null; - } - const data = await res.json().catch(() => ({})); - const errorMsg = data.error?.message || data.error || t("failedSaveConnection"); - return errorMsg; - } catch (error) { - console.log("Error saving connection:", error); - return t("failedSaveConnectionRetry"); - } - }; - - const handleUpdateConnection = async (formData) => { - try { - const res = await fetch(`/api/providers/${selectedConnection.id}`, { - method: "PUT", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify(formData), - }); - if (res.ok) { - await fetchConnections(); - setShowEditModal(false); - return null; - } - const data = await res.json().catch(() => ({})); - return data.error?.message || data.error || t("failedSaveConnection"); - } catch (error) { - console.log("Error updating connection:", error); - return t("failedSaveConnectionRetry"); - } - }; - - // handleUpdateConnectionStatus, handleToggleProxyEnabled, handleTogglePerKeyProxyEnabled, - // handleDistributeProxies, handleToggleRateLimit, handleToggleClaudeExtraUsage, - // handleToggleCliproxyapiMode, handleToggleCodexLimit, handleToggleClaudeRoutingPreference, - // handleChangeCodexGlobalServiceMode, handleRetestConnection, runBatchTest, - // handleBatchTestAll, handleBatchRetest, parseApiErrorMessage, getAttachmentFilename, - // handleRefreshToken → hooks/useProviderConnections.ts + useProviderSettings.ts (Phase 1f) - - // handleToggleProxyEnabled → useProviderConnections (Phase 1f) - - // handleTogglePerKeyProxyEnabled → useProviderConnections (Phase 1f) - - // handleDistributeProxies → useProviderConnections (Phase 1f) - - // handleToggleRateLimit → useProviderConnections (Phase 1f) - - // handleToggleClaudeExtraUsage → useProviderConnections (Phase 1f) - - // [cpaProviderEnabled] state + useEffect + handleToggleCliproxyapiMode → useProviderConnections (Phase 1f) - - // handleToggleCodexLimit → useProviderConnections (Phase 1f) - - // handleToggleClaudeRoutingPreference + handleChangeCodexGlobalServiceMode → useProviderSettings (Phase 1f) - - // handleRetestConnection, runBatchTest, handleBatchTestAll, handleBatchRetest, - // [refreshingId], parseApiErrorMessage, getAttachmentFilename, handleRefreshToken - // → useProviderConnections (Phase 1f) - - const handleApplyCodexAuthLocal = async (connectionId: string) => { - if (applyingCodexAuthId) return; - setApplyingCodexAuthId(connectionId); - - const defaultSuccess = - typeof t.has === "function" && t.has("codexAuthAppliedLocal") - ? t("codexAuthAppliedLocal") - : "Codex auth.json applied locally"; - const defaultError = - typeof t.has === "function" && t.has("codexAuthApplyFailed") - ? t("codexAuthApplyFailed") - : "Failed to apply Codex auth.json locally"; - - try { - const res = await fetch(`/api/providers/${connectionId}/codex-auth/apply-local`, { - method: "POST", - }); - - if (!res.ok) { - notify.error(await parseApiErrorMessage(res, defaultError)); - return; - } - - notify.success(defaultSuccess); - setApplyCodexModalConnectionId(null); - } catch (error) { - console.error("Error applying Codex auth locally:", error); - notify.error(defaultError); - } finally { - setApplyingCodexAuthId(null); - } - }; - - const handleExportCodexAuthFile = async (connectionId: string) => { - if (exportingCodexAuthId) return; - setExportingCodexAuthId(connectionId); - - const defaultSuccess = - typeof t.has === "function" && t.has("codexAuthExported") - ? t("codexAuthExported") - : "Codex auth.json exported"; - const defaultError = - typeof t.has === "function" && t.has("codexAuthExportFailed") - ? t("codexAuthExportFailed") - : "Failed to export Codex auth.json"; - - try { - const res = await fetch(`/api/providers/${connectionId}/codex-auth/export`, { - method: "POST", - }); - - if (!res.ok) { - notify.error(await parseApiErrorMessage(res, defaultError)); - return; - } - - const blob = await res.blob(); - const filename = getAttachmentFilename(res, "codex-auth.json"); - const objectUrl = window.URL.createObjectURL(blob); - const link = document.createElement("a"); - - link.href = objectUrl; - link.download = filename; - document.body.appendChild(link); - link.click(); - document.body.removeChild(link); - window.setTimeout(() => window.URL.revokeObjectURL(objectUrl), 1000); - - notify.success(defaultSuccess); - } catch (error) { - console.error("Error exporting Codex auth file:", error); - notify.error(defaultError); - } finally { - setExportingCodexAuthId(null); - } - }; - - const handleApplyClaudeAuthLocal = async (connectionId: string) => { - if (applyingClaudeAuthId) return; - setApplyingClaudeAuthId(connectionId); - - const defaultSuccess = - typeof t.has === "function" && t.has("claudeAuthAppliedLocal") - ? t("claudeAuthAppliedLocal") - : "Claude auth applied locally"; - const defaultError = - typeof t.has === "function" && t.has("claudeAuthApplyFailed") - ? t("claudeAuthApplyFailed") - : "Failed to apply Claude auth locally"; - - try { - const res = await fetch(`/api/providers/${connectionId}/claude-auth/apply-local`, { - method: "POST", - }); - - if (!res.ok) { - notify.error(await parseApiErrorMessage(res, defaultError)); - return; - } - - notify.success(defaultSuccess); - setApplyClaudeModalConnectionId(null); - } catch (error) { - console.error("Error applying Claude auth locally:", error); - notify.error(defaultError); - } finally { - setApplyingClaudeAuthId(null); - } - }; - - const handleExportClaudeAuthFile = async (connectionId: string) => { - if (exportingClaudeAuthId) return; - setExportingClaudeAuthId(connectionId); - - const defaultSuccess = - typeof t.has === "function" && t.has("claudeAuthExported") - ? t("claudeAuthExported") - : "Claude auth file exported"; - const defaultError = - typeof t.has === "function" && t.has("claudeAuthExportFailed") - ? t("claudeAuthExportFailed") - : "Failed to export Claude auth file"; - - try { - const res = await fetch(`/api/providers/${connectionId}/claude-auth/export`, { - method: "POST", - }); - - if (!res.ok) { - notify.error(await parseApiErrorMessage(res, defaultError)); - return; - } - - const blob = await res.blob(); - const filename = getAttachmentFilename(res, "claude-auth.json"); - const objectUrl = window.URL.createObjectURL(blob); - const link = document.createElement("a"); - - link.href = objectUrl; - link.download = filename; - document.body.appendChild(link); - link.click(); - document.body.removeChild(link); - window.setTimeout(() => window.URL.revokeObjectURL(objectUrl), 1000); - - notify.success(defaultSuccess); - } catch (error) { - console.error("Error exporting Claude auth file:", error); - notify.error(defaultError); - } finally { - setExportingClaudeAuthId(null); - } - }; - - const handleApplyGeminiAuthLocal = async (connectionId: string) => { - if (applyingGeminiAuthId) return; - setApplyingGeminiAuthId(connectionId); - - const defaultSuccess = - typeof t.has === "function" && t.has("geminiAuthAppliedLocal") - ? t("geminiAuthAppliedLocal") - : "Gemini auth applied locally"; - const defaultError = - typeof t.has === "function" && t.has("geminiAuthApplyFailed") - ? t("geminiAuthApplyFailed") - : "Failed to apply Gemini auth locally"; - - try { - const res = await fetch(`/api/providers/${connectionId}/gemini-cli-auth/apply-local`, { - method: "POST", - }); - - if (!res.ok) { - notify.error(await parseApiErrorMessage(res, defaultError)); - return; - } - - notify.success(defaultSuccess); - setApplyGeminiModalConnectionId(null); - } catch (error) { - console.error("Error applying Gemini auth locally:", error); - notify.error(defaultError); - } finally { - setApplyingGeminiAuthId(null); - } - }; - - const handleExportGeminiAuthFile = async (connectionId: string) => { - if (exportingGeminiAuthId) return; - setExportingGeminiAuthId(connectionId); - - const defaultSuccess = - typeof t.has === "function" && t.has("geminiAuthExported") - ? t("geminiAuthExported") - : "Gemini auth file exported"; - const defaultError = - typeof t.has === "function" && t.has("geminiAuthExportFailed") - ? t("geminiAuthExportFailed") - : "Failed to export Gemini auth file"; - - try { - const res = await fetch(`/api/providers/${connectionId}/gemini-cli-auth/export`, { - method: "POST", - }); - - if (!res.ok) { - notify.error(await parseApiErrorMessage(res, defaultError)); - return; - } - - const blob = await res.blob(); - const filename = getAttachmentFilename(res, "gemini-auth.json"); - const objectUrl = window.URL.createObjectURL(blob); - const link = document.createElement("a"); - - link.href = objectUrl; - link.download = filename; - document.body.appendChild(link); - link.click(); - document.body.removeChild(link); - window.setTimeout(() => window.URL.revokeObjectURL(objectUrl), 1000); - - notify.success(defaultSuccess); - } catch (error) { - console.error("Error exporting Gemini auth file:", error); - notify.error(defaultError); - } finally { - setExportingGeminiAuthId(null); - } - }; - - // handleSwapPriority → useProviderConnections (Phase 1f) - - const handleImportModels = async () => { - if (importingModels) return; - const activeConnection = connections.find((conn) => conn.isActive !== false); - // #3047 — no-auth providers (e.g. OpenCode Free) have no connection rows; - // fall back to the provider id so the models route can serve the public - // catalog instead of the button silently doing nothing. - if (!activeConnection && !isFreeNoAuth) return; - const importTargetId = activeConnection?.id ?? providerId; - - setImportingModels(true); - setShowImportModal(true); - setImportProgress({ - current: 0, - total: 0, - phase: "fetching", - status: t("fetchingModels"), - logs: [], - error: "", - importedCount: 0, - }); - - try { - const res = await fetch(`/api/providers/${importTargetId}/models?refresh=true`); - const data = await res.json(); - if (!res.ok) { - setImportProgress((prev) => ({ - ...prev, - phase: "error", - status: t("failedFetchModels"), - error: data.error || t("failedImportModels"), - })); - return; - } - const fetchedModels = data.models || []; - if (fetchedModels.length === 0) { - setImportProgress((prev) => ({ - ...prev, - phase: "done", - status: t("noModelsFound"), - logs: [t("noModelsReturnedFromEndpoint")], - })); - return; - } - - const existingIds = new Set([ - ...(modelMeta.customModels || []).map((m: any) => m.id), - ...models.map((m: any) => m.id), - ]); - const newModels = fetchedModels.filter( - (model: any) => !existingIds.has(model.id || model.name || model.model) - ); - - if (newModels.length === 0) { - setImportProgress((prev) => ({ - ...prev, - phase: "done", - status: t("allModelsAlreadyImported") || "All models already imported", - logs: [t("noNewModelsToImport") || "No new models to import"], - importedCount: 0, - total: 0, - current: 0, - })); - return; - } - - setImportProgress((prev) => ({ - ...prev, - phase: "importing", - total: newModels.length, - current: 0, - status: t("importingModelsProgress", { current: 0, total: newModels.length }), - logs: [ - t("foundModelsStartingImport", { count: newModels.length }), - ...(newModels.length < fetchedModels.length - ? [ - t("skippingExistingModels", { count: fetchedModels.length - newModels.length }) || - `Skipping ${fetchedModels.length - newModels.length} existing models`, - ] - : []), - ], - })); - - let importedCount = 0; - for (let i = 0; i < newModels.length; i++) { - const model = newModels[i]; - const modelId = model.id || model.name || model.model; - if (!modelId) continue; - const parts = modelId.split("/"); - const baseAlias = parts[parts.length - 1]; - - setImportProgress((prev) => ({ - ...prev, - current: i + 1, - status: t("importingModelsProgress", { current: i + 1, total: newModels.length }), - logs: [...prev.logs, t("importingModelById", { modelId })], - })); - - // Save as imported (default) model in the DB - await fetch("/api/provider-models", { - method: "POST", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ - provider: providerId, - modelId, - modelName: model.name || modelId, - source: "imported", - ...(typeof model.apiFormat === "string" ? { apiFormat: model.apiFormat } : {}), - ...(Array.isArray(model.supportedEndpoints) - ? { supportedEndpoints: model.supportedEndpoints } - : {}), - }), - }); - // Also create an alias for routing - if (!modelAliases[baseAlias]) { - await handleSetAlias(modelId, baseAlias, providerStorageAlias); - } - importedCount += 1; - } - - await fetchAliases(); - - setImportProgress((prev) => ({ - ...prev, - phase: "done", - current: newModels.length, - status: - importedCount > 0 - ? t("importSuccessCount", { count: importedCount }) - : t("noNewModelsAddedExisting"), - logs: [ - ...prev.logs, - importedCount > 0 - ? t("importDoneCount", { count: importedCount }) - : t("noNewModelsAdded"), - ], - importedCount, - })); - - // Auto-reload after success - if (importedCount > 0) { - setTimeout(() => { - window.location.reload(); - }, 2000); - } - } catch (error) { - console.log("Error importing models:", error); - setImportProgress((prev) => ({ - ...prev, - phase: "error", - status: t("importFailed"), - error: error instanceof Error ? error.message : t("unexpectedErrorOccurred"), - })); - } finally { - setImportingModels(false); - } - }; - - // Shared import handler for CompatibleModelsSection - const handleCompatibleImportWithProgress = async (connectionId: string) => { - setShowImportModal(true); - setImportProgress({ - current: 0, - total: 0, - phase: "fetching", - status: t("fetchingModels"), - logs: [], - error: "", - importedCount: 0, - }); - - try { - const response = await fetch(`/api/providers/${connectionId}/sync-models?mode=import`, { - method: "POST", - signal: AbortSignal.timeout(60_000), - }); - const data = await response.json(); - if (!response.ok) { - throw new Error(data.error || t("failedImportModels")); - } - - const importedModels = Array.isArray(data.importedModels) ? data.importedModels : []; - const importedCount = - typeof data.importedCount === "number" ? data.importedCount : importedModels.length; - const changedCount = - typeof data.importedChanges?.total === "number" - ? data.importedChanges.total - : importedCount; - const totalChangedCount = - changedCount + - (typeof data.customModelChanges?.total === "number" ? data.customModelChanges.total : 0); - - if (importedModels.length === 0) { - setImportProgress((prev) => ({ - ...prev, - phase: "done", - status: - importedCount > 0 - ? t("importSuccessCount", { count: importedCount }) - : t("noNewModelsAdded"), - logs: [ - importedCount > 0 - ? t("importDoneCount", { count: importedCount }) - : t("noNewModelsAdded"), - ], - importedCount, - })); - if (totalChangedCount > 0) { - setTimeout(() => { - window.location.reload(); - }, 2000); - } - return; - } - - setImportProgress((prev) => ({ - ...prev, - phase: "done", - total: importedModels.length, - current: importedModels.length, - status: - importedCount > 0 - ? t("importSuccessCount", { count: importedCount }) - : t("noNewModelsAdded"), - logs: [ - t("foundModelsStartingImport", { count: importedModels.length }), - ...importedModels.map((model: any) => - t("importingModelById", { modelId: model.id || model.name || model.model }) - ), - importedCount > 0 - ? t("importDoneCount", { count: importedCount }) - : t("noNewModelsAdded"), - ], - importedCount, - })); - - if (totalChangedCount > 0) { - setTimeout(() => { - window.location.reload(); - }, 2000); - } - } catch (error) { - console.log("Error importing models:", error); - setImportProgress((prev) => ({ - ...prev, - phase: "error", - status: t("importFailed"), - error: error instanceof Error ? error.message : t("unexpectedErrorOccurred"), - })); - } - }; - - const canImportModels = isFreeNoAuth || connections.some((conn) => conn.isActive !== false); - - // Auto-sync toggle state: read from first active connection's providerSpecificData - const autoSyncConnection = connections.find((conn: any) => conn.isActive !== false); - const isAutoSyncEnabled = !!(autoSyncConnection as any)?.providerSpecificData?.autoSync; - const [togglingAutoSync, setTogglingAutoSync] = useState(false); - - const handleToggleAutoSync = async () => { - if (!autoSyncConnection || togglingAutoSync) return; - setTogglingAutoSync(true); - try { - const newValue = !isAutoSyncEnabled; - await fetch(`/api/providers/${(autoSyncConnection as any).id}`, { - method: "PUT", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ - providerSpecificData: { autoSync: newValue }, - }), - }); - await fetchConnections(); - notify[newValue ? "success" : "info"]( - newValue ? t("autoSyncEnabled") : t("autoSyncDisabled") - ); - } catch (error) { - console.log("Error toggling auto-sync:", error); - notify.error(t("autoSyncToggleFailed")); - } finally { - setTogglingAutoSync(false); - } - }; + }); - const [clearingModels, setClearingModels] = useState(false); - const providerAliasEntries = useMemo( - () => - Object.entries(modelAliases).filter( - ([, model]) => typeof model === "string" && model.startsWith(`${providerStorageAlias}/`) - ), - [modelAliases, providerStorageAlias] - ); + // Phase 1s: handleSaveApiKey extracted to hooks/useApiKeySave.ts + const { handleSaveApiKey } = useApiKeySave({ + providerId, + fetchConnections, + fetchProviderModelMeta, + setImportProgress, + setShowImportModal, + setShowAddApiKeyModal, + setSiliconFlowInitialBaseUrl, + notify, + t, + }); - const handleClearAllModels = async () => { - if (clearingModels) return; - if (!confirm(t("clearAllModelsConfirm"))) return; - setClearingModels(true); - try { - const res = await fetch( - `/api/provider-models?provider=${encodeURIComponent(providerStorageAlias)}&all=true`, - { method: "DELETE" } - ); - if (res.ok) { - // Also delete all aliases that belong to this provider - await Promise.all( - providerAliasEntries.map(([alias]) => - fetch(`/api/models/alias?alias=${encodeURIComponent(alias)}`, { - method: "DELETE", - }).catch(() => {}) - ) - ); - await fetchProviderModelMeta(); - await fetchAliases(); - notify.success(t("clearAllModelsSuccess")); - } else { - notify.error(t("clearAllModelsFailed")); - } - } catch { - notify.error(t("clearAllModelsFailed")); - } finally { - setClearingModels(false); - } - }; + // ── Phase 1t.4: node/connection update handlers ────────────────────────── + const { handleUpdateNode, handleUpdateConnection } = useProviderNodeActions({ + providerId, + fetchConnections, + selectedConnection, + setProviderNode, + setShowEditNodeModal, + setShowEditModal, + t, + }); - // Phase 1e: compat-state derivations moved to useModelCompatState hook. + // Phase 1j: auth file handlers + const { + applyingCodexAuthId, + applyCodexModalConnectionId, + setApplyCodexModalConnectionId, + exportingCodexAuthId, + handleApplyCodexAuthLocal, + handleExportCodexAuthFile, + applyingClaudeAuthId, + applyClaudeModalConnectionId, + setApplyClaudeModalConnectionId, + exportingClaudeAuthId, + handleApplyClaudeAuthLocal, + handleExportClaudeAuthFile, + applyingGeminiAuthId, + applyGeminiModalConnectionId, + setApplyGeminiModalConnectionId, + exportingGeminiAuthId, + handleApplyGeminiAuthLocal, + handleExportGeminiAuthFile, + } = useAuthFileHandlers({ parseApiErrorMessage, getAttachmentFilename, notify, t }); + + // Phase 1e: compat-state derivations const compat = useModelCompatState( modelMeta.customModels, modelMeta.modelCompatOverrides ); - const { customMap, overrideMap } = compat; + const { customMap } = compat; const effectiveModelNormalize = compat.effectiveModelNormalize; const effectiveModelPreserveDeveloper = compat.effectiveModelPreserveDeveloper; const effectiveModelHidden = compat.isModelHidden; @@ -1809,429 +375,45 @@ export default function ProviderDetailPageClient() { [providerId, modelMeta.customModels] ); - const saveModelCompatFlags = async (modelId: string, patch: ModelCompatSavePatch) => { - setCompatSavingModelId(modelId); - try { - const c = customMap.get(modelId) as Record | undefined; - let body: Record; - const onlyCompatByProtocol = - patch.compatByProtocol && - patch.normalizeToolCallId === undefined && - patch.preserveOpenAIDeveloperRole === undefined && - !("upstreamHeaders" in patch); - - if (c) { - if (onlyCompatByProtocol) { - body = { - provider: providerId, - modelId, - compatByProtocol: patch.compatByProtocol, - }; - } else { - body = { - provider: providerId, - modelId, - modelName: (c.name as string) || modelId, - source: (c.source as string) || "manual", - apiFormat: (c.apiFormat as string) || "chat-completions", - supportedEndpoints: - Array.isArray(c.supportedEndpoints) && (c.supportedEndpoints as unknown[]).length - ? c.supportedEndpoints - : ["chat"], - normalizeToolCallId: - patch.normalizeToolCallId !== undefined - ? patch.normalizeToolCallId - : Boolean(c.normalizeToolCallId), - preserveOpenAIDeveloperRole: - patch.preserveOpenAIDeveloperRole !== undefined - ? patch.preserveOpenAIDeveloperRole - : Object.prototype.hasOwnProperty.call(c, "preserveOpenAIDeveloperRole") - ? Boolean(c.preserveOpenAIDeveloperRole) - : true, - }; - if (patch.compatByProtocol) body.compatByProtocol = patch.compatByProtocol; - } - } else { - body = { provider: providerId, modelId, ...patch }; - } - const res = await fetch("/api/provider-models", { - method: "PUT", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify(body), - }); - if (!res.ok) { - const detail = await formatProviderModelsErrorResponse(res); - notify.error( - detail ? `${t("failedSaveCustomModel")} — ${detail}` : t("failedSaveCustomModel") - ); - return; - } - } catch { - notify.error(t("failedSaveCustomModel")); - return; - } finally { - setCompatSavingModelId(null); - } - try { - await fetchProviderModelMeta(); - } catch { - /* refresh failure is non-critical — data was already saved */ - } - }; - - const handleToggleModelHidden = async ( - providerKey: string, - modelId: string, - hidden: boolean - ): Promise => { - setTogglingModelId(modelId); - try { - const res = await fetch( - `/api/provider-models?provider=${encodeURIComponent(providerKey)}&modelId=${encodeURIComponent(modelId)}`, - { - method: "PATCH", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ isHidden: hidden }), - } - ); - if (!res.ok) { - const detail = await res.text().catch(() => ""); - notify.error(detail || t("failedSaveCustomModel")); - return; - } - await Promise.all([fetchProviderModelMeta().catch(() => {}), fetchAliases().catch(() => {})]); - } catch { - notify.error(t("failedSaveCustomModel")); - } finally { - setTogglingModelId(null); - } - }; - - const handleBulkToggleModelHidden = async ( - providerKey: string, - modelIds: string[], - hidden: boolean - ): Promise => { - if (modelIds.length === 0) return; - setBulkVisibilityAction(hidden ? "deselect" : "select"); - try { - const res = await fetch(`/api/provider-models?provider=${encodeURIComponent(providerKey)}`, { - method: "PATCH", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ isHidden: hidden, modelIds }), - }); - if (!res.ok) { - const detail = await res.text().catch(() => ""); - notify.error(detail || t("failedSaveCustomModel")); - return; - } - await Promise.all([fetchProviderModelMeta().catch(() => {}), fetchAliases().catch(() => {})]); - } catch { - notify.error(t("failedSaveCustomModel")); - } finally { - setBulkVisibilityAction(null); - } - }; - - const renderModelsSection = () => { - const autoSyncToggle = compatibleSupportsModelImport && canImportModels && ( - - ); - - const clearAllButton = (modelMeta.customModels.length > 0 || - providerAliasEntries.length > 0) && ( - - ); - - if (isManagedAvailableModelsProvider) { - const description = - providerId === "openrouter" - ? t("openRouterAnyModelHint") - : isCcCompatible - ? t("ccCompatibleModelsDescription") - : t("compatibleModelsDescription", { - type: isAnthropicCompatible ? t("anthropic") : t("openai"), - }); - const inputLabel = providerId === "openrouter" ? t("modelIdFromOpenRouter") : t("modelId"); - const inputPlaceholder = - providerId === "openrouter" - ? t("openRouterModelPlaceholder") - : isCcCompatible - ? "claude-sonnet-4-6" - : isAnthropicCompatible - ? t("anthropicCompatibleModelPlaceholder") - : t("openaiCompatibleModelPlaceholder"); - - return ( -
-
- {autoSyncToggle} - {clearAllButton} -
- - handleToggleModelHidden(providerStorageAlias, modelId, hidden) - } - onBulkToggleHidden={(modelIds, hidden) => - handleBulkToggleModelHidden(providerStorageAlias, modelIds, hidden) - } - bulkTogglePending={bulkVisibilityAction !== null} - togglingModelId={togglingModelId} - onTestModel={onTestModel} - modelTestStatus={modelTestStatus} - testingModelId={testingModelId} - onTestAll={handleTestAll} - testingAll={testingAll} - testProgress={testProgress} - autoHideFailed={autoHideFailed} - onAutoHideFailedChange={setAutoHideFailed} - /> -
- ); - } - - if (providerInfo.passthroughModels) { - const passthroughDescription = - providerId === "openrouter" - ? t("openRouterAnyModelHint") - : providerId === "bedrock" - ? t("bedrockModelsDescription") - : t("passthroughModelsDescription", { provider: providerInfo?.name || providerId }); - const passthroughInputLabel = - providerId === "openrouter" ? t("modelIdFromOpenRouter") : t("modelId"); - const passthroughInputPlaceholder = - providerId === "openrouter" - ? t("openRouterModelPlaceholder") - : providerId === "bedrock" - ? t("bedrockModelPlaceholder") - : t("openaiCompatibleModelPlaceholder"); - - return ( -
-
- - {autoSyncToggle} - {clearAllButton} - {!canImportModels && ( - {t("addConnectionToImport")} - )} -
- - handleToggleModelHidden(providerStorageAlias, modelId, hidden) - } - onBulkToggleHidden={(modelIds, hidden) => - handleBulkToggleModelHidden(providerStorageAlias, modelIds, hidden) - } - bulkTogglePending={bulkVisibilityAction !== null} - togglingModelId={togglingModelId} - onTestModel={onTestModel} - modelTestStatus={modelTestStatus} - testingModelId={testingModelId} - providerId={providerId} - connectionId={selectedConnection?.id ?? ""} - autoHideFailed={autoHideFailed} - onAutoHideFailedChange={setAutoHideFailed} - /> -
- ); - } + // ── Phase 1l: model visibility handlers ───────────────────────────────── + const { + compatSavingModelId, + togglingModelId, + bulkVisibilityAction, + clearingModels, + modelFilter, + testingModelId, + modelTestStatus, + testingAll, + testProgress, + autoHideFailed, + visibilityFilter, + providerAliasEntries, + setModelFilter, + setAutoHideFailed, + setVisibilityFilter, + saveModelCompatFlags, + handleToggleModelHidden, + handleBulkToggleModelHidden, + handleClearAllModels, + onTestModel, + handleTestAll, + onModelTestStatusChange, + } = useModelVisibilityHandlers({ + providerId, + modelAliases, + customMap, + providerStorageAlias, + fetchProviderModelMeta, + fetchAliases, + notify, + t, + selectedConnection, + providerNode, + }); - const importButton = ( -
- - {autoSyncToggle} - {!canImportModels && ( - {t("addConnectionToImport")} - )} -
- ); + // renderModelsSection → components/ProviderModelsSection.tsx (Phase 1m) - if (models.length === 0) { - return ( -
- {importButton} -

{t("noModelsConfigured")}

-
- ); - } - const modelsWithVisibility = models.map((model) => ({ - ...model, - isHidden: effectiveModelHidden(model.id), - })); - const filteredModels = modelsWithVisibility.filter((model) => { - const matchesQuery = matchesModelCatalogQuery(modelFilter, { - modelId: model.id, - modelName: model.name, - source: model.source, - }); - const matchesVisibility = - visibilityFilter === "all" - ? true - : visibilityFilter === "visible" - ? !model.isHidden - : model.isHidden; - return matchesQuery && matchesVisibility; - }); - const activeCount = modelsWithVisibility.filter((m) => !m.isHidden).length; - const hiddenFilteredCount = filteredModels.filter((m) => m.isHidden).length; - const visibleFilteredCount = filteredModels.length - hiddenFilteredCount; - const testAllTargets = filteredModels - .filter((m) => !m.isHidden) - .map((m) => ({ modelId: m.id, fullModel: `${providerDisplayAlias}/${m.id}` })); - return ( -
- {importButton} - {modelsWithVisibility.length > 0 && ( - - handleBulkToggleModelHidden( - providerId, - filteredModels.map((model) => model.id), - false - ) - } - onDeselectAll={() => - handleBulkToggleModelHidden( - providerId, - filteredModels.map((model) => model.id), - true - ) - } - selectAllDisabled={hiddenFilteredCount === 0 || bulkVisibilityAction !== null} - deselectAllDisabled={visibleFilteredCount === 0 || bulkVisibilityAction !== null} - onTestAll={() => handleTestAll(testAllTargets)} - testingAll={testingAll} - testProgress={testProgress} - visibilityFilter={visibilityFilter} - onVisibilityFilterChange={setVisibilityFilter} - autoHideFailed={autoHideFailed} - onAutoHideFailedChange={setAutoHideFailed} - /> - )} -
- {filteredModels.map((model) => { - return ( - getUpstreamHeadersRecordForModel(model.id, p)} - saveModelCompatFlags={saveModelCompatFlags} - compatDisabled={compatSavingModelId === model.id} - onToggleHidden={(modelId, hidden) => - handleToggleModelHidden(providerId, modelId, hidden) - } - togglingHidden={togglingModelId === model.id} - onTestModel={onTestModel} - testStatus={modelTestStatus[model.id] || null} - testingModel={testingModelId === model.id} - /> - ); - })} - {filteredModels.length === 0 && modelFilter && ( -

- {providerText(t, "noModelsMatch", `No models match "${modelFilter}"`, { - filter: modelFilter, - })} -

- )} -
-
- ); - }; if (loading) { return ( @@ -2253,226 +435,34 @@ export default function ProviderDetailPageClient() { ); } - // OpenAI/Anthropic compatible providers use their specialized pseudo-provider icons. - const getHeaderIconProviderId = () => { - if (isOpenAICompatible && providerInfo.apiType) { - return providerInfo.apiType === "responses" ? "oai-r" : "oai-cc"; - } - if (isAnthropicProtocolCompatible) { - return "anthropic-m"; - } - return providerInfo.id; - }; - return (
- {/* Header */} -
- - arrow_back - {t("backToProviders")} - -
-
- -
-
- {providerInfo.website ? ( - - {providerInfo.name} - open_in_new - - ) : ( -

{providerInfo.name}

- )} -
-

- {t("connectionCountLabel", { count: connections.length })} -

- - {providerId === "adapta-web" && ( - - )} -
-
-
-
+ {/* Header — Phase 1t.1: extracted to components/ProviderPageHeader.tsx */} + setShowTutorialModal(true)} + t={t} + /> - {providerId === "zed" && ( - <> - -
-
-

- download - Import from Zed Keychain -

-

- Discover AI provider credentials (OpenAI, Anthropic, Google, Mistral, xAI) that - Zed IDE stored in the OS keychain and import them as connections. Requires Zed IDE - installed on this machine. -

-
- -
-
- -
- - {showZedManual && ( -
-

- Use this when OmniRoute runs in Docker or the keychain is unavailable. Paste the - API key that Zed stored under{" "} - ~/.config/zed/settings.json or copy - it from the Zed AI settings panel. -

-
- - setZedManualToken(e.target.value)} - /> - -
-
- )} -
-
- - )} + {providerId === "zed" && } + {/* CompatibleNodeCard — Phase 1t.2: extracted to components/CompatibleNodeCard.tsx */} {isCompatible && providerNode && ( - -
-
-

- {isCcCompatible - ? t("ccCompatibleDetailsTitle") - : isAnthropicCompatible - ? t("anthropicCompatibleDetails") - : t("openaiCompatibleDetails")} -

-

- {getApiLabel()} · {(providerNode.baseUrl || "").replace(/\/$/, "")}/{getApiPath()} -

-
-
- - - -
-
- {isCcCompatible && ( -
-
- - warning - -

{t("ccCompatibleValidationHint")}

-
-
- )} -
+ setShowEditNodeModal(true)} + t={t} + /> )} {/* Connections */} @@ -2496,961 +486,198 @@ export default function ProviderDetailPageClient() { providerId !== "opencode" && } {!isUpstreamProxyProvider && !isFreeNoAuth && ( -
-
-

{t("connections")}

- {providerId === "claude" && ( -
- - alt_route - - - {providerText( - t, - "preferClaudeCodeForUnprefixedClaudeModelsLabel", - "Claude Code default" - )} - - - - {preferClaudeCodeForUnprefixedClaudeModels - ? providerText(t, "toggleOnShort", "On") - : providerText(t, "toggleOffShort", "Off")} - - {claudeRoutingSettingsLoadError ? ( - - ) : null} -
- )} - {providerId === "codex" && ( -
- - {providerText(t, "providerDetailServiceModeLabel", "Global service mode:")} - - - {codexSettingsLoadError ? ( - - ) : null} -
- )} - {/* Provider-level proxy indicator/button */} - -
-
- {connections.length > 0 && ( - - )} - {connections.length > 1 && ( - - )} - {!isCompatible ? ( - <> - {isCommandCode ? ( - <> - - - - ) : ( - <> - - {providerId === "qoder" && ( - - )} - {providerId === "codex" && ( - - )} - {providerId === "codex" && ( - - )} - {providerId === "codex" && ( - - )} - {providerId === "claude" && ( - - )} - {providerId === "gemini-cli" && ( - - )} - - )} - - ) : ( - connections.length === 0 && ( - - ) - )} -
-
+ setShowOAuthModal(true)} + onOpenCodexCliGuide={() => setCodexCliGuideOpen(true)} + onOpenImportCodex={() => setImportCodexModalOpen(true)} + onOpenImportClaude={() => setImportClaudeModalOpen(true)} + onOpenImportGemini={() => setImportGeminiModalOpen(true)} + t={t} + /> {connections.length === 0 ? ( -
-
- - {isOAuth ? "lock" : "key"} - -
-

{t("noConnectionsYet")}

-

{t("addFirstConnectionHint")}

- {!isCompatible && ( -
- {isCommandCode ? ( - <> - - - - ) : ( - <> - - {providerId === "qoder" && ( - - )} - {providerId === "codex" && ( - - )} - {providerId === "claude" && ( - - )} - {providerId === "gemini-cli" && ( - - )} - - )} -
- )} -
+ setShowOAuthModal(true)} + onOpenImportCodex={() => setImportCodexModalOpen(true)} + onOpenImportClaude={() => setImportClaudeModalOpen(true)} + onOpenImportGemini={() => setImportGeminiModalOpen(true)} + t={t} + /> ) : ( - (() => { - const sorted = [...connections].sort((a, b) => (a.priority || 0) - (b.priority || 0)); - const hasAnyTag = sorted.some( - (c) => c.providerSpecificData?.tag as string | undefined - ); - const allSelected = selectedIds.size === connections.length && connections.length > 0; - const someSelected = selectedIds.size > 0 && selectedIds.size < connections.length; - const bulkBusy = - batchUpdating !== null || batchRetesting || batchDeleting || batchTesting; - const bulkActions = selectedIds.size > 0 && ( -
- - - - -
- ); - - const isHealthy = (c: ConnectionRowConnection): boolean => { - const s = c.testStatus; - return c.isActive !== false && (!s || s === "active" || s === "success"); - }; - const STATUS_FILTER_OPTIONS = [ - { value: "all", label: t("filterAll", "All") }, - { value: "active", label: t("filterActive", "Active") }, - { value: "error", label: t("filterError", "Error") }, - { value: "banned", label: t("filterBanned", "Banned") }, - { - value: "credits_exhausted", - label: t("filterCreditsExhausted", "Credits Exhausted"), - }, - ]; - const filtered = - healthFilter === "all" - ? sorted - : sorted.filter((c) => { - if (healthFilter === "active") return isHealthy(c); - if (healthFilter === "error") - return ( - !isHealthy(c) && - c.testStatus !== "banned" && - c.testStatus !== "credits_exhausted" - ); - return c.testStatus === healthFilter; - }); - - const totalFilteredPages = Math.max(1, Math.ceil(filtered.length / PAGE_SIZE)); - const clampedPage = Math.min(page, totalFilteredPages - 1); - const pageStart = clampedPage * PAGE_SIZE; - const pageEnd = pageStart + PAGE_SIZE; - - const filterPills = ( -
- {STATUS_FILTER_OPTIONS.map((opt) => ( - - ))} -
- ); - - const paginationBar = - totalFilteredPages > 1 ? ( -
- - {pageStart + 1}–{Math.min(pageEnd, filtered.length)} / {filtered.length} - -
-
-
- ) : null; - - if (!hasAnyTag) { - const pageConnections = filtered.slice(pageStart, pageEnd); - const allSelected = - pageConnections.length > 0 && pageConnections.every((c) => selectedIds.has(c.id)); - const someSelected = pageConnections.some((c) => selectedIds.has(c.id)); - return ( - <> -
-
- - {filterPills} -
- - {bulkActions} -
-
- {pageConnections.length === 0 ? ( -
- {t("noFilteredConnections", "No connections match the current filter.")} -
- ) : ( - pageConnections.map((conn, index) => ( - handleToggleSelectOne(conn.id)} - onMoveUp={() => handleSwapPriority(conn, sorted[index - 1])} - onMoveDown={() => handleSwapPriority(conn, sorted[index + 1])} - onToggleActive={(isActive) => - handleUpdateConnectionStatus(conn.id, isActive) - } - onToggleRateLimit={(enabled) => handleToggleRateLimit(conn.id, enabled)} - onToggleClaudeExtraUsage={(enabled) => - handleToggleClaudeExtraUsage(conn.id, enabled) - } - isCodex={providerId === "codex"} - isGeminiCli={providerId === "gemini-cli"} - isCcCompatible={isCcCompatible} - cliproxyapiEnabled={cpaProviderEnabled} - onToggleCliproxyapiMode={(enabled) => - handleToggleCliproxyapiMode(conn.id, enabled) - } - onToggleCodex5h={(enabled) => - handleToggleCodexLimit(conn.id, "use5h", enabled) - } - onToggleCodexWeekly={(enabled) => - handleToggleCodexLimit(conn.id, "useWeekly", enabled) - } - onRetest={() => handleRetestConnection(conn.id)} - isRetesting={retestingId === conn.id} - onEdit={() => { - setSelectedConnection(conn); - setShowEditModal(true); - }} - onDelete={() => handleDelete(conn.id)} - onReauth={ - conn.authType === "oauth" - ? () => gateConnectionFlow(() => setShowOAuthModal(true, conn)) - : undefined - } - onRefreshToken={ - conn.authType === "oauth" - ? () => handleRefreshToken(conn.id) - : undefined - } - isRefreshing={refreshingId === conn.id} - onApplyCodexAuthLocal={ - providerId === "codex" - ? () => setApplyCodexModalConnectionId(conn.id) - : undefined - } - isApplyingCodexAuthLocal={applyingCodexAuthId === conn.id} - onExportCodexAuthFile={ - providerId === "codex" - ? () => handleExportCodexAuthFile(conn.id) - : undefined - } - isExportingCodexAuthFile={exportingCodexAuthId === conn.id} - onApplyClaudeAuthLocal={ - providerId === "claude" - ? () => setApplyClaudeModalConnectionId(conn.id) - : undefined - } - isApplyingClaudeAuthLocal={applyingClaudeAuthId === conn.id} - onExportClaudeAuthFile={ - providerId === "claude" - ? () => handleExportClaudeAuthFile(conn.id) - : undefined - } - isExportingClaudeAuthFile={exportingClaudeAuthId === conn.id} - onApplyGeminiAuthLocal={ - providerId === "gemini-cli" - ? () => setApplyGeminiModalConnectionId(conn.id) - : undefined - } - isApplyingGeminiAuthLocal={applyingGeminiAuthId === conn.id} - onExportGeminiAuthFile={ - providerId === "gemini-cli" - ? () => handleExportGeminiAuthFile(conn.id) - : undefined - } - isExportingGeminiAuthFile={exportingGeminiAuthId === conn.id} - onProxy={() => - setProxyTarget({ - level: "key", - id: conn.id, - label: pickDisplayValue( - [conn.name, conn.email], - emailsVisible, - conn.id - ), - }) - } - hasProxy={!!connProxyMap[conn.id]?.proxy} - proxySource={connProxyMap[conn.id]?.level || null} - proxyHost={connProxyMap[conn.id]?.proxy?.host || null} - proxyEnabled={readBooleanToggle(conn.proxyEnabled, true)} - onToggleProxyEnabled={(enabled) => - handleToggleProxyEnabled(conn.id, enabled) - } - perKeyProxyEnabled={readBooleanToggle(conn.perKeyProxyEnabled, false)} - onTogglePerKeyProxyEnabled={(enabled) => - handleTogglePerKeyProxyEnabled(conn.id, enabled) - } - /> - )) - )} -
- {paginationBar} - - ); - } - - // Build ordered tag groups: untagged first, then alphabetically - const groupMap = new Map(); - for (const conn of filtered) { - const tag = (conn.providerSpecificData?.tag as string | undefined)?.trim() || ""; - if (!groupMap.has(tag)) groupMap.set(tag, []); - groupMap.get(tag)!.push(conn); - } - const groupKeys = Array.from(groupMap.keys()).sort((a, b) => { - if (a === "") return -1; - if (b === "") return 1; - return compareTr(a, b); - }); - - return ( - <> - {selectedIds.size > 0 || connections.length > 0 ? ( -
-
- - {filterPills} -
- -
- {/* Distribute Proxies lives in the provider toolbar (top action bar); - removed the duplicate here that rendered simultaneously when nothing - was selected. Per-tag groups keep their own scoped button. */} - {bulkActions} -
-
- ) : null} -
- {groupKeys.map((tag, gi) => { - const groupConns = groupMap.get(tag)!; - return ( -
0 - ? "border-t border-black/[0.06] dark:border-white/[0.06] mt-1 pt-1" - : "" - } - > - {tag && ( -
- - label - - - {tag} - -
- - - {groupConns.length} - -
- )} -
- {groupConns.map((conn, index) => ( - handleToggleSelectOne(conn.id)} - onMoveUp={() => - handleSwapPriority(conn, sorted[sorted.indexOf(conn) - 1]) - } - onMoveDown={() => - handleSwapPriority(conn, sorted[sorted.indexOf(conn) + 1]) - } - onToggleActive={(isActive) => - handleUpdateConnectionStatus(conn.id, isActive) - } - onToggleRateLimit={(enabled) => - handleToggleRateLimit(conn.id, enabled) - } - onToggleClaudeExtraUsage={(enabled) => - handleToggleClaudeExtraUsage(conn.id, enabled) - } - isCodex={providerId === "codex"} - isGeminiCli={providerId === "gemini-cli"} - isCcCompatible={isCcCompatible} - cliproxyapiEnabled={cpaProviderEnabled} - onToggleCodex5h={(enabled) => - handleToggleCodexLimit(conn.id, "use5h", enabled) - } - onToggleCodexWeekly={(enabled) => - handleToggleCodexLimit(conn.id, "useWeekly", enabled) - } - onRetest={() => handleRetestConnection(conn.id)} - isRetesting={retestingId === conn.id} - onEdit={() => { - setSelectedConnection(conn); - setShowEditModal(true); - }} - onDelete={() => handleDelete(conn.id)} - onReauth={ - conn.authType === "oauth" - ? () => gateConnectionFlow(() => setShowOAuthModal(true, conn)) - : undefined - } - onRefreshToken={ - conn.authType === "oauth" - ? () => handleRefreshToken(conn.id) - : undefined - } - isRefreshing={refreshingId === conn.id} - onApplyCodexAuthLocal={ - providerId === "codex" - ? () => setApplyCodexModalConnectionId(conn.id) - : undefined - } - isApplyingCodexAuthLocal={applyingCodexAuthId === conn.id} - onExportCodexAuthFile={ - providerId === "codex" - ? () => handleExportCodexAuthFile(conn.id) - : undefined - } - isExportingCodexAuthFile={exportingCodexAuthId === conn.id} - onApplyClaudeAuthLocal={ - providerId === "claude" - ? () => setApplyClaudeModalConnectionId(conn.id) - : undefined - } - isApplyingClaudeAuthLocal={applyingClaudeAuthId === conn.id} - onExportClaudeAuthFile={ - providerId === "claude" - ? () => handleExportClaudeAuthFile(conn.id) - : undefined - } - isExportingClaudeAuthFile={exportingClaudeAuthId === conn.id} - onApplyGeminiAuthLocal={ - providerId === "gemini-cli" - ? () => setApplyGeminiModalConnectionId(conn.id) - : undefined - } - isApplyingGeminiAuthLocal={applyingGeminiAuthId === conn.id} - onExportGeminiAuthFile={ - providerId === "gemini-cli" - ? () => handleExportGeminiAuthFile(conn.id) - : undefined - } - isExportingGeminiAuthFile={exportingGeminiAuthId === conn.id} - onProxy={() => - setProxyTarget({ - level: "key", - id: conn.id, - label: pickDisplayValue( - [conn.name, conn.email], - emailsVisible, - conn.id - ), - }) - } - hasProxy={!!connProxyMap[conn.id]?.proxy} - proxySource={connProxyMap[conn.id]?.level || null} - proxyHost={connProxyMap[conn.id]?.proxy?.host || null} - proxyEnabled={readBooleanToggle(conn.proxyEnabled, true)} - onToggleProxyEnabled={(enabled) => - handleToggleProxyEnabled(conn.id, enabled) - } - perKeyProxyEnabled={readBooleanToggle( - conn.perKeyProxyEnabled, - false - )} - onTogglePerKeyProxyEnabled={(enabled) => - handleTogglePerKeyProxyEnabled(conn.id, enabled) - } - /> - ))} -
-
- ); - })} -
- - ); - })() + { + setSelectedConnection(conn); + setShowEditModal(true); + }} + onOpenOAuth={(conn) => gateConnectionFlow(() => setShowOAuthModal(true, conn))} + onSetProxyTarget={setProxyTarget} + onOpenApplyCodexModal={setApplyCodexModalConnectionId} + onExportCodexAuthFile={handleExportCodexAuthFile} + onOpenApplyClaudeModal={setApplyClaudeModalConnectionId} + onExportClaudeAuthFile={handleExportClaudeAuthFile} + onOpenApplyGeminiModal={setApplyGeminiModalConnectionId} + onExportGeminiAuthFile={handleExportGeminiAuthFile} + gateConnectionFlow={gateConnectionFlow} + t={t} + /> )} )} - - {isUpstreamProxyProvider && ( - -
-
-

- {providerText( - t, - "upstreamProxyManagedTitle", - "Managed via Upstream Proxy Settings" - )} -

-

- {providerText( - t, - "upstreamProxyManagedDescription", - "CLIProxyAPI is configured as an upstream proxy layer, not as a direct provider connection. Manage the binary/runtime in CLI Tools and enable proxy routing on each provider via the provider proxy controls." - )} -

-
-
- - terminal - {t("openCliTools")} - - - settings - {t("openSettings")} - -
-
-
- )} + {isUpstreamProxyProvider && } {/* Models — hidden for search providers (they don't have models) */} {!isSearchProvider && !isUpstreamProxyProvider && (

{t("availableModels")}

- {renderModelsSection()} + {/* Phase 1m: extracted to components/ProviderModelsSection.tsx */} + {/* Custom Models — available for all providers */} -

{t("searchProvider")}

-

{t("searchProviderDesc")}

- {providerId === "perplexity-search" && ( -
- link -

{t("perplexitySearchSharedKeyInfo")}

-
- )} - {providerId === "google-pse-search" && ( -
- tune -

{t("googlePseInfo")}

-
- )} - {providerId === "searxng-search" && ( -
- dns -

{t("searxngInfo")}

-
- )} -
- )} + {isSearchProvider && } {/* Playground panel — rendered for providers that declare serviceKinds */} - {/* Modals */} - {showRiskNoticeModal && subscriptionRisk && ( - - )} - {!isUpstreamProxyProvider && - (providerId === "kiro" || providerId === "amazon-q" ? ( - { - setShowOAuthModal(false); - }} - /> - ) : providerId === "cursor" ? ( - { - setShowOAuthModal(false); - }} - /> - ) : providerId === "trae" ? ( - { - setShowOAuthModal(false); - }} - /> - ) : ( - { - setShowOAuthModal(false); - }} - /> - ))} - {providerId === "siliconflow" && ( - { - setSiliconFlowInitialBaseUrl(baseUrl); - setShowSiliconFlowEndpointModal(false); - setShowAddApiKeyModal(true); - }} - onClose={() => { - setShowSiliconFlowEndpointModal(false); - setSiliconFlowInitialBaseUrl(undefined); - }} - /> - )} - {!isUpstreamProxyProvider && ( - - )} - setBatchDeleteConfirmOpen(false)} - onConfirm={handleBatchDeleteConfirm} - title={t("batchDeleteConfirmTitle", "Delete connections")} - message={t("batchDeleteConfirm", { count: selectedIds.size })} - confirmText={t("batchDeleteConfirmButton", "Delete")} - cancelText={t("cancel", "Cancel")} - loading={batchDeleting} + {/* Modals — Phase 1t.5: extracted to components/ProviderModalsPanel.tsx */} + - {providerId === "codex" && applyCodexModalConnectionId && ( - setApplyCodexModalConnectionId(null)} - /> - )} - {!isUpstreamProxyProvider && ( - setShowEditModal(false)} - /> - )} - {!isUpstreamProxyProvider && isCompatible && ( - setShowEditNodeModal(false)} - isAnthropic={isAnthropicProtocolCompatible} - isCcCompatible={isCcCompatible} - /> - )} - {/* Codex CLI Guide Modal */} - setCodexCliGuideOpen(false)} /> - {/* Codex Import Auth Modal */} - {providerId === "codex" && importCodexModalOpen && ( - setImportCodexModalOpen(false)} - onSuccess={() => { - setImportCodexModalOpen(false); - void fetchConnections(); - }} - /> - )} - {providerId === "codex" && externalLinkModalOpen && ( - setExternalLinkModalOpen(false)} - title="Adicionar Externo — link do Codex" - > -
-

- Compartilhe este link com quem vai autenticar a conta do Codex. A pessoa abre a - página, faz o login da OpenAI no próprio navegador e a conexão é cadastrada aqui. Uso - único, expira em 15 minutos. -

- {externalLinkLoading ? ( -

Gerando link…

- ) : externalLinkError ? ( -

{externalLinkError}

- ) : externalLinkUrl ? ( - <> -
- {externalLinkUrl} -
-
- - -
-

- sync - Aguardando a autenticação no navegador da pessoa… esta janela atualiza sozinha. -

- - ) : null} -
-
- )} - {/* Claude Apply Auth Modal */} - {providerId === "claude" && applyClaudeModalConnectionId && ( - setApplyClaudeModalConnectionId(null)} - /> - )} - {/* Claude Import Auth Modal */} - {providerId === "claude" && importClaudeModalOpen && ( - setImportClaudeModalOpen(false)} - onSuccess={() => { - setImportClaudeModalOpen(false); - void fetchConnections(); - }} - /> - )} - {/* Gemini Apply Auth Modal */} - {providerId === "gemini-cli" && applyGeminiModalConnectionId && ( - setApplyGeminiModalConnectionId(null)} - /> - )} - {/* Gemini Import Auth Modal */} - {providerId === "gemini-cli" && importGeminiModalOpen && ( - setImportGeminiModalOpen(false)} - onSuccess={() => { - setImportGeminiModalOpen(false); - void fetchConnections(); - }} - /> - )} - {/* Batch Test Results Modal */} - {batchTestResults && ( -
setBatchTestResults(null)} - > -
-
e.stopPropagation()} - > -
-

{t("testResults")}

- -
-
- {batchTestResults.error && - (!batchTestResults.results || batchTestResults.results.length === 0) ? ( -
- - error - -

{String(batchTestResults.error)}

-
- ) : ( -
- {batchTestResults.summary && ( -
- {providerInfo?.name || providerId} - - {t("passedCount", { count: batchTestResults.summary.passed })} - - {batchTestResults.summary.failed > 0 && ( - - {t("failedCount", { count: batchTestResults.summary.failed })} - - )} - - {t("testedCount", { count: batchTestResults.summary.total })} - -
- )} - {(batchTestResults.results || []).map((r: any, i: number) => ( -
- - {r.valid ? "check_circle" : "error"} - -
- - {pickDisplayValue([r.connectionName], emailsVisible, r.connectionName)} - -
- {r.latencyMs !== undefined && ( - - {t("millisecondsAbbr", { value: r.latencyMs })} - - )} - - {r.valid ? t("okShort") : r.diagnosis?.type || t("errorShort")} - -
- ))} - {(!batchTestResults.results || batchTestResults.results.length === 0) && ( -
- {t("noActiveConnectionsInGroup")} -
- )} -
- )} -
-
-
- )} - {/* Proxy Config Modal */} - {proxyTarget && ( - setProxyTarget(null)} - level={proxyTarget.level} - levelId={proxyTarget.id} - levelLabel={proxyTarget.label} - onSaved={() => { - void fetchProxyConfig(); - void loadConnProxies(connections); - }} - /> - )} - {/* Import Progress Modal */} - { - if (importProgress.phase === "done" || importProgress.phase === "error") { - setShowImportModal(false); - } - }} - title={t("importingModelsTitle")} - size="md" - closeOnOverlay={false} - showCloseButton={importProgress.phase === "done" || importProgress.phase === "error"} - > -
- {/* Status text */} -
- {importProgress.phase === "fetching" && ( - - progress_activity - - )} - {importProgress.phase === "importing" && ( - - progress_activity - - )} - {importProgress.phase === "done" && ( - check_circle - )} - {importProgress.phase === "error" && ( - error - )} - {importProgress.status} -
- - {/* Progress bar */} - {(importProgress.phase === "importing" || importProgress.phase === "done") && - importProgress.total > 0 && ( -
-
- - {importProgress.current} / {importProgress.total} - - - {Math.round((importProgress.current / importProgress.total) * 100)}% - -
-
-
-
-
- )} - - {/* Fetching indeterminate bar */} - {importProgress.phase === "fetching" && ( -
-
-
- )} - - {/* Error message */} - {importProgress.phase === "error" && importProgress.error && ( -
-

{importProgress.error}

-
- )} - - {/* Log list */} - {importProgress.logs.length > 0 && ( -
-
- {importProgress.logs.map((log, i) => ( -

- {log} -

- ))} -
-
- )} - - {/* Close button */} - {importProgress.phase === "done" && ( -
- -
- )} -
- - - {/* Adapta Web — Tutorial Modal */} - {providerId === "adapta-web" && ( - setShowTutorialModal(false)} - title="Como conectar o Adapta Web" - size="md" - > -
-

- O Adapta usa autenticação via Clerk. O token{" "} - __client é um JWT - de longa duração que permite renovar sessões automaticamente. -

- -
    -
  1. - - 1 - -
    -

    Acesse o chat do Adapta

    -

    - Abra{" "} - - agent.adapta.one/agentic-chat - {" "} - e faça login com sua conta Gold ou Business. -

    -
    -
  2. - -
  3. - - 2 - -
    -

    Abra o DevTools

    -

    - Pressione{" "} - F12{" "} - ou{" "} - - Cmd+Option+I - {" "} - para abrir as Ferramentas do Desenvolvedor. -

    -
    -
  4. - -
  5. - - 3 - -
    -

    Vá em Application → Cookies

    -

    - Na aba Application (Chrome/Edge) ou Storage{" "} - (Firefox), expanda Cookies e clique em{" "} - - .clerk.agent.adapta.one - - . -

    -
    -
  6. - -
  7. - - 4 - -
    -

    - Copie o valor do cookie{" "} - __client -

    -

    - Localize o cookie chamado{" "} - __client na - lista. Clique nele e copie o conteúdo da coluna Value — começa - com eyJ…. -

    -
    -
  8. - -
  9. - - 5 - -
    -

    Cole aqui e salve

    -

    - Clique em Add Connection, cole o valor do{" "} - __client no - campo de API Key e salve. O OmniRoute renovará a sessão automaticamente. -

    -
    -
  10. -
- -
- Dica: O cookie __client tem - validade longa (meses). Só será necessário renová-lo se você sair da conta ou o Adapta - invalidar a sessão. -
-
-
- )}
); } -// ModelRow, ModelVisibilityToolbar, PassthroughModelsSection, PassthroughModelRow, -// CustomModelsSection, CompatibleModelsSection → components/ (Phase 1e — Issue #3501) - -// Phase 1d: CooldownTimer, inferErrorType, getStatusPresentation, ConnectionRow → components/ConnectionRow.tsx -// Phase 1d: ModelCompatPopover, recordToHeaderRows → components/ModelCompatPopover.tsx -// Phase 1d: SiliconFlowEndpointModal → components/SiliconFlowEndpointModal.tsx diff --git a/src/app/(dashboard)/dashboard/providers/[id]/__tests__/phase1f.test.tsx b/src/app/(dashboard)/dashboard/providers/[id]/__tests__/phase1f.test.tsx index 9ee6137d206..3ed902b55dd 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/__tests__/phase1f.test.tsx +++ b/src/app/(dashboard)/dashboard/providers/[id]/__tests__/phase1f.test.tsx @@ -10,9 +10,10 @@ // Uses createRoot + act to mount each hook inside a minimal wrapper component // so we test real React hook semantics without a full Next.js server context. -import React, { act } from "react"; +import React, { act, useEffect } from "react"; import { createRoot } from "react-dom/client"; import { describe, it, expect, vi, beforeEach, afterEach } from "vitest"; +import path from "node:path"; // --------------------------------------------------------------------------- // Global mocks required by the extracted hooks @@ -27,10 +28,7 @@ vi.mock("next/navigation", () => ({ vi.mock("next-intl", () => ({ useTranslations: () => (key: string, values?: Record) => { if (values) { - return Object.entries(values).reduce( - (acc, [k, v]) => acc.replace(`{${k}}`, String(v)), - key - ); + return Object.entries(values).reduce((acc, [k, v]) => acc.replace(`{${k}}`, String(v)), key); } return key; }, @@ -83,10 +81,13 @@ describe("useProviderConnections — initial state", () => { let result: HookResult | null = null; function TestWrapper() { - result = useProviderConnections("openai", true, false); + const hookResult = useProviderConnections("openai", true, false); + useEffect(() => { + result = hookResult; + }, [hookResult]); return ( - {String(result.connections.length)}|{String(result.batchTesting)} + {String(hookResult.connections.length)}|{String(hookResult.batchTesting)} ); } @@ -112,7 +113,10 @@ describe("useProviderConnections — initial state", () => { let result: HookResult | null = null; function TestWrapper() { - result = useProviderConnections("openai", true, false); + const hookResult = useProviderConnections("openai", true, false); + useEffect(() => { + result = hookResult; + }, [hookResult]); return ; } @@ -181,7 +185,10 @@ describe("useProviderSettings — initial state", () => { let result: HookResult | null = null; function TestWrapper() { - result = useProviderSettings("openai"); + const hookResult = useProviderSettings("openai"); + useEffect(() => { + result = hookResult; + }, [hookResult]); return ; } @@ -207,7 +214,10 @@ describe("useProviderSettings — initial state", () => { let result: HookResult | null = null; function TestWrapper() { - result = useProviderSettings("codex"); + const hookResult = useProviderSettings("codex"); + useEffect(() => { + result = hookResult; + }, [hookResult]); return ; } @@ -253,7 +263,10 @@ describe("useProviderModels — initial state", () => { let result: HookResult | null = null; function TestWrapper() { - result = useProviderModels("openai", false); + const hookResult = useProviderModels("openai", false); + useEffect(() => { + result = hookResult; + }, [hookResult]); return ; } @@ -275,7 +288,10 @@ describe("useProviderModels — initial state", () => { let result: HookResult | null = null; function TestWrapper() { - result = useProviderModels("openai", false); + const hookResult = useProviderModels("openai", false); + useEffect(() => { + result = hookResult; + }, [hookResult]); return ; } @@ -316,8 +332,13 @@ describe("useProviderModels — initial state", () => { // Cycle-safety: hooks must NOT import from ProviderDetailPageClient // --------------------------------------------------------------------------- -const HOOKS_DIR = - "/home/diegosouzapw/dev/proxys/OmniRoute/.worktrees/fix-3501-phase1f/src/app/(dashboard)/dashboard/providers/[id]/hooks"; +// Resolve the hooks dir from the repo root (vitest runs from cwd). Was a +// hardcoded absolute worktree path that broke the test outside that worktree +// (#3501 Phase 1g-1j). +const HOOKS_DIR = path.join( + process.cwd(), + "src/app/(dashboard)/dashboard/providers/[id]/hooks" +); describe("Cycle-safety — hooks do not import ProviderDetailPageClient", () => { // We allow the name in JSDoc comments; what we forbid is an actual ES import statement. diff --git a/src/app/(dashboard)/dashboard/providers/[id]/authModals.tsx b/src/app/(dashboard)/dashboard/providers/[id]/authModals.tsx new file mode 100644 index 00000000000..e990829aaab --- /dev/null +++ b/src/app/(dashboard)/dashboard/providers/[id]/authModals.tsx @@ -0,0 +1,2909 @@ +"use client"; +import { useState, useEffect, useRef } from "react"; +import { useNotificationStore } from "@/store/notificationStore"; +import Link from "next/link"; +import { useTranslations } from "next-intl"; +import { + Button, + Badge, + Input, + Modal, + Toggle, + Select +} from "@/shared/components"; +import { + providerAllowsOptionalApiKey, + supportsBulkApiKey +} from "@/shared/constants/providers"; +import { parseBulkApiKeys } from "@/shared/utils/bulkApiKeyParser"; +import { + getWebSessionCredentialRequirement +} from "./webSessionCredentials"; + +import type { + CompatByProtocolMap, + ModelCompatSavePatch, + CompatModelRow, + CompatModelMap, + ModelRowProps, + PassthroughModelRowProps, + PassthroughModelsSectionProps, + CustomModelsSectionProps, + CompatibleModelsSectionProps, + ProviderMessageTranslator, + HeaderDraftRow, + ProviderModelsApiErrorBody, + CommandCodeAuthFlowState, + CooldownTimerProps, + ConnectionRowConnection, + ConnectionRowProps, + AddApiKeyModalProps, + EditConnectionModalConnection, + EditConnectionModalProps, + EditCompatibleNodeModalNode, + EditCompatibleNodeModalProps, + ImportTopTab, + ClaudeImportTopTab, + GeminiImportTopTab, + LocalProviderMetadata +} from "./types"; +import { + providerText, + getWebSessionCredentialLabel, + getWebSessionCredentialHint, + getWebSessionCredentialCheckLabel, + getAddCredentialModalTitle, + WebSessionCredentialGuide, + getLocalProviderMetadata, + isBaseUrlConfigurableProvider, + getProviderBaseUrlDefault, + getProviderBaseUrlHint, + getProviderBaseUrlPlaceholder, + isGlmProvider, + parseRoutingTagsInput, + parseExcludedModelsInput, + extractCommandCodeCredentialInput, + normalizeAndValidateHttpBaseUrl, + previewCodexJson, + parseBulkPasteText, + extractEmailFromClaudeJson, + previewClaudeJson, + previewGeminiJson +} from "./utils"; + +function AddApiKeyModal({ + isOpen, + provider, + providerName, + initialBaseUrl, + isCompatible, + isAnthropic, + isCcCompatible, + isCommandCode, + commandCodeAuthState, + onStartCommandCodeAuth, + onSave, + onClose, +}: AddApiKeyModalProps) { + const t = useTranslations("providers"); + const usesBaseUrl = isBaseUrlConfigurableProvider(provider); + const defaultBaseUrl = getProviderBaseUrlDefault(provider); + const isVertex = provider === "vertex" || provider === "vertex-partner"; + const isBedrock = provider === "bedrock"; + const showsRegion = isVertex || isBedrock; + const defaultRegion = isBedrock ? "eu-west-2" : "us-central1"; + const isGlm = isGlmProvider(provider); + const isQoder = provider === "qoder"; + const isCloudflare = provider === "cloudflare-ai"; + const localProviderMetadata = getLocalProviderMetadata(provider); + const isLocalSelfHostedProvider = !!localProviderMetadata; + const isGooglePse = provider === "google-pse-search"; + const webSessionCredential = getWebSessionCredentialRequirement(provider); + const isNoAuthWebSessionCredential = webSessionCredential?.kind === "none"; + const isWebSessionCredential = !!webSessionCredential && webSessionCredential.kind !== "none"; + const providerDisplayName = providerName || provider || ""; + const apiKeyOptional = + providerAllowsOptionalApiKey(provider) || Boolean(isNoAuthWebSessionCredential); + const commandCodeAuthPhaseLabel = commandCodeAuthState + ? { + idle: "Ready", + starting: "Starting…", + polling: "Waiting for browser…", + received: "Browser approved", + applying: "Applying key…", + applied: "Connected", + expired: "Link expired", + error: "Connection failed", + }[commandCodeAuthState.phase] + : null; + + const [formData, setFormData] = useState({ + name: "", + apiKey: "", + priority: 1, + baseUrl: initialBaseUrl || defaultBaseUrl, + cx: "", + region: showsRegion ? defaultRegion : "", + apiRegion: "international", + validationModelId: "", + routingTags: "", + excludedModels: "", + customUserAgent: "", + accountId: "", + consoleApiKey: "", + ccCompatibleContext1m: false, + passthroughModels: false, + }); + const [validating, setValidating] = useState(false); + const [validationResult, setValidationResult] = useState(null); + const [saving, setSaving] = useState(false); + const [saveError, setSaveError] = useState(null); + const [showAdvanced, setShowAdvanced] = useState(false); + const [copiedCommandCodeField, setCopiedCommandCodeField] = useState(null); + const wasOpenRef = useRef(false); + + useEffect(() => { + const wasOpen = wasOpenRef.current; + wasOpenRef.current = isOpen; + if (!isOpen || wasOpen) return; + setFormData((current) => ({ + ...current, + baseUrl: initialBaseUrl || defaultBaseUrl, + })); + }, [defaultBaseUrl, initialBaseUrl, isOpen]); + + const bulkSupported = supportsBulkApiKey(provider); + const [mode, setMode] = useState<"single" | "bulk">("single"); + const [bulkText, setBulkText] = useState(""); + const [bulkValidateKeys, setBulkValidateKeys] = useState(false); + const [bulkResult, setBulkResult] = useState<{ + success: number; + failed: number; + total: number; + errors: Array<{ index: number; name: string; message: string }>; + } | null>(null); + const [bulkWarnings, setBulkWarnings] = useState([]); + const apiCredentialLabel = isQoder + ? t("personalAccessTokenLabel") + : webSessionCredential + ? getWebSessionCredentialLabel(t, webSessionCredential, apiKeyOptional) + : apiKeyOptional + ? `${t("apiKeyLabel")} (${t("optional").toLowerCase()})` + : t("apiKeyLabel"); + const apiCredentialPlaceholder = isVertex + ? t("vertexServiceAccountPlaceholder") + : isWebSessionCredential + ? webSessionCredential.placeholder + : isQoder + ? t("qoderPatPlaceholder") + : apiKeyOptional + ? t("optional") + : undefined; + const apiCredentialHint = isQoder + ? t("qoderPatHint") + : isWebSessionCredential + ? getWebSessionCredentialHint(t, webSessionCredential, providerDisplayName, false) + : isLocalSelfHostedProvider + ? t("localProviderApiKeyOptionalHint", { + provider: localProviderMetadata?.name || providerName || provider || "", + }) + : apiKeyOptional + ? t("apiKeyOptionalHint") + : undefined; + const credentialValidationFailedMessage = isWebSessionCredential + ? providerText( + t, + "webSessionCredentialValidationFailed", + "Session credential validation failed. Sign in again, copy a fresh credential, and try again." + ) + : t("apiKeyValidationFailed"); + + const handleValidate = async () => { + setValidating(true); + setSaveError(null); + try { + const credentialInput = isCommandCode + ? extractCommandCodeCredentialInput(formData.apiKey) + : formData.apiKey; + const res = await fetch("/api/providers/validate", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ + provider, + apiKey: credentialInput, + validationModelId: formData.validationModelId || undefined, + customUserAgent: formData.customUserAgent.trim() || undefined, + baseUrl: formData.baseUrl.trim() || undefined, + region: showsRegion ? formData.region.trim() || defaultRegion : undefined, + cx: formData.cx.trim() || undefined, + }), + }); + const data = await res.json(); + setValidationResult(data.valid ? "success" : "failed"); + } catch { + setValidationResult("failed"); + } finally { + setValidating(false); + } + }; + + const copyCommandCodeValue = async (value: string | undefined, key: string) => { + if (!value) return; + try { + await navigator.clipboard.writeText(value); + setCopiedCommandCodeField(key); + window.setTimeout(() => setCopiedCommandCodeField(null), 1500); + } catch { + setSaveError("Copy failed. Select the text and copy it manually."); + } + }; + + const handleSubmit = async () => { + const credentialInput = isCommandCode + ? extractCommandCodeCredentialInput(formData.apiKey) + : formData.apiKey; + if (!provider || (!isCompatible && !apiKeyOptional && !credentialInput)) return; + + setSaving(true); + setSaveError(null); + try { + if (isGooglePse && !formData.cx.trim()) { + setSaveError(t("searchEngineIdRequired")); + return; + } + + let validatedBaseUrl = null; + if (usesBaseUrl) { + const checked = normalizeAndValidateHttpBaseUrl(formData.baseUrl, defaultBaseUrl); + if (checked.error) { + setSaveError(checked.error); + return; + } + validatedBaseUrl = checked.value; + } + + let isValid = Boolean(isNoAuthWebSessionCredential && !credentialInput); + let validationError: string | null = null; + if (!isValid) { + try { + setValidating(true); + setValidationResult(null); + const res = await fetch("/api/providers/validate", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ + provider, + apiKey: credentialInput, + validationModelId: formData.validationModelId || undefined, + customUserAgent: formData.customUserAgent.trim() || undefined, + baseUrl: formData.baseUrl.trim() || undefined, + region: showsRegion ? formData.region.trim() || defaultRegion : undefined, + cx: formData.cx.trim() || undefined, + }), + }); + const data = await res.json(); + isValid = !!data.valid; + if (!isValid && data.error) { + validationError = data.error; + } + setValidationResult(isValid ? "success" : "failed"); + } catch { + setValidationResult("failed"); + } finally { + setValidating(false); + } + } + + if (!isValid) { + if (apiKeyOptional && !credentialInput) { + // Bypass validation block for local/optional providers when no key is provided + console.debug("Validation failed but apiKey is optional; proceeding to save."); + } else { + setSaveError(validationError || credentialValidationFailedMessage); + return; + } + } + + const providerSpecificData: Record = {}; + if (formData.customUserAgent.trim()) { + providerSpecificData.customUserAgent = formData.customUserAgent.trim(); + } + if (formData.routingTags.trim()) { + providerSpecificData.tags = parseRoutingTagsInput(formData.routingTags); + } + if (formData.excludedModels.trim()) { + providerSpecificData.excludedModels = parseExcludedModelsInput(formData.excludedModels); + } + if (formData.passthroughModels) { + providerSpecificData.passthroughModels = true; + } + if (provider === "bailian-coding-plan" && formData.consoleApiKey.trim()) { + providerSpecificData.consoleApiKey = formData.consoleApiKey.trim(); + } + if (isGooglePse && formData.cx.trim()) { + providerSpecificData.cx = formData.cx.trim(); + } + if (usesBaseUrl) { + providerSpecificData.baseUrl = validatedBaseUrl; + } else if (showsRegion) { + providerSpecificData.region = formData.region.trim() || defaultRegion; + } else if (isGlm) { + providerSpecificData.apiRegion = formData.apiRegion; + } else if (isCloudflare && formData.accountId.trim()) { + providerSpecificData.accountId = formData.accountId.trim(); + } + if (isCcCompatible && formData.ccCompatibleContext1m) { + providerSpecificData.requestDefaults = { context1m: true }; + } + + const payload = { + name: formData.name, + apiKey: credentialInput.trim() || undefined, + priority: formData.priority, + testStatus: "active", + providerSpecificData: + Object.keys(providerSpecificData).length > 0 ? providerSpecificData : undefined, + }; + + const error = await onSave(payload); + if (error) { + setSaveError(typeof error === "string" ? error : t("failedSaveConnection")); + } + } finally { + setSaving(false); + } + }; + + const handleBulkSubmit = async () => { + if (!provider) return; + const parsed = parseBulkApiKeys(bulkText); + setBulkWarnings(parsed.warnings); + if (parsed.entries.length === 0) return; + + setSaving(true); + setBulkResult(null); + setSaveError(null); + + try { + let providerSpecificData: Record | undefined; + if (usesBaseUrl) { + const checked = normalizeAndValidateHttpBaseUrl(formData.baseUrl, defaultBaseUrl); + if (checked.error) { + setSaveError(checked.error); + return; + } + providerSpecificData = { baseUrl: checked.value }; + } + + const res = await fetch("/api/providers/bulk", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ + provider, + entries: parsed.entries.map((e) => ({ name: e.name, apiKey: e.apiKey })), + priority: formData.priority || 1, + providerSpecificData, + validateKeys: bulkValidateKeys, + }), + }); + const data = await res.json(); + if (!res.ok) { + setSaveError(typeof data?.error === "string" ? data.error : t("failedSaveConnection")); + return; + } + setBulkResult({ + success: data.success || 0, + failed: data.failed || 0, + total: data.total || 0, + errors: Array.isArray(data.errors) ? data.errors : [], + }); + } catch (err) { + setSaveError(err instanceof Error ? err.message : t("failedSaveConnection")); + } finally { + setSaving(false); + } + }; + + if (!provider) return null; + + return ( + +
+ {bulkSupported && ( +
+ + +
+ )} + + {bulkSupported && mode === "bulk" && ( +
+

{t("bulkAddFormatHint")}

+